git: 9d6498310f5c - stable/15 - processes: add zombie references, each of them prevents reap

Konstantin Belousov <[email protected]>
Newsgroups gmane.os.freebsd.devel.cvs.src
Message-ID <6a8124b1.446eb.4fcfa53__580.631689966538$1786848847$gmane$org@gitrepo.freebsd.org>
The branch stable/15 has been updated by kib:

URL: https://cgit.FreeBSD.org/src/commit/?id=9d6498310f5c49668a3ed402dc307345c8f0a2ea

commit 9d6498310f5c49668a3ed402dc307345c8f0a2ea
Author:     Konstantin Belousov <[email protected]>
AuthorDate: 2026-07-15 17:30:51 +0000
Commit:     Konstantin Belousov <[email protected]>
CommitDate: 2026-08-16 02:41:36 +0000

    processes: add zombie references, each of them prevents reap
    
    (cherry picked from commit bcdb6ba94d08554eeb9bde4d2468707a612f5d90)
---
 sys/kern/kern_exit.c    | 58 ++++++++++++++++++++++++++++---------------------
 sys/kern/kern_fork.c    |  9 ++++++++
 sys/kern/sys_procdesc.c | 48 +++++++++++++++++++++++++++-------------
 sys/sys/proc.h          | 12 +++++++++-
 sys/sys/procdesc.h      |  6 +++--
 5 files changed, 90 insertions(+), 43 deletions(-)

diff --git a/sys/kern/kern_exit.c b/sys/kern/kern_exit.c
index b61fb7bcf2f1..efdd51e2ec97 100644
--- a/sys/kern/kern_exit.c
+++ b/sys/kern/kern_exit.c
@@ -683,8 +683,7 @@ exit1(struct thread *td, int rval, int signo)
 	 * exit().
 	 */
 	signal_parent = 0;
-	procdesc_exit(p);
-	if (p->p_procdesc == NULL) {
+	if (p->p_procdesc == NULL || procdesc_exit(p)) {
 		/*
 		 * Notify parent that we're gone.  If parent has the
 		 * PS_NOCLDWAIT flag set, or if the handler is set to SIG_IGN,
@@ -986,7 +985,8 @@ sys_pdwait(struct thread *td, struct pdwait_args *uap)
  * lock as part of its work.
  */
 void
-proc_reap(struct thread *td, struct proc *p, int *status, int options)
+proc_reap(struct thread *td, struct proc *p, int *status, int options,
+    int zombieref)
 {
 	struct proc *q, *t;
 
@@ -1010,6 +1010,13 @@ proc_reap(struct thread *td, struct proc *p, int *status, int options)
 		return;
 	}
 
+	p->p_zombieref &= ~zombieref;
+	if ((p->p_zombieref & PZOMBIEREF_REFMASK) != 0) {
+		sx_xunlock(&proctree_lock);
+		PROC_UNLOCK(p);
+		return;
+	}
+
 	PROC_LOCK(q);
 	sigqueue_take(p->p_ksi);
 	PROC_UNLOCK(q);
@@ -1192,9 +1199,8 @@ proc_to_reap(struct thread *td, struct proc *p, idtype_t idtype, id_t id,
 
 	switch (idtype) {
 	case P_ALL:
-		if (p->p_procdesc == NULL ||
-		   (p->p_pptr == td->td_proc &&
-		   (p->p_flag & P_TRACED) != 0)) {
+		if ((p->p_zombieref & PZOMBIEREF_PARENT) != 0 ||
+		   (p->p_pptr == td->td_proc && (p->p_flag & P_TRACED) != 0)) {
 			break;
 		}
 
@@ -1246,14 +1252,17 @@ proc_to_reap(struct thread *td, struct proc *p, idtype_t idtype, id_t id,
 		return (0);
 	}
 
-	if (p_canwait(td, p)) {
+	if (p_canwait(td, p) != 0 ||
+	    ((options & WEXITED) == 0 && p->p_state == PRS_ZOMBIE) ||
+	    /* waitpid() is disabled and waiter is not the debugger */
+	    ((p->p_zombieref & PZOMBIEREF_PARENT) == 0 &&
+	    (p->p_pptr != td->td_proc || (p->p_flag & P_TRACED) == 0))) {
 		PROC_UNLOCK(p);
 		return (0);
 	}
-
-	if ((options & WEXITED) == 0 && p->p_state == PRS_ZOMBIE) {
+	if (check_only) {
 		PROC_UNLOCK(p);
-		return (0);
+		return (1);
 	}
 
 	/*
@@ -1279,8 +1288,8 @@ proc_to_reap(struct thread *td, struct proc *p, idtype_t idtype, id_t id,
 	 */
 	wait_fill_wrusage(p, wrusage);
 
-	if (p->p_state == PRS_ZOMBIE && !check_only) {
-		proc_reap(td, p, status, options);
+	if (p->p_state == PRS_ZOMBIE) {
+		proc_reap(td, p, status, options, PZOMBIEREF_PARENT);
 		return (-1);
 	}
 	return (1);
@@ -1481,18 +1490,6 @@ loop_locked:
 			return (0);
 		}
 
-		/*
-		 * When running in capsicum(4) mode, make wait(2) ignore
-		 * processes created with pdfork(2).  This is because one can
-		 * disown them - by passing their process descriptor to another
-		 * process - which means it needs to be prevented from touching
-		 * them afterwards.
-		 */
-		if (IN_CAPABILITY_MODE(td) && p->p_procdesc != NULL) {
-			PROC_UNLOCK(p);
-			continue;
-		}
-
 		nfound++;
 		PROC_LOCK_ASSERT(p, MA_OWNED);
 
@@ -1523,7 +1520,6 @@ loop_locked:
 			if (ret != 0) {
 				KASSERT(ret != -1, ("reaped an orphan (pid %d)",
 				    (int)td->td_retval[0]));
-				PROC_UNLOCK(p);
 				nfound++;
 				break;
 			}
@@ -1678,6 +1674,18 @@ proc_reparent(struct proc *child, struct proc *parent, bool set_oppid)
 	child->p_pptr = parent;
 	if (set_oppid)
 		child->p_oppid = parent->p_pid;
+
+	/*
+	 * When reparenting the child to the real parent which expects
+	 * to be able to call waitpid(), or reaper, re-enable
+	 * waitpid(2) for it, so that the zombie can be collected.
+	 */
+	if ((child->p_flag & P_TRACED) == 0 &&
+	    ((proc_realparent(child) == parent &&
+	    (child->p_zombieref & PZOMBIEREF_NEEDPARENT) != 0)
+	    || child->p_reaper == parent) &&
+	    (child->p_zombieref & PZOMBIEREF_PARENT) == 0)
+		child->p_zombieref |= PZOMBIEREF_PARENT;
 }
 
 static void
diff --git a/sys/kern/kern_fork.c b/sys/kern/kern_fork.c
index 45052c53d2e3..3c8193542eb2 100644
--- a/sys/kern/kern_fork.c
+++ b/sys/kern/kern_fork.c
@@ -548,6 +548,15 @@ do_fork(struct thread *td, struct fork_req *fr, struct proc *p2, struct thread *
 	    P2_STKGAP_DISABLE | P2_STKGAP_DISABLE_EXEC | P2_NO_NEW_PRIVS |
 	    P2_WXORX_DISABLE | P2_WXORX_ENABLE_EXEC | P2_LOGSIGEXIT_CTL |
 	    P2_LOGSIGEXIT_ENABLE);
+	if ((fr->fr_flags & RFPROCDESC) != 0) {
+		p2->p_zombieref = PZOMBIEREF_PROCDESC;
+		if ((fr->fr_pd_flags & PD_NOWAITPID) == 0 &&
+		    (fr->fr_flags & RFNOWAIT) == 0)
+			p2->p_zombieref |= (PZOMBIEREF_PARENT |
+			    PZOMBIEREF_NEEDPARENT);
+	} else {
+		p2->p_zombieref = PZOMBIEREF_PARENT | PZOMBIEREF_NEEDPARENT;
+	}
 	p2->p_swtick = ticks;
 	if (p1->p_flag & P_PROFIL)
 		startprofclock(p2);
diff --git a/sys/kern/sys_procdesc.c b/sys/kern/sys_procdesc.c
index a605633fbca5..289b9f100897 100644
--- a/sys/kern/sys_procdesc.c
+++ b/sys/kern/sys_procdesc.c
@@ -277,8 +277,10 @@ procdesc_free(struct procdesc *pd)
  * procdesc_exit() - notify a process descriptor that its process is exiting.
  * We use the proctree_lock to ensure that process exit either happens
  * strictly before or strictly after a concurrent call to procdesc_close().
+ * Return true if the process' parent is responsible for reaping the child,
+ * false otherwise.
  */
-void
+bool
 procdesc_exit(struct proc *p)
 {
 	struct procdesc *pd;
@@ -289,7 +291,7 @@ procdesc_exit(struct proc *p)
 
 	pd = p->p_procdesc;
 	if (pd == NULL)
-		return;
+		goto out;
 
 	PROCDESC_LOCK(pd);
 	KASSERT(pd->pd_fpcount > 0, ("%s: closed procdesc %p", __func__, pd));
@@ -304,6 +306,8 @@ procdesc_exit(struct proc *p)
 
 	/* Wakeup all waiters for this procdesc' process exit. */
 	wakeup(&p->p_procdesc);
+out:
+	return ((p->p_zombieref & PZOMBIEREF_PARENT) != 0);
 }
 
 void
@@ -418,15 +422,7 @@ procdesc_close(struct file *fp, struct thread *td)
 	} else {
 		PROC_LOCK(p);
 		AUDIT_ARG_PROCESS(p);
-		if (p->p_state == PRS_ZOMBIE) {
-			/*
-			 * If the process is already dead and just awaiting
-			 * reaping, do that now.  This will release the
-			 * process's reference to the process descriptor when it
-			 * calls back into procdesc_reap().
-			 */
-			proc_reap(curthread, p, NULL, 0);
-		} else if (pd->pd_fpcount == 0) /* last procdesc */ {
+		if (pd->pd_fpcount == 0) /* last procdesc */ {
 			/*
 			 * If the process is not yet dead, we need to kill it,
 			 * but we can't wait around synchronously for it to go
@@ -438,9 +434,27 @@ procdesc_close(struct file *fp, struct thread *td)
 			p->p_procdesc = NULL;
 			pd->pd_pid = -1;
 			procdesc_free(pd);
+			if (p->p_state == PRS_ZOMBIE) {
+				proc_reap(curthread, p, NULL, 0,
+				    PZOMBIEREF_PROCDESC);
+				goto out;
+			}
 
-			/* Failed finstall() should not cause reaping. */
-			if ((fp->f_pdflags & F_PD_NOFINSTALL) == 0) {
+			/*
+			 * Not a zombie, and no more opened process
+			 * descriptors. Clear PZOMBIEREF_PROCDESC
+			 * since right now nobody would call
+			 * proc_reap(p, PZOMBIEREF_PROCDESC).  The
+			 * flag is re-added if pdopenpid() is called.
+			 */
+			p->p_zombieref &= ~PZOMBIEREF_PROCDESC;
+
+			/*
+			 * A reference for waitpid() or failed
+			 * finstall() should not cause reaping.
+			 */
+			if ((fp->f_pdflags & F_PD_NOFINSTALL) == 0 &&
+			    (p->p_zombieref & PZOMBIEREF_PARENT) == 0) {
 				/*
 				 * Next, reparent it to its reaper
 				 * (usually init(8)) so that there's
@@ -456,12 +470,13 @@ procdesc_close(struct file *fp, struct thread *td)
 					proc_add_orphan(p, p->p_reaper);
 				}
 			}
+
 			procdesc_close_tail(fp, p);
 		} else {
 			procdesc_close_tail(fp, p);
 		}
 	}
-
+out:
 	/*
 	 * Release the file descriptor's reference on the process descriptor.
 	 */
@@ -652,6 +667,7 @@ pdopenpid1(struct thread *td, pid_t pid, struct procdesc **pdf, struct file *fp)
 	}
 	pd = p->p_procdesc;
 	if (pd != NULL) {
+		MPASS((p->p_zombieref & PZOMBIEREF_PROCDESC) != 0);
 		refcount_acquire(&pd->pd_refcount);
 		PROCDESC_LOCK(pd);
 		MPASS(pd->pd_fpcount > 0);
@@ -663,6 +679,8 @@ pdopenpid1(struct thread *td, pid_t pid, struct procdesc **pdf, struct file *fp)
 		pd->pd_proc = p;
 		pd->pd_pid = p->p_pid;
 		p->p_procdesc = pd;
+		MPASS((p->p_zombieref & PZOMBIEREF_PROCDESC) == 0);
+		p->p_zombieref |= PZOMBIEREF_PROCDESC;
 	}
 	procdesc_finit(pd, fp);
 	PROC_UNLOCK(p);
@@ -718,7 +736,7 @@ sys_pdopenpid(struct thread *td, struct pdopenpid_args *args)
 	AUDIT_ARG_PID(args->pid);
 	AUDIT_ARG_FFLAGS(args->flags);
 
-	if ((args->flags & ~(PD_ALLOWED_AT_FORK)) != 0)
+	if ((args->flags & ~(PD_ALLOWED_AT_OPENPID)) != 0)
 		return (EINVAL);
 	return (kern_pdopenpid(td, args->pid, args->flags));
 }
diff --git a/sys/sys/proc.h b/sys/sys/proc.h
index 59958bd97a32..3082ffacb726 100644
--- a/sys/sys/proc.h
+++ b/sys/sys/proc.h
@@ -782,6 +782,7 @@ struct proc {
 	u_int		p_asig;		/* (c) ASYNCEXIT pending signal. */
 	u_int		p_tree_refcnt;	/* (e) proctree refcount */
 	u_int		p_execblock;	/* (c) Blockers for execve. */
+	u_int		p_zombieref;	/* (e) References for reap. */
 };
 
 #define	p_session	p_pgrp->pg_session
@@ -911,6 +912,14 @@ struct proc {
 #define	P_TREE_REAPER		0x00000004	/* Reaper of subtree */
 #define	P_TREE_GRPEXITED	0x00000008	/* exit1() done with job ctl */
 
+/*
+ * p_zombieref; protected by proctree_lock.
+ */
+#define	PZOMBIEREF_PARENT	0x00000001	/* Ref for waitpid() */
+#define	PZOMBIEREF_PROCDESC	0x00000002	/* Ref for pdwait() */
+#define	PZOMBIEREF_NEEDPARENT	0x80000000	/* Had ref for waitpid() */
+#define	PZOMBIEREF_REFMASK	(PZOMBIEREF_PARENT | PZOMBIEREF_PROCDESC)
+
 /*
  * These were process status values (p_stat), now they are only used in
  * legacy conversion code.
@@ -1205,7 +1214,8 @@ int	proc_iterate(int (*cb)(struct proc *, void *), void *cbarg);
 void	proc_linkup0(struct proc *p, struct thread *td);
 void	proc_linkup(struct proc *p, struct thread *td);
 struct proc *proc_realparent(struct proc *child);
-void	proc_reap(struct thread *td, struct proc *p, int *status, int options);
+void	proc_reap(struct thread *td, struct proc *p, int *status, int options,
+	    int zombieref);
 void	proc_reparent(struct proc *child, struct proc *newparent, bool set_oppid);
 void	proc_set_p2_wexit(struct proc *p);
 void	proc_set_traced(struct proc *p, bool stop);
diff --git a/sys/sys/procdesc.h b/sys/sys/procdesc.h
index f912f8787fd9..0c923509f6eb 100644
--- a/sys/sys/procdesc.h
+++ b/sys/sys/procdesc.h
@@ -108,7 +108,7 @@ struct procdesc {
 /*
  * In-kernel interfaces to process descriptors.
  */
-void	 procdesc_exit(struct proc *);
+bool	 procdesc_exit(struct proc *);
 void	 procdesc_fork(struct proc *p, pid_t child_pid);
 void	 procdesc_jobstate(struct proc *p);
 int	 kern_pdgetpid(struct thread *, int fd, const cap_rights_t *,
@@ -160,7 +160,9 @@ __END_DECLS
  */
 #define	PD_DAEMON	0x00000001	/* Don't exit when procdesc closes. */
 #define	PD_CLOEXEC	0x00000002	/* Close file descriptor on exec. */
+#define	PD_NOWAITPID	0x00000004	/* Reap without waitpid(). */
 
-#define	PD_ALLOWED_AT_FORK	(PD_DAEMON | PD_CLOEXEC)
+#define	PD_ALLOWED_AT_FORK	(PD_DAEMON | PD_CLOEXEC | PD_NOWAITPID)
+#define	PD_ALLOWED_AT_OPENPID	(PD_DAEMON | PD_CLOEXEC)
 
 #endif /* !_SYS_PROCDESC_H_ */
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.