Skip site navigation (1)Skip section navigation (2)
Date:      Sat, 01 Aug 2026 12:19:27 +0000
From:      Konstantin Belousov <kib@FreeBSD.org>
To:        src-committers@FreeBSD.org, dev-commits-src-all@FreeBSD.org, dev-commits-src-main@FreeBSD.org
Subject:   git: bcdb6ba94d08 - main - processes: add zombie references, each of them prevents reap
Message-ID:  <6a6de44f.1dd12.2bfa5fa6@gitrepo.freebsd.org>

index | next in thread | raw e-mail

The branch main has been updated by kib:

URL: https://cgit.FreeBSD.org/src/commit/?id=bcdb6ba94d08554eeb9bde4d2468707a612f5d90

commit bcdb6ba94d08554eeb9bde4d2468707a612f5d90
Author:     Konstantin Belousov <kib@FreeBSD.org>
AuthorDate: 2026-07-15 17:30:51 +0000
Commit:     Konstantin Belousov <kib@FreeBSD.org>
CommitDate: 2026-08-01 10:41:45 +0000

    processes: add zombie references, each of them prevents reap
    
    Add the p_zombieref bitmask into struct proc, which enumerates all
    legitimate waiters on the process exit status. Among them are parent
    for PZOMBIEREF_PARENT, and the holder of the process descriptor for
    PZOMBIEREF_PROCDESC, if the process was created by pdfork().
    
    Require all zombie refs to be cleared to reap zombie. This prevents
    stealing the exit status from the parent by pdwait()ing on a procdesc
    obtained by pdopenpid(), or by waitpid() by debugger from the real
    parent.
    
    Reviewed by:    markj
    Tested by:      pho
    Sponsored by:   The FreeBSD Foundation
    MFC after:      1 week
    Differential revision:  https://reviews.freebsd.org/D58264
---
 sys/kern/kern_exit.c    | 58 ++++++++++++++++++++++++++++---------------------
 sys/kern/kern_fork.c    |  9 ++++++++
 sys/kern/sys_procdesc.c | 48 +++++++++++++++++++++++++++-------------
 sys/sys/proc.h          | 12 +++++++++-
 sys/sys/procdesc.h      |  6 +++--
 5 files changed, 90 insertions(+), 43 deletions(-)

diff --git a/sys/kern/kern_exit.c b/sys/kern/kern_exit.c
index b61fb7bcf2f1..efdd51e2ec97 100644
--- a/sys/kern/kern_exit.c
+++ b/sys/kern/kern_exit.c
@@ -683,8 +683,7 @@ exit1(struct thread *td, int rval, int signo)
 	 * exit().
 	 */
 	signal_parent = 0;
-	procdesc_exit(p);
-	if (p->p_procdesc == NULL) {
+	if (p->p_procdesc == NULL || procdesc_exit(p)) {
 		/*
 		 * Notify parent that we're gone.  If parent has the
 		 * PS_NOCLDWAIT flag set, or if the handler is set to SIG_IGN,
@@ -986,7 +985,8 @@ sys_pdwait(struct thread *td, struct pdwait_args *uap)
  * lock as part of its work.
  */
 void
-proc_reap(struct thread *td, struct proc *p, int *status, int options)
+proc_reap(struct thread *td, struct proc *p, int *status, int options,
+    int zombieref)
 {
 	struct proc *q, *t;
 
@@ -1010,6 +1010,13 @@ proc_reap(struct thread *td, struct proc *p, int *status, int options)
 		return;
 	}
 
+	p->p_zombieref &= ~zombieref;
+	if ((p->p_zombieref & PZOMBIEREF_REFMASK) != 0) {
+		sx_xunlock(&proctree_lock);
+		PROC_UNLOCK(p);
+		return;
+	}
+
 	PROC_LOCK(q);
 	sigqueue_take(p->p_ksi);
 	PROC_UNLOCK(q);
@@ -1192,9 +1199,8 @@ proc_to_reap(struct thread *td, struct proc *p, idtype_t idtype, id_t id,
 
 	switch (idtype) {
 	case P_ALL:
-		if (p->p_procdesc == NULL ||
-		   (p->p_pptr == td->td_proc &&
-		   (p->p_flag & P_TRACED) != 0)) {
+		if ((p->p_zombieref & PZOMBIEREF_PARENT) != 0 ||
+		   (p->p_pptr == td->td_proc && (p->p_flag & P_TRACED) != 0)) {
 			break;
 		}
 
@@ -1246,14 +1252,17 @@ proc_to_reap(struct thread *td, struct proc *p, idtype_t idtype, id_t id,
 		return (0);
 	}
 
-	if (p_canwait(td, p)) {
+	if (p_canwait(td, p) != 0 ||
+	    ((options & WEXITED) == 0 && p->p_state == PRS_ZOMBIE) ||
+	    /* waitpid() is disabled and waiter is not the debugger */
+	    ((p->p_zombieref & PZOMBIEREF_PARENT) == 0 &&
+	    (p->p_pptr != td->td_proc || (p->p_flag & P_TRACED) == 0))) {
 		PROC_UNLOCK(p);
 		return (0);
 	}
-
-	if ((options & WEXITED) == 0 && p->p_state == PRS_ZOMBIE) {
+	if (check_only) {
 		PROC_UNLOCK(p);
-		return (0);
+		return (1);
 	}
 
 	/*
@@ -1279,8 +1288,8 @@ proc_to_reap(struct thread *td, struct proc *p, idtype_t idtype, id_t id,
 	 */
 	wait_fill_wrusage(p, wrusage);
 
-	if (p->p_state == PRS_ZOMBIE && !check_only) {
-		proc_reap(td, p, status, options);
+	if (p->p_state == PRS_ZOMBIE) {
+		proc_reap(td, p, status, options, PZOMBIEREF_PARENT);
 		return (-1);
 	}
 	return (1);
@@ -1481,18 +1490,6 @@ loop_locked:
 			return (0);
 		}
 
-		/*
-		 * When running in capsicum(4) mode, make wait(2) ignore
-		 * processes created with pdfork(2).  This is because one can
-		 * disown them - by passing their process descriptor to another
-		 * process - which means it needs to be prevented from touching
-		 * them afterwards.
-		 */
-		if (IN_CAPABILITY_MODE(td) && p->p_procdesc != NULL) {
-			PROC_UNLOCK(p);
-			continue;
-		}
-
 		nfound++;
 		PROC_LOCK_ASSERT(p, MA_OWNED);
 
@@ -1523,7 +1520,6 @@ loop_locked:
 			if (ret != 0) {
 				KASSERT(ret != -1, ("reaped an orphan (pid %d)",
 				    (int)td->td_retval[0]));
-				PROC_UNLOCK(p);
 				nfound++;
 				break;
 			}
@@ -1678,6 +1674,18 @@ proc_reparent(struct proc *child, struct proc *parent, bool set_oppid)
 	child->p_pptr = parent;
 	if (set_oppid)
 		child->p_oppid = parent->p_pid;
+
+	/*
+	 * When reparenting the child to the real parent which expects
+	 * to be able to call waitpid(), or reaper, re-enable
+	 * waitpid(2) for it, so that the zombie can be collected.
+	 */
+	if ((child->p_flag & P_TRACED) == 0 &&
+	    ((proc_realparent(child) == parent &&
+	    (child->p_zombieref & PZOMBIEREF_NEEDPARENT) != 0)
+	    || child->p_reaper == parent) &&
+	    (child->p_zombieref & PZOMBIEREF_PARENT) == 0)
+		child->p_zombieref |= PZOMBIEREF_PARENT;
 }
 
 static void
diff --git a/sys/kern/kern_fork.c b/sys/kern/kern_fork.c
index e3d7b39e687a..7d2535de987f 100644
--- a/sys/kern/kern_fork.c
+++ b/sys/kern/kern_fork.c
@@ -547,6 +547,15 @@ do_fork(struct thread *td, struct fork_req *fr, struct proc *p2, struct thread *
 	    P2_STKGAP_DISABLE | P2_STKGAP_DISABLE_EXEC | P2_NO_NEW_PRIVS |
 	    P2_WXORX_DISABLE | P2_WXORX_ENABLE_EXEC | P2_LOGSIGEXIT_CTL |
 	    P2_LOGSIGEXIT_ENABLE);
+	if ((fr->fr_flags & RFPROCDESC) != 0) {
+		p2->p_zombieref = PZOMBIEREF_PROCDESC;
+		if ((fr->fr_pd_flags & PD_NOWAITPID) == 0 &&
+		    (fr->fr_flags & RFNOWAIT) == 0)
+			p2->p_zombieref |= (PZOMBIEREF_PARENT |
+			    PZOMBIEREF_NEEDPARENT);
+	} else {
+		p2->p_zombieref = PZOMBIEREF_PARENT | PZOMBIEREF_NEEDPARENT;
+	}
 	p2->p_swtick = ticks;
 	if (p1->p_flag & P_PROFIL)
 		startprofclock(p2);
diff --git a/sys/kern/sys_procdesc.c b/sys/kern/sys_procdesc.c
index a605633fbca5..289b9f100897 100644
--- a/sys/kern/sys_procdesc.c
+++ b/sys/kern/sys_procdesc.c
@@ -277,8 +277,10 @@ procdesc_free(struct procdesc *pd)
  * procdesc_exit() - notify a process descriptor that its process is exiting.
  * We use the proctree_lock to ensure that process exit either happens
  * strictly before or strictly after a concurrent call to procdesc_close().
+ * Return true if the process' parent is responsible for reaping the child,
+ * false otherwise.
  */
-void
+bool
 procdesc_exit(struct proc *p)
 {
 	struct procdesc *pd;
@@ -289,7 +291,7 @@ procdesc_exit(struct proc *p)
 
 	pd = p->p_procdesc;
 	if (pd == NULL)
-		return;
+		goto out;
 
 	PROCDESC_LOCK(pd);
 	KASSERT(pd->pd_fpcount > 0, ("%s: closed procdesc %p", __func__, pd));
@@ -304,6 +306,8 @@ procdesc_exit(struct proc *p)
 
 	/* Wakeup all waiters for this procdesc' process exit. */
 	wakeup(&p->p_procdesc);
+out:
+	return ((p->p_zombieref & PZOMBIEREF_PARENT) != 0);
 }
 
 void
@@ -418,15 +422,7 @@ procdesc_close(struct file *fp, struct thread *td)
 	} else {
 		PROC_LOCK(p);
 		AUDIT_ARG_PROCESS(p);
-		if (p->p_state == PRS_ZOMBIE) {
-			/*
-			 * If the process is already dead and just awaiting
-			 * reaping, do that now.  This will release the
-			 * process's reference to the process descriptor when it
-			 * calls back into procdesc_reap().
-			 */
-			proc_reap(curthread, p, NULL, 0);
-		} else if (pd->pd_fpcount == 0) /* last procdesc */ {
+		if (pd->pd_fpcount == 0) /* last procdesc */ {
 			/*
 			 * If the process is not yet dead, we need to kill it,
 			 * but we can't wait around synchronously for it to go
@@ -438,9 +434,27 @@ procdesc_close(struct file *fp, struct thread *td)
 			p->p_procdesc = NULL;
 			pd->pd_pid = -1;
 			procdesc_free(pd);
+			if (p->p_state == PRS_ZOMBIE) {
+				proc_reap(curthread, p, NULL, 0,
+				    PZOMBIEREF_PROCDESC);
+				goto out;
+			}
 
-			/* Failed finstall() should not cause reaping. */
-			if ((fp->f_pdflags & F_PD_NOFINSTALL) == 0) {
+			/*
+			 * Not a zombie, and no more opened process
+			 * descriptors. Clear PZOMBIEREF_PROCDESC
+			 * since right now nobody would call
+			 * proc_reap(p, PZOMBIEREF_PROCDESC).  The
+			 * flag is re-added if pdopenpid() is called.
+			 */
+			p->p_zombieref &= ~PZOMBIEREF_PROCDESC;
+
+			/*
+			 * A reference for waitpid() or failed
+			 * finstall() should not cause reaping.
+			 */
+			if ((fp->f_pdflags & F_PD_NOFINSTALL) == 0 &&
+			    (p->p_zombieref & PZOMBIEREF_PARENT) == 0) {
 				/*
 				 * Next, reparent it to its reaper
 				 * (usually init(8)) so that there's
@@ -456,12 +470,13 @@ procdesc_close(struct file *fp, struct thread *td)
 					proc_add_orphan(p, p->p_reaper);
 				}
 			}
+
 			procdesc_close_tail(fp, p);
 		} else {
 			procdesc_close_tail(fp, p);
 		}
 	}
-
+out:
 	/*
 	 * Release the file descriptor's reference on the process descriptor.
 	 */
@@ -652,6 +667,7 @@ pdopenpid1(struct thread *td, pid_t pid, struct procdesc **pdf, struct file *fp)
 	}
 	pd = p->p_procdesc;
 	if (pd != NULL) {
+		MPASS((p->p_zombieref & PZOMBIEREF_PROCDESC) != 0);
 		refcount_acquire(&pd->pd_refcount);
 		PROCDESC_LOCK(pd);
 		MPASS(pd->pd_fpcount > 0);
@@ -663,6 +679,8 @@ pdopenpid1(struct thread *td, pid_t pid, struct procdesc **pdf, struct file *fp)
 		pd->pd_proc = p;
 		pd->pd_pid = p->p_pid;
 		p->p_procdesc = pd;
+		MPASS((p->p_zombieref & PZOMBIEREF_PROCDESC) == 0);
+		p->p_zombieref |= PZOMBIEREF_PROCDESC;
 	}
 	procdesc_finit(pd, fp);
 	PROC_UNLOCK(p);
@@ -718,7 +736,7 @@ sys_pdopenpid(struct thread *td, struct pdopenpid_args *args)
 	AUDIT_ARG_PID(args->pid);
 	AUDIT_ARG_FFLAGS(args->flags);
 
-	if ((args->flags & ~(PD_ALLOWED_AT_FORK)) != 0)
+	if ((args->flags & ~(PD_ALLOWED_AT_OPENPID)) != 0)
 		return (EINVAL);
 	return (kern_pdopenpid(td, args->pid, args->flags));
 }
diff --git a/sys/sys/proc.h b/sys/sys/proc.h
index cd2d4a318992..c052540cece5 100644
--- a/sys/sys/proc.h
+++ b/sys/sys/proc.h
@@ -782,6 +782,7 @@ struct proc {
 	LIST_ENTRY(proc) p_jaillist;	/* (d) Jail process linkage. */
 	u_int		p_asig;		/* (c) ASYNCEXIT pending signal. */
 	u_int		p_tree_refcnt;	/* (e) proctree refcount */
+	u_int		p_zombieref;	/* (e) References for reap. */
 };
 
 #define	p_session	p_pgrp->pg_session
@@ -911,6 +912,14 @@ struct proc {
 #define	P_TREE_REAPER		0x00000004	/* Reaper of subtree */
 #define	P_TREE_GRPEXITED	0x00000008	/* exit1() done with job ctl */
 
+/*
+ * p_zombieref; protected by proctree_lock.
+ */
+#define	PZOMBIEREF_PARENT	0x00000001	/* Ref for waitpid() */
+#define	PZOMBIEREF_PROCDESC	0x00000002	/* Ref for pdwait() */
+#define	PZOMBIEREF_NEEDPARENT	0x80000000	/* Had ref for waitpid() */
+#define	PZOMBIEREF_REFMASK	(PZOMBIEREF_PARENT | PZOMBIEREF_PROCDESC)
+
 /*
  * These were process status values (p_stat), now they are only used in
  * legacy conversion code.
@@ -1205,7 +1214,8 @@ int	proc_iterate(int (*cb)(struct proc *, void *), void *cbarg);
 void	proc_linkup0(struct proc *p, struct thread *td);
 void	proc_linkup(struct proc *p, struct thread *td);
 struct proc *proc_realparent(struct proc *child);
-void	proc_reap(struct thread *td, struct proc *p, int *status, int options);
+void	proc_reap(struct thread *td, struct proc *p, int *status, int options,
+	    int zombieref);
 void	proc_reparent(struct proc *child, struct proc *newparent, bool set_oppid);
 void	proc_set_p2_wexit(struct proc *p);
 void	proc_set_traced(struct proc *p, bool stop);
diff --git a/sys/sys/procdesc.h b/sys/sys/procdesc.h
index f912f8787fd9..0c923509f6eb 100644
--- a/sys/sys/procdesc.h
+++ b/sys/sys/procdesc.h
@@ -108,7 +108,7 @@ struct procdesc {
 /*
  * In-kernel interfaces to process descriptors.
  */
-void	 procdesc_exit(struct proc *);
+bool	 procdesc_exit(struct proc *);
 void	 procdesc_fork(struct proc *p, pid_t child_pid);
 void	 procdesc_jobstate(struct proc *p);
 int	 kern_pdgetpid(struct thread *, int fd, const cap_rights_t *,
@@ -160,7 +160,9 @@ __END_DECLS
  */
 #define	PD_DAEMON	0x00000001	/* Don't exit when procdesc closes. */
 #define	PD_CLOEXEC	0x00000002	/* Close file descriptor on exec. */
+#define	PD_NOWAITPID	0x00000004	/* Reap without waitpid(). */
 
-#define	PD_ALLOWED_AT_FORK	(PD_DAEMON | PD_CLOEXEC)
+#define	PD_ALLOWED_AT_FORK	(PD_DAEMON | PD_CLOEXEC | PD_NOWAITPID)
+#define	PD_ALLOWED_AT_OPENPID	(PD_DAEMON | PD_CLOEXEC)
 
 #endif /* !_SYS_PROCDESC_H_ */


home | help

Want to link to this message? Use this
URL: <https://mail-archive.FreeBSD.org/cgi/mid.cgi?6a6de44f.1dd12.2bfa5fa6>