git: 151a8512bdec - main - truss(1): capsicumize

From: Konstantin Belousov <kib_at_FreeBSD.org>
Date: Wed, 16 Sep 2026 15:34:14 UTC
The branch main has been updated by kib:

URL: https://cgit.FreeBSD.org/src/commit/?id=151a8512bdeced5a4e84f4d9a36c16cd8b99e977

commit 151a8512bdeced5a4e84f4d9a36c16cd8b99e977
Author:     Konstantin Belousov <kib@FreeBSD.org>
AuthorDate: 2026-07-09 08:25:47 +0000
Commit:     Konstantin Belousov <kib@FreeBSD.org>
CommitDate: 2026-09-16 15:33:40 +0000

    truss(1): capsicumize
    
    The new ptrace(2) features allow to change truss(1) to systematically
    operate on the process descriptors instead of pids.
    
    Allocate the global kqueue that tracks all noted children
    by pdopenpid()-ing them and adding to the kqueue with
    EVFILT_PROCDESC/NOTE_PDSIGCHLD. The activated knote triggers the
    pdwait() call to return the child tracing info. This replaces the
    waitid(P_ALL) call in the non-capsicumized truss(1) eventloop.
    
    Reviewed by:    markj
    Sponsored by:   The FreeBSD Foundation
    MFC after:      1 week
    Differential revision:  https://reviews.freebsd.org/D58094
---
 usr.bin/truss/extern.h   |  27 ++-
 usr.bin/truss/main.c     |  32 ++--
 usr.bin/truss/setup.c    | 419 +++++++++++++++++++++++++++++++++++------------
 usr.bin/truss/syscalls.c | 157 ++++++++++--------
 usr.bin/truss/truss.1    |  27 ++-
 usr.bin/truss/truss.h    |   9 +
 6 files changed, 479 insertions(+), 192 deletions(-)

diff --git a/usr.bin/truss/extern.h b/usr.bin/truss/extern.h
index 26a1ba06f7c1..4a9180019ac5 100644
--- a/usr.bin/truss/extern.h
+++ b/usr.bin/truss/extern.h
@@ -31,12 +31,21 @@
  * SUCH DAMAGE.
  */
 
-extern void add_syscall_filter(const char *);
-extern void list_syscall_groups(void);
-extern bool syscall_filter_match(const char *, u_int);
-extern int print_line_prefix(struct trussinfo *);
-extern void setup_and_wait(struct trussinfo *, char **);
-extern void start_tracing(struct trussinfo *, pid_t);
-extern void restore_proc(int);
-extern void decode_siginfo(FILE *, siginfo_t *);
-extern void eventloop(struct trussinfo *);
+#ifndef __TRUSS_EXTERN_H__
+#define	__TRUSS_EXTERN_H__
+
+void add_syscall_filter(const char *);
+void list_syscall_groups(void);
+bool syscall_filter_match(const char *, u_int);
+int print_line_prefix(struct trussinfo *);
+void setup_and_wait(struct trussinfo *, char **);
+void start_tracing(struct trussinfo *, pid_t);
+void restore_proc(int);
+void decode_siginfo(FILE *, siginfo_t *);
+void eventloop(struct trussinfo *);
+
+int truss_kill(struct trussinfo *info, struct procinfo *p, int sig);
+int truss_ptrace(struct trussinfo *info, int req, struct procinfo *p,
+    void *addr, int data);
+
+#endif
diff --git a/usr.bin/truss/main.c b/usr.bin/truss/main.c
index e28fb64f3265..6e55574b0a4c 100644
--- a/usr.bin/truss/main.c
+++ b/usr.bin/truss/main.c
@@ -31,13 +31,14 @@
  * SUCH DAMAGE.
  */
 
-#include <sys/cdefs.h>
 /*
  * The main module for truss.  Surprisingly simple, but, then, the other
  * files handle the bulk of the work.  And, of course, the kernel has to
  * do a lot of the work :).
  */
 
+#include <sys/capsicum.h>
+#include <sys/event.h>
 #include <sys/ptrace.h>
 
 #include <err.h>
@@ -57,8 +58,8 @@ static __dead2 void
 usage(void)
 {
 	fprintf(stderr, "%s\n%s\n%s\n",
-	    "usage: truss [-cfaedDHS] [-o file] [-s strsize] [-t expr] -p pid",
-	    "       truss [-cfaedDHS] [-o file] [-s strsize] [-t expr] "
+	    "usage: truss [-cfaedyDHS] [-o file] [-s strsize] [-t expr] -p pid",
+	    "       truss [-cfaedyDHS] [-o file] [-s strsize] [-t expr] "
 	    "command [args]",
 	    "       truss -t");
 	exit(1);
@@ -69,6 +70,7 @@ main(int ac, char **av)
 {
 	struct sigaction sa;
 	struct trussinfo *trussinfo;
+	struct procinfo *np;
 	char *fname;
 	char **command;
 	const char *errstr;
@@ -87,13 +89,15 @@ main(int ac, char **av)
 	trussinfo->strsize = 32;
 	trussinfo->curthread = NULL;
 	LIST_INIT(&trussinfo->proclist);
+	trussinfo->cap_mode = true;
+
 	/*
 	 * The leading ':' asks getopt() to report a missing option
 	 * argument as ':' rather than '?' so that a bare -t, which lists
 	 * the system call groups, can be told from a malformed option.
 	 * Diagnosing the other two cases then falls to us.
 	 */
-	while ((c = getopt(ac, av, ":p:o:facedDs:t:SH")) != -1) {
+	while ((c = getopt(ac, av, ":p:o:facedyDs:t:SH")) != -1) {
 		switch (c) {
 		case 'p':	/* specified pid */
 			pid = atoi(optarg);
@@ -132,6 +136,9 @@ main(int ac, char **av)
 		case 't':	/* Select the system calls to trace */
 			add_syscall_filter(optarg);
 			break;
+		case 'y':
+			trussinfo->cap_mode = false;
+			break;
 		case 'S':	/* Don't trace signals */
 			trussinfo->flags |= NOSIGS;
 			break;
@@ -166,6 +173,12 @@ main(int ac, char **av)
 			err(1, "cannot open %s", fname);
 	}
 
+	if (trussinfo->cap_mode) {
+		trussinfo->pdkq = kqueue();
+		if (trussinfo->pdkq == -1)
+			err(1, "kqueue");
+	}
+
 	/*
 	 * If truss starts the process itself, it will ignore some signals --
 	 * they should be passed off to the process, which may or may not
@@ -193,7 +206,8 @@ main(int ac, char **av)
 	 * At this point, if we started the process, it is stopped waiting to
 	 * be woken up, either in exit() or in execve().
 	 */
-	if (LIST_FIRST(&trussinfo->proclist)->abi == NULL) {
+	np = LIST_FIRST(&trussinfo->proclist);
+	if (np->abi == NULL) {
 		/*
 		 * If we are not able to handle this ABI, detach from the
 		 * process and exit.  If we just created a new process to
@@ -201,13 +215,11 @@ main(int ac, char **av)
 		 * it run untraced.
 		 */
 		if (pid == 0)
-			kill(LIST_FIRST(&trussinfo->proclist)->pid, SIGKILL);
-		ptrace(PT_DETACH, LIST_FIRST(&trussinfo->proclist)->pid, NULL,
-		    0);
+			truss_kill(trussinfo, np, SIGKILL);
+		truss_ptrace(trussinfo, PT_DETACH, np, NULL, 0);
 		return (1);
 	}
-	ptrace(PT_SYSCALL, LIST_FIRST(&trussinfo->proclist)->pid, (caddr_t)1,
-	    0);
+	truss_ptrace(trussinfo, PT_SYSCALL, np, (caddr_t)1, 0);
 
 	/*
 	 * At this point, it's a simple loop, waiting for the process to
diff --git a/usr.bin/truss/setup.c b/usr.bin/truss/setup.c
index 5d89855e9823..fdd56752a904 100644
--- a/usr.bin/truss/setup.c
+++ b/usr.bin/truss/setup.c
@@ -31,18 +31,22 @@
  * SUCH DAMAGE.
  */
 
-#include <sys/cdefs.h>
 /*
  * Various setup functions for truss.  Not the cleanest-written code,
  * I'm afraid.
  */
 
+#include <sys/capsicum.h>
+#include <sys/event.h>
 #include <sys/ptrace.h>
+#include <sys/procdesc.h>
+#include <sys/syscall.h>
 #include <sys/sysctl.h>
 #include <sys/time.h>
 #include <sys/wait.h>
 
 #include <assert.h>
+#include <capsicum_helpers.h>
 #include <err.h>
 #include <errno.h>
 #include <signal.h>
@@ -59,6 +63,8 @@
 #include "syscall.h"
 #include "extern.h"
 
+#define	WFLAGS	(WTRAPPED | WEXITED | WCONTINUED | WUNTRACED)
+
 struct procabi_table {
 	const char *name;
 	struct procabi *abi;
@@ -68,8 +74,9 @@ static sig_atomic_t detaching;
 
 static void	enter_syscall(struct trussinfo *, struct threadinfo *,
 		    struct ptrace_lwpinfo *);
-static void	new_proc(struct trussinfo *, pid_t, lwpid_t);
-
+static bool	new_proc(struct trussinfo *, pid_t, lwpid_t, int, bool, bool);
+static void	new_proc_register_kev(struct trussinfo *info, int pfd);
+static struct procinfo *find_proc(struct trussinfo *info, pid_t pid);
 
 static struct procabi freebsd = {
 	.type = "FreeBSD",
@@ -138,6 +145,49 @@ static struct procabi_table abis[] = {
 #endif
 };
 
+int
+truss_ptrace(struct trussinfo *info, int req, struct procinfo *p, void *addr,
+    int data)
+{
+	if (info->cap_mode)
+		return (pdptrace(req, p->pfd, -1, addr, data));
+	return (ptrace(req, p->pid, addr, data));
+}
+
+static int
+truss_ptrace_lwp(struct trussinfo *info, int req, struct procinfo *p,
+    lwpid_t lwpid, void *addr, int data)
+{
+	if (info->cap_mode)
+		return (pdptrace(req, p->pfd, lwpid, addr, data));
+	return (ptrace(req, lwpid, addr, data));
+}
+
+static int
+t_wait(struct trussinfo *info, int pid, int *status, int wflags)
+{
+	if (info->cap_mode)
+		return (pdwait(pid, status, wflags, NULL, NULL));
+	return (waitpid(pid, status, wflags));
+}
+
+static int
+truss_wait(struct trussinfo *info, struct procinfo *p, int *status,
+    int wflags)
+{
+	if (info->cap_mode)
+		return (pdwait(p->pfd, status, wflags, NULL, NULL));
+	return (waitpid(p->pid, status, wflags));
+}
+
+int
+truss_kill(struct trussinfo *info, struct procinfo *p, int sig)
+{
+	if (info->cap_mode)
+		return (pdkill(p->pfd, sig));
+	return (kill(p->pid, sig));
+}
+
 /*
  * setup_and_wait() is called to start a process.  All it really does
  * is fork(), enable tracing in the child, and then exec the given
@@ -148,21 +198,34 @@ void
 setup_and_wait(struct trussinfo *info, char *command[])
 {
 	pid_t pid;
-
-	pid = vfork();
-	if (pid == -1)
-		err(1, "fork failed");
+	int fd, res;
+
+	if (info->cap_mode) {
+		pid = pdfork(&fd, PD_DAEMON | PD_CLOEXEC | PD_PTRACE_CAP);
+		if (pid == -1)
+			err(1, "fork failed");
+	} else {
+		pid = vfork();
+		fd = -1;
+	}
 	if (pid == 0) {	/* Child */
 		ptrace(PT_TRACE_ME, 0, 0, 0);
 		execvp(command[0], command);
 		err(1, "execvp %s", command[0]);
 	}
 
+	if (info->cap_mode) {
+		if (caph_enter() == -1)
+			err(1, "cap_enter");
+		new_proc_register_kev(info, fd);
+	}
+
 	/* Only in the parent here */
-	if (waitpid(pid, NULL, 0) < 0)
+	res = t_wait(info, info->cap_mode ? fd : pid, NULL, WFLAGS);
+	if (res < 0)
 		err(1, "unexpected stop in waitpid");
 
-	new_proc(info, pid, 0);
+	new_proc(info, pid, 0, fd, false, false);
 }
 
 /*
@@ -171,20 +234,33 @@ setup_and_wait(struct trussinfo *info, char *command[])
 void
 start_tracing(struct trussinfo *info, pid_t pid)
 {
-	int ret, retry;
+	int fd, ret, retry;
+
+	if (info->cap_mode) {
+		fd = pdopenpid(pid, PD_DAEMON | PD_CLOEXEC | PD_PTRACE_CAP);
+		if (fd == -1)
+			err(1, "Cannot open the target process");
+		if (caph_enter() == -1)
+			err(1, "cap_enter");
+		new_proc_register_kev(info, fd);
+	} else {
+		fd = -1;
+	}
 
 	retry = 10;
 	do {
-		ret = ptrace(PT_ATTACH, pid, NULL, 0);
+		ret = info->cap_mode ? pdptrace(PT_ATTACH, fd, -1,
+		    NULL, 0) : ptrace(PT_ATTACH, pid, NULL, 0);
 		usleep(200);
 	} while (ret && retry-- > 0);
 	if (ret)
 		err(1, "Cannot attach to target process");
 
-	if (waitpid(pid, NULL, 0) < 0)
+	ret = t_wait(info, info->cap_mode ? fd : pid, NULL, WFLAGS);
+	if (ret < 0)
 		err(1, "Unexpected stop in waitpid");
 
-	new_proc(info, pid, 0);
+	new_proc(info, pid, 0, fd, false, false);
 }
 
 /*
@@ -201,31 +277,32 @@ restore_proc(int signo __unused)
 }
 
 static void
-detach_proc(pid_t pid)
+detach_proc(struct trussinfo *info, struct procinfo *p)
 {
-	int sig, status;
+	int error, sig, status;
 
 	/*
 	 * Stop the child so that we can detach.  Filter out possible
 	 * lingering SIGTRAP events buffered in the threads.
 	 */
-	kill(pid, SIGSTOP);
+	truss_kill(info, p, SIGSTOP);
 	for (;;) {
-		if (waitpid(pid, &status, 0) < 0)
+		error = truss_wait(info, p, &status, WFLAGS);
+		if (error < 0)
 			err(1, "Unexpected error in waitpid");
 		sig = WIFSTOPPED(status) ? WSTOPSIG(status) : 0;
 		if (sig == SIGSTOP)
 			break;
 		if (sig == SIGTRAP)
 			sig = 0;
-		if (ptrace(PT_CONTINUE, pid, (caddr_t)1, sig) < 0)
+		if (truss_ptrace(info, PT_CONTINUE, p, (caddr_t)1, sig) < 0)
 			err(1, "Can not continue for detach");
 	}
 
-	if (ptrace(PT_DETACH, pid, (caddr_t)1, 0) < 0)
+	if (truss_ptrace(info, PT_DETACH, p, (caddr_t)1, 0) < 0)
 		err(1, "Can not detach the process");
 
-	kill(pid, SIGCONT);
+	truss_kill(info, p, SIGCONT);
 }
 
 /*
@@ -233,28 +310,22 @@ detach_proc(pid_t pid)
  * a process is first monitored.
  */
 static struct procabi *
-find_abi(pid_t pid)
+find_abi(struct trussinfo *info, struct procinfo *p)
 {
-	size_t len;
-	unsigned int i;
-	int error;
-	int mib[4];
+	unsigned i;
 	char progt[32];
 
-	len = sizeof(progt);
-	mib[0] = CTL_KERN;
-	mib[1] = KERN_PROC;
-	mib[2] = KERN_PROC_SV_NAME;
-	mib[3] = pid;
-	error = sysctl(mib, 4, progt, &len, NULL, 0);
-	if (error != 0)
-		err(2, "can not get sysvec name");
+	if (truss_ptrace(info, PT_GET_ABI_NAME, p, progt,
+	    sizeof(progt)) == -1) {
+		warn("cannot get ABI for proc %ld", (long)p->pid);
+		return (NULL);
+	}
 
 	for (i = 0; i < nitems(abis); i++) {
 		if (strcmp(abis[i].name, progt) == 0)
 			return (abis[i].abi);
 	}
-	warnx("ABI %s for pid %ld is not supported", progt, (long)pid);
+	warnx("ABI %s for pid %ld is not supported", progt, (long)p->pid);
 	return (NULL);
 }
 
@@ -297,17 +368,18 @@ add_threads(struct trussinfo *info, struct procinfo *p)
 	lwpid_t *lwps;
 	int i, nlwps;
 
-	nlwps = ptrace(PT_GETNUMLWPS, p->pid, NULL, 0);
+	nlwps = truss_ptrace(info, PT_GETNUMLWPS, p, NULL, 0);
 	if (nlwps == -1)
 		err(1, "Unable to fetch number of LWPs");
 	assert(nlwps > 0);
 	lwps = calloc(nlwps, sizeof(*lwps));
-	nlwps = ptrace(PT_GETLWPLIST, p->pid, (caddr_t)lwps, nlwps);
+	nlwps = truss_ptrace(info, PT_GETLWPLIST, p, lwps, nlwps);
 	if (nlwps == -1)
 		err(1, "Unable to fetch LWP list");
 	for (i = 0; i < nlwps; i++) {
 		t = new_thread(p, lwps[i]);
-		if (ptrace(PT_LWPINFO, lwps[i], (caddr_t)&pl, sizeof(pl)) == -1)
+		if (truss_ptrace_lwp(info, PT_LWPINFO, p, lwps[i], &pl,
+		    sizeof(pl)) == -1)
 			err(1, "ptrace(PT_LWPINFO)");
 		if (pl.pl_flags & PL_FLAG_SCE) {
 			info->curthread = t;
@@ -318,27 +390,56 @@ add_threads(struct trussinfo *info, struct procinfo *p)
 }
 
 static void
-new_proc(struct trussinfo *info, pid_t pid, lwpid_t lwpid)
+new_proc_register_kev(struct trussinfo *info, int pfd)
+{
+	struct kevent ev[1];
+	int error;
+
+	if (!info->cap_mode)
+		return;
+	EV_SET(&ev[0], pfd, EVFILT_PROCDESC, EV_ADD, NOTE_EXIT |
+	    NOTE_PDSIGCHLD | NOTE_FORK, 0, 0);
+	error = kevent(info->pdkq, ev, nitems(ev), NULL, 0, NULL);
+	if (error == -1)
+		err(1, "Unable to register pfd %d for notifications", pfd);
+}
+
+static bool
+new_proc(struct trussinfo *info, pid_t pid, lwpid_t lwpid, int pfd,
+    bool allow_known, bool wait_for)
 {
 	struct procinfo *np;
 
-	/*
-	 * If this happens it means there is a bug in truss.  Unfortunately
-	 * this will kill any processes truss is attached to.
-	 */
-	LIST_FOREACH(np, &info->proclist, entries) {
-		if (np->pid == pid)
-			errx(1, "Duplicate process for pid %ld", (long)pid);
+	if (find_proc(info, pid) != NULL) {
+		if (allow_known)
+			return (false);
+
+		/*
+		 * If this happens it means there is a bug in truss.
+		 * Unfortunately this will kill any processes truss is
+		 * attached to.
+		 */
+		errx(1, "Duplicate process for pid %ld", (long)pid);
+	}
+	if (pfd == -1 && info->cap_mode) {
+		pfd = pdopenpid(pid, PD_DAEMON | PD_CLOEXEC | PD_PTRACE_CAP);
+		if (pfd == -1)
+			err(1, "pdopenpid %d", pid);
+		if (wait_for && t_wait(info, pfd, NULL, WFLAGS) < 0)
+			err(1, "waitpid on attach to %d", pid);
+		new_proc_register_kev(info, pfd);
 	}
 
-	if (info->flags & FOLLOWFORKS)
-		if (ptrace(PT_FOLLOW_FORK, pid, NULL, 1) == -1)
-			err(1, "Unable to follow forks for pid %ld", (long)pid);
-	if (ptrace(PT_LWP_EVENTS, pid, NULL, 1) == -1)
-		err(1, "Unable to enable LWP events for pid %ld", (long)pid);
 	np = calloc(1, sizeof(struct procinfo));
 	np->pid = pid;
-	np->abi = find_abi(pid);
+	np->pfd = pfd;
+	np->abi = find_abi(info, np);
+	np->herald_printed = false;
+	if ((info->flags & FOLLOWFORKS) != 0 && truss_ptrace(info,
+	    PT_FOLLOW_FORK, np, NULL, 1) == -1)
+		err(1, "Unable to follow forks for pid %ld", (long)pid);
+	if (truss_ptrace(info, PT_LWP_EVENTS, np, NULL, 1) == -1)
+		err(1, "Unable to enable LWP events for pid %ld", (long)pid);
 	LIST_INIT(&np->threadlist);
 	LIST_INIT(&np->fdlist);
 	LIST_INSERT_HEAD(&info->proclist, np, entries);
@@ -347,13 +448,21 @@ new_proc(struct trussinfo *info, pid_t pid, lwpid_t lwpid)
 		new_thread(np, lwpid);
 	else
 		add_threads(info, np);
+	return (true);
 }
 
 static void
-free_proc(struct procinfo *p)
+free_proc(struct trussinfo *info, struct procinfo *p)
 {
 	struct threadinfo *t, *t2;
 	struct fd_domain *f, *f2;
+	struct kevent ev[1];
+
+	if (info->cap_mode) {
+		EV_SET(&ev[0], p->pfd, EVFILT_PROCDESC, EV_DELETE, 0, 0, 0);
+		(void)kevent(info->pdkq, ev, nitems(ev), NULL, 0, 0);
+		close(p->pfd);
+	}
 
 	LIST_FOREACH_SAFE(t, &p->threadlist, entries, t2) {
 		free(t);
@@ -373,8 +482,8 @@ detach_all_procs(struct trussinfo *info)
 	struct procinfo *p, *p2;
 
 	LIST_FOREACH_SAFE(p, &info->proclist, entries, p2) {
-		detach_proc(p->pid);
-		free_proc(p);
+		detach_proc(info, p);
+		free_proc(info, p);
 	}
 }
 
@@ -465,8 +574,8 @@ enter_syscall(struct trussinfo *info, struct threadinfo *t,
 
 	alloc_syscall(t, pl);
 	narg = MIN(pl->pl_syscall_narg, nitems(t->cs.args));
-	if (narg != 0 && ptrace(PT_GET_SC_ARGS, t->tid, (caddr_t)t->cs.args,
-	    sizeof(t->cs.args)) != 0) {
+	if (narg != 0 && truss_ptrace_lwp(info, PT_GET_SC_ARGS, t->proc,
+	    t->tid, (caddr_t)t->cs.args, sizeof(t->cs.args)) != 0) {
 		free_syscall(t);
 		return;
 	}
@@ -555,7 +664,8 @@ exit_syscall(struct trussinfo *info, struct ptrace_lwpinfo *pl)
 
 	clock_gettime(CLOCK_REALTIME, &t->after);
 	p = t->proc;
-	if (ptrace(PT_GET_SC_RET, t->tid, (caddr_t)&psr, sizeof(psr)) != 0) {
+	if (truss_ptrace_lwp(info, PT_GET_SC_RET, p, t->tid, &psr,
+	    sizeof(psr)) != 0) {
 		free_syscall(t);
 		return;
 	}
@@ -624,11 +734,11 @@ exit_syscall(struct trussinfo *info, struct ptrace_lwpinfo *pl)
 	 */
 	if (pl->pl_flags & PL_FLAG_EXEC) {
 		assert(LIST_NEXT(LIST_FIRST(&p->threadlist), entries) == NULL);
-		p->abi = find_abi(p->pid);
+		p->abi = find_abi(info, p);
 		if (p->abi == NULL) {
-			if (ptrace(PT_DETACH, p->pid, (caddr_t)1, 0) < 0)
+			if (truss_ptrace(info, PT_DETACH, p, (caddr_t)1, 0) < 0)
 				err(1, "Can not detach the process");
-			free_proc(p);
+			free_proc(info, p);
 		}
 	}
 }
@@ -711,6 +821,9 @@ report_new_child(struct trussinfo *info)
 	struct threadinfo *t;
 
 	t = info->curthread;
+	if (t->proc->herald_printed)
+		return;
+	t->proc->herald_printed = true;
 	clock_gettime(CLOCK_REALTIME, &t->after);
 	t->before = t->after;
 	print_line_prefix(info);
@@ -790,6 +903,96 @@ report_signal(struct trussinfo *info, siginfo_t *si, struct ptrace_lwpinfo *pl)
 	
 }
 
+static void
+eventloop_handle_trapped(struct trussinfo *info, pid_t si_pid, int si_status,
+    siginfo_t *si)
+{
+	struct procinfo *np;
+	struct ptrace_lwpinfo pl;
+	int pending_signal;
+
+	np = find_proc(info, si_pid);
+	if (np == NULL) {
+		new_proc(info, si_pid, 0, -1, true, false);
+		np = find_proc(info, si_pid);
+	}
+	if (truss_ptrace(info, PT_LWPINFO, np, &pl, sizeof(pl)) == -1)
+		err(1, "ptrace(PT_LWPINFO)");
+
+	if ((pl.pl_flags & PL_FLAG_CHILD) != 0) {
+		assert(LIST_FIRST(&info->proclist)->abi != NULL);
+	} else if ((pl.pl_flags & PL_FLAG_BORN) != 0) {
+		new_thread(np, pl.pl_lwpid);
+	}
+	find_thread(info, si_pid, pl.pl_lwpid);
+
+	pending_signal = 0;
+	if (si_status == SIGTRAP && (pl.pl_flags & (PL_FLAG_BORN |
+	    PL_FLAG_EXITED | PL_FLAG_SCE | PL_FLAG_SCX)) != 0) {
+		if ((pl.pl_flags & PL_FLAG_BORN) != 0) {
+			if ((info->flags & COUNTONLY) == 0)
+				report_thread_birth(info);
+		} else if ((pl.pl_flags & PL_FLAG_EXITED) != 0) {
+			if ((info->flags & COUNTONLY) == 0)
+				report_thread_death(info);
+			free_thread(info->curthread);
+			info->curthread = NULL;
+		} else if ((pl.pl_flags & PL_FLAG_SCE) != 0) {
+			enter_syscall(info, info->curthread, &pl);
+		} else if ((pl.pl_flags & PL_FLAG_SCX) != 0) {
+			exit_syscall(info, &pl);
+		}
+	} else if ((pl.pl_flags & PL_FLAG_CHILD) != 0) {
+		if ((info->flags & COUNTONLY) == 0)
+			report_new_child(info);
+	} else if (si != NULL) {
+		if ((info->flags & NOSIGS) == 0)
+			report_signal(info, si, &pl);
+		pending_signal = si->si_status;
+	}
+	if (truss_ptrace(info, PT_SYSCALL, np, (caddr_t)1,
+	    pending_signal) == -1)
+		err(1, "ptrace(PT_SYSCALL)");
+}
+
+static void
+eventloop_handle_note_fork(struct trussinfo *info)
+{
+	struct ptrace_child *ptcs;
+	int cnt, i;
+
+again:
+	cnt = ptrace(PT_GET_CHILDREN, getpid(), NULL, 0);
+	if (cnt == -1)
+		err(1, "Unexpected error from ptrace(PT_GET_CHILDREN) size");
+	if (cnt == 0)
+		return;
+	ptcs = calloc(cnt, sizeof(*ptcs));
+	if (ptcs == NULL)
+		err(1, "No memory");
+	cnt = ptrace(PT_GET_CHILDREN, getpid(), (caddr_t)ptcs,
+	    cnt * sizeof(*ptcs));
+	if (cnt == -1) {
+		if (errno == ENOMEM) {
+			free(ptcs);
+			goto again;
+		}
+		err(1, "Unexpected error from ptrace(PT_GET_CHILDREN) data");
+	}
+	for (i = 0; i < cnt; i++) {
+		if ((ptcs[i].flags & (PTCHLD_TRACED | PTCHLD_TRACED_BY_ME |
+		    PTCHLD_EXITED)) != (PTCHLD_TRACED | PTCHLD_TRACED_BY_ME))
+			continue;
+		if (new_proc(info, ptcs[i].pid, 0, -1, true, true)) {
+			if ((info->flags & COUNTONLY) == 0)
+				report_new_child(info);
+			eventloop_handle_trapped(info, ptcs[i].pid, SIGTRAP,
+			    NULL);
+		}
+	}
+	free(ptcs);
+}
+
 /*
  * Wait for events until all the processes have exited or truss has been
  * asked to stop.
@@ -797,9 +1000,10 @@ report_signal(struct trussinfo *info, siginfo_t *si, struct ptrace_lwpinfo *pl)
 void
 eventloop(struct trussinfo *info)
 {
-	struct ptrace_lwpinfo pl;
 	siginfo_t si;
-	int pending_signal;
+	struct kevent ev[1];
+	int cnt, error;
+	bool has_si;
 
 	while (!LIST_EMPTY(&info->proclist)) {
 		if (detaching) {
@@ -807,11 +1011,52 @@ eventloop(struct trussinfo *info)
 			return;
 		}
 
-		if (waitid(P_ALL, 0, &si, WTRAPPED | WEXITED) == -1) {
-			if (errno == EINTR)
+		has_si = false;
+		if (info->cap_mode) {
+			cnt = kevent(info->pdkq, NULL, 0, ev, nitems(ev),
+			    NULL);
+			if (cnt == -1) {
+				if (errno == EINTR)
+					continue;
+				err(1, "Unexpected error from kevent");
+			}
+			if (cnt == 0) {
+				/* XXXKIB ? */
 				continue;
-			err(1, "Unexpected error from waitid");
+			}
+			if ((ev[0].fflags & (NOTE_EXIT | NOTE_PDSIGCHLD)) !=
+			    0) {
+				error = pdwait(ev[0].ident, NULL,
+				    WFLAGS | WNOHANG, NULL, &si);
+				if (error == -1) {
+					if (errno == EINTR ||
+					    errno == EWOULDBLOCK)
+						continue;
+					err(1, "Unexpected error from pdwait");
+				}
+				has_si = true;
+
+				/*
+				 * To get rid of zombie, we need to
+				 * waitpid() on it in addition to the
+				 * pdwait() above, because we are the
+				 * debugger, and the child was
+				 * reparented to us.
+				 */
+				waitpid(si.si_pid, NULL, WEXITED | WNOHANG);
+			}
+			if ((ev[0].fflags & NOTE_FORK) != 0)
+				eventloop_handle_note_fork(info);
+		} else {
+			if (waitid(P_ALL, 0, &si, WTRAPPED | WEXITED) == -1) {
+				if (errno == EINTR)
+					continue;
+				err(1, "Unexpected error from waitid");
+			}
+			has_si = true;
 		}
+		if (!has_si)
+			continue;
 
 		assert(si.si_signo == SIGCHLD);
 
@@ -825,50 +1070,12 @@ eventloop(struct trussinfo *info)
 					thread_exit_syscall(info);
 				report_exit(info, &si);
 			}
-			free_proc(info->curthread->proc);
+			free_proc(info, info->curthread->proc);
 			info->curthread = NULL;
 			break;
 		case CLD_TRAPPED:
-			if (ptrace(PT_LWPINFO, si.si_pid, (caddr_t)&pl,
-			    sizeof(pl)) == -1)
-				err(1, "ptrace(PT_LWPINFO)");
-
-			if (pl.pl_flags & PL_FLAG_CHILD) {
-				new_proc(info, si.si_pid, pl.pl_lwpid);
-				assert(LIST_FIRST(&info->proclist)->abi !=
-				    NULL);
-			} else if (pl.pl_flags & PL_FLAG_BORN)
-				new_thread(find_proc(info, si.si_pid),
-				    pl.pl_lwpid);
-			find_thread(info, si.si_pid, pl.pl_lwpid);
-
-			if (si.si_status == SIGTRAP &&
-			    (pl.pl_flags & (PL_FLAG_BORN|PL_FLAG_EXITED|
-			    PL_FLAG_SCE|PL_FLAG_SCX)) != 0) {
-				if (pl.pl_flags & PL_FLAG_BORN) {
-					if ((info->flags & COUNTONLY) == 0)
-						report_thread_birth(info);
-				} else if (pl.pl_flags & PL_FLAG_EXITED) {
-					if ((info->flags & COUNTONLY) == 0)
-						report_thread_death(info);
-					free_thread(info->curthread);
-					info->curthread = NULL;
-				} else if (pl.pl_flags & PL_FLAG_SCE)
-					enter_syscall(info, info->curthread, &pl);
-				else if (pl.pl_flags & PL_FLAG_SCX)
-					exit_syscall(info, &pl);
-				pending_signal = 0;
-			} else if (pl.pl_flags & PL_FLAG_CHILD) {
-				if ((info->flags & COUNTONLY) == 0)
-					report_new_child(info);
-				pending_signal = 0;
-			} else {
-				if ((info->flags & NOSIGS) == 0)
-					report_signal(info, &si, &pl);
-				pending_signal = si.si_status;
-			}
-			ptrace(PT_SYSCALL, si.si_pid, (caddr_t)1,
-			    pending_signal);
+			eventloop_handle_trapped(info, si.si_pid,
+			    si.si_status, &si);
 			break;
 		case CLD_STOPPED:
 			errx(1, "waitid reported CLD_STOPPED");
diff --git a/usr.bin/truss/syscalls.c b/usr.bin/truss/syscalls.c
index 6a78a4cf1007..5502ce83922b 100644
--- a/usr.bin/truss/syscalls.c
+++ b/usr.bin/truss/syscalls.c
@@ -950,7 +950,8 @@ get_syscall(struct threadinfo *t, u_int number, u_int nargs)
  * Copy a fixed amount of bytes from the process.
  */
 static int
-get_struct(pid_t pid, psaddr_t offset, void *buf, size_t len)
+get_struct(struct trussinfo *info, struct procinfo *p, psaddr_t offset,
+    void *buf, size_t len)
 {
 	struct ptrace_io_desc iorequest;
 
@@ -958,7 +959,7 @@ get_struct(pid_t pid, psaddr_t offset, void *buf, size_t len)
 	iorequest.piod_offs = (void *)(uintptr_t)offset;
 	iorequest.piod_addr = buf;
 	iorequest.piod_len = len;
-	if (ptrace(PT_IO, pid, (caddr_t)&iorequest, 0) < 0)
+	if (truss_ptrace(info, PT_IO, p, (caddr_t)&iorequest, 0) < 0)
 		return (-1);
 	return (0);
 }
@@ -971,7 +972,7 @@ get_struct(pid_t pid, psaddr_t offset, void *buf, size_t len)
  * only get that much.
  */
 static char *
-get_string(pid_t pid, psaddr_t addr, int max)
+get_string(struct trussinfo *info, struct procinfo *p, psaddr_t addr, int max)
 {
 	struct ptrace_io_desc iorequest;
 	char *buf, *nbuf;
@@ -995,7 +996,7 @@ get_string(pid_t pid, psaddr_t addr, int max)
 		iorequest.piod_offs = (void *)((uintptr_t)addr + offset);
 		iorequest.piod_addr = buf + offset;
 		iorequest.piod_len = size;
-		if (ptrace(PT_IO, pid, (caddr_t)&iorequest, 0) < 0) {
+		if (truss_ptrace(info, PT_IO, p, (caddr_t)&iorequest, 0) < 0) {
 			free(buf);
 			return (NULL);
 		}
@@ -1098,7 +1099,9 @@ print_sockaddr(FILE *fp, struct trussinfo *trussinfo, uintptr_t arg,
 	struct sockaddr_un *sun;
 	struct sockaddr *sa;
 	u_char *q;
-	pid_t pid = trussinfo->curthread->proc->pid;
+	struct procinfo *p;
+
+	p = trussinfo->curthread->proc;
 
 	if (arg == 0) {
 		fputs("NULL", fp);
@@ -1111,7 +1114,7 @@ print_sockaddr(FILE *fp, struct trussinfo *trussinfo, uintptr_t arg,
 	}
 
 	sa = calloc(1, len);
-	if (get_struct(pid, arg, sa, len) == -1) {
+	if (get_struct(trussinfo, p, arg, sa, len) == -1) {
 		free(sa);
 		print_pointer(fp, arg);
 		return;
@@ -1161,11 +1164,11 @@ print_sockaddr(FILE *fp, struct trussinfo *trussinfo, uintptr_t arg,
 static void
 print_iovec(FILE *fp, struct trussinfo *trussinfo, uintptr_t arg, int iovcnt)
 {
+	struct procinfo *p;
 	struct iovec iov[IOV_LIMIT];
 	size_t max_string = trussinfo->strsize;
 	char tmp2[max_string + 1], *tmp3;
 	size_t len;
-	pid_t pid = trussinfo->curthread->proc->pid;
 	int i;
 	bool buf_truncated, iov_truncated;
 
@@ -1179,7 +1182,9 @@ print_iovec(FILE *fp, struct trussinfo *trussinfo, uintptr_t arg, int iovcnt)
 	} else {
 		iov_truncated = false;
 	}
-	if (get_struct(pid, arg, &iov, iovcnt * sizeof(struct iovec)) == -1) {
+	p = trussinfo->curthread->proc;
+	if (get_struct(trussinfo, p, arg, &iov, iovcnt *
+	    sizeof(struct iovec)) == -1) {
 		print_pointer(fp, arg);
 		return;
 	}
@@ -1194,7 +1199,8 @@ print_iovec(FILE *fp, struct trussinfo *trussinfo, uintptr_t arg, int iovcnt)
 			buf_truncated = false;
 		}
 		fprintf(fp, "%s{", (i > 0) ? "," : "");
-		if (len && get_struct(pid, (uintptr_t)iov[i].iov_base, &tmp2, len) != -1) {
+		if (len && get_struct(trussinfo, p, (uintptr_t)iov[i].iov_base,
+		    &tmp2, len) != -1) {
 			tmp3 = malloc(len * 4 + 1);
 			while (len) {
 				if (strvisx(tmp3, tmp2, len,
@@ -1466,7 +1472,8 @@ print_sctp_cmsg(FILE *fp, bool receive, struct cmsghdr *cmsghdr)
 }
 
 static void
-print_cmsgs(FILE *fp, pid_t pid, bool receive, struct msghdr *msghdr)
+print_cmsgs(FILE *fp, struct trussinfo *info, struct procinfo *p,
+    bool receive, struct msghdr *msghdr)
 {
 	struct cmsghdr *cmsghdr;
 	char *cmsgbuf;
@@ -1481,7 +1488,8 @@ print_cmsgs(FILE *fp, pid_t pid, bool receive, struct msghdr *msghdr)
 		return;
 	}
 	cmsgbuf = calloc(1, len);
-	if (get_struct(pid, (uintptr_t)msghdr->msg_control, cmsgbuf, len) == -1) {
+	if (get_struct(info, p, (uintptr_t)msghdr->msg_control, cmsgbuf,
+	    len) == -1) {
 		print_pointer(fp, (uintptr_t)msghdr->msg_control);
 		free(cmsgbuf);
 		return;
@@ -1599,7 +1607,7 @@ print_netlink(FILE *fp, struct trussinfo *trussinfo, void *msg, size_t len,
     int protocol)
 {
 	char *buf;
-	pid_t pid = trussinfo->curthread->proc->pid;
+	struct procinfo *p;
 	bool success = false;
 
 	if (msg == NULL || len == 0)
@@ -1613,7 +1621,8 @@ print_netlink(FILE *fp, struct trussinfo *trussinfo, void *msg, size_t len,
 	if (buf == NULL)
 		return (false);
 
-	if (get_struct(pid, (uintptr_t)msg, buf, read_len) == -1) {
+	p = trussinfo->curthread->proc;
+	if (get_struct(trussinfo, p, (uintptr_t)msg, buf, read_len) == -1) {
 		free(buf);
 		return (false);
 	}
@@ -1637,12 +1646,12 @@ print_arg(struct syscall_arg *sc, syscallarg_t *args, syscallarg_t *retval,
     struct trussinfo *trussinfo, struct syscall_decode *decode)
 {
 	FILE *fp;
+	struct procinfo *p;
 	char *tmp;
 	size_t tmplen;
-	pid_t pid;
 
 	fp = open_memstream(&tmp, &tmplen);
-	pid = trussinfo->curthread->proc->pid;
+	p = trussinfo->curthread->proc;
 	switch (sc->type & ARG_MASK) {
 	case Hex:
 		fprintf(fp, "0x%x", (int)args[sc->offset]);
@@ -1659,7 +1668,7 @@ print_arg(struct syscall_arg *sc, syscallarg_t *args, syscallarg_t *retval,
 	case PUInt: {
 		unsigned int val;
 
-		if (get_struct(pid, args[sc->offset], &val,
+		if (get_struct(trussinfo, p, args[sc->offset], &val,
 		    sizeof(val)) == 0) 
 			fprintf(fp, "{ %u }", val);
 		else
@@ -1686,7 +1695,7 @@ print_arg(struct syscall_arg *sc, syscallarg_t *args, syscallarg_t *retval,
 		/* NULL-terminated string. */
 		char *tmp2;
 
-		tmp2 = get_string(pid, args[sc->offset], 0);
+		tmp2 = get_string(trussinfo, p, args[sc->offset], 0);
*** 525 LINES SKIPPED ***