2026-08-22: ryao: Did anyone know that struct lockf_range from flock/lockf can leak if a filesystem is force unmounted? If it doesn't leak due to a race with unmount before it locks the vnode, it looks like a UAF could happen. FreeBSD avoids this by having lf_purgelocks(), as far as I can tell. ryao: I caught it while doing an experiment porting a filesystem driver for fun. As for figuring it out, I guess I can do that. I will need to get back to you on it, as I have a busy schedule. 2026-08-23: ryao: I also managed to confirm the hypothetical UAF is there: https://bpa.st/DL2C6 ryao: I also have a candidate patch for fixing it: https://bpa.st/TN4G2 https://bpa.st/DL2C6 ``` holder locked waiter entering unmount -> 0 errno=0 holder=1 waiter=1 spray SIGUSR1 waiter 1047 panic: Bad link elm 0xfffff80066a440c8 prev->next != elm cpuid = 1 lf_setlock() at lf_setlock+0xcbd lf_advlock() at lf_advlock+0x1ce vop_advlock() at vop_advlock+0xa2 sys_flock() at sys_flock+0xc0 Debugger("panic") ``` https://bpa.st/TN4G2 ``` diff --git a/sys/kern/kern_lockf.c b/sys/kern/kern_lockf.c index 78f076660d..684eba22c2 100644 --- a/sys/kern/kern_lockf.c +++ b/sys/kern/kern_lockf.c @@ -247,11 +247,24 @@ lf_advlock(struct vop_advlock_args *ap, struct lockf *lock, u_quad_t size) */ token = lwkt_getpooltoken(lock); + /* + * Purge first: never re-init a lockf that vclean is tearing + * down, and do not republish vp->v_lockf once it has been + * NULLed under this token. + */ + if (lock->init_done && lock->lf_purging) { + lwkt_reltoken(token); + return (EBADF); + } if (lock->init_done == 0) { TAILQ_INIT(&lock->lf_range); TAILQ_INIT(&lock->lf_blocked); + lock->lf_threads = 0; + lock->lf_purging = 0; lock->init_done = 1; } + lock->lf_threads++; + ap->a_vp->v_lockf = lock; switch(ap->a_op) { case F_SETLK: @@ -289,6 +302,9 @@ lf_advlock(struct vop_advlock_args *ap, struct lockf *lock, u_quad_t size) error = EINVAL; break; } + lock->lf_threads--; + if (lock->lf_purging) + wakeup(lock); lwkt_reltoken(token); return(error); } @@ -317,6 +333,10 @@ lf_setlock(struct lockf *lock, struct proc *owner, int type, int flags, count = 0; restart: + if (lock->lf_purging) { + error = EBADF; + goto do_cleanup; + } /* * Preallocate two ranges so we don't have to worry about blocking * in the middle of the lock code. @@ -762,6 +782,67 @@ lf_setlock(struct lockf *lock, struct proc *owner, int type, int flags, return(error); } +/* + * Called from vclean_vxlocked before VOP_RECLAIM, while the node + * that embeds the lockf is still allocated. + * + * Same contract as FreeBSD lf_purgelocks(): mark this lockf dying, + * wake blocked waiters with the existing lf_wakeup protocol, wait + * for every lf_advlock thread to leave, then lf_destroy_range the + * leftover granted locks (the M_LOCKF leak). Waiters own their + * lf_blocked nodes and free them after tsleep. + * + * lf_purging is cleared on the way out so a surviving FS node + * (tmpfs/HAMMER2 vnode recycle with nlink > 0) can lock again + * on a new vnode. New lockers on *this* vnode are already + * rejected by VRECLAIMED in vop_advlock. + */ +void +lf_purgelocks(struct vnode *vp) +{ + struct lockf *lock; + struct lockf_range *range, *nrange; + lwkt_token_t token; + + lock = vp->v_lockf; + if (lock == NULL) + return; + + token = lwkt_getpooltoken(lock); + lock = vp->v_lockf; + if (lock == NULL || lock->init_done == 0) { + vp->v_lockf = NULL; + lwkt_reltoken(token); + return; + } + + lock->lf_purging = 1; + vp->v_lockf = NULL; + + TAILQ_FOREACH_MUTABLE(range, &lock->lf_blocked, lf_link, nrange) { + TAILQ_REMOVE(&lock->lf_blocked, range, lf_link); + range->lf_flags = 1; /* match lf_wakeup() */ + wakeup(range); + } + + while (lock->lf_threads > 0) { + lwkt_reltoken(token); + tsleep(lock, 0, "lfpurg", hz); + token = lwkt_getpooltoken(lock); + } + + while ((range = TAILQ_FIRST(&lock->lf_range)) != NULL) { + TAILQ_REMOVE(&lock->lf_range, range, lf_link); + if (range->lf_flags & F_POSIX) + (void)lf_count_change(range->lf_owner, -1); + lf_destroy_range(range); + } + + lock->init_done = 0; + lock->lf_purging = 0; + lwkt_reltoken(token); +} + /* * Check whether there is a blocking lock, * and if so return its process identifier. diff --git a/sys/kern/vfs_subr.c b/sys/kern/vfs_subr.c index 91477e9433..ecd77ec26f 100644 --- a/sys/kern/vfs_subr.c +++ b/sys/kern/vfs_subr.c @@ -67,6 +67,7 @@ #include #include #include +#include #include @@ -1296,10 +1297,13 @@ vclean_vxlocked(struct vnode *vp, int flags) return; /* - * Set flag to interlock operation, flag finalization to ensure - * that the vnode winds up on the inactive list, and set v_act to 0. + * Set VRECLAIMED under v_token so vop_advlock's check+VTOI + * cannot race us. flock/fcntl drop the vnode lock before + * VOP_ADVLOCK; v_token is the interlock that vn_lock is not. */ + lwkt_gettoken(&vp->v_token); vsetflags(vp, VRECLAIMED); + lwkt_reltoken(&vp->v_token); atomic_set_int(&vp->v_refcnt, VREF_FINALIZE); vp->v_act = 0; @@ -1396,6 +1400,16 @@ vclean_vxlocked(struct vnode *vp, int flags) if (vp->v_flag & VOBJDIRTY) vclrobjdirty(vp); + /* + * POSIX/flock ranges live in the filesystem node, not on + * the vnode. lf_advlock published vp->v_lockf. Purge + * while the inode still exists: wake waiters, drain + * lf_threads, lf_destroy_range leftover M_LOCKF nodes. + * After VOP_RECLAIM, deadfs vop_advlock is EBADF and + * those ranges would leak (or UAF the freed node). + */ + lf_purgelocks(vp); + /* * Reclaim the vnode if not already dead. */ diff --git a/sys/kern/vfs_vopops.c b/sys/kern/vfs_vopops.c index e183666b0f..96979df71f 100644 --- a/sys/kern/vfs_vopops.c +++ b/sys/kern/vfs_vopops.c @@ -1024,9 +1024,22 @@ vop_advlock(struct vop_ops *ops, struct vnode *vp, caddr_t id, int op, ap.a_fl = fl; ap.a_flags = flags; + /* + * flock/fcntl drop vn_lock before this wrapper, so vclean + * can run concurrently. Hold v_token across the VRECLAIMED + * check and the FS method (VTOI/VTOZ + lf_advlock). tsleep + * in lf_setlock drops the token; those threads are counted + * in lf_threads and lf_purgelocks waits for them. + */ + lwkt_gettoken(&vp->v_token); + if (vp->v_flag & VRECLAIMED) { + lwkt_reltoken(&vp->v_token); + return (EBADF); + } VFS_MPLOCK(vp->v_mount); DO_OPS(ops, error, &ap, vop_advlock); VFS_MPUNLOCK(); + lwkt_reltoken(&vp->v_token); return(error); } @@ -2004,8 +2017,15 @@ int vop_advlock_ap(struct vop_advlock_args *ap) { int error; + struct vnode *vp = ap->a_vp; + lwkt_gettoken(&vp->v_token); + if (vp->v_flag & VRECLAIMED) { + lwkt_reltoken(&vp->v_token); + return (EBADF); + } DO_OPS(ap->a_head.a_ops, error, ap, vop_advlock); + lwkt_reltoken(&vp->v_token); return(error); } diff --git a/sys/sys/lockf.h b/sys/sys/lockf.h index 8b0e281da6..0934a5d5c1 100644 --- a/sys/sys/lockf.h +++ b/sys/sys/lockf.h @@ -49,11 +49,14 @@ #endif struct vop_advlock_args; +struct vnode; /* * The lockf structure is a kernel structure which contains the information * associated with the byte range locks on an inode. The lockf structure is - * embedded in the inode structure. + * embedded in the filesystem node (inode, tmpfs_node, hammer2_inode, ...). + * lf_advlock also stores a back-pointer in vp->v_lockf so vclean can + * purge waiters and leftover ranges before VOP_RECLAIM. */ struct lockf_range { @@ -71,10 +74,13 @@ struct lockf { struct lockf_range_list lf_range; struct lockf_range_list lf_blocked; int init_done; + int lf_threads; /* threads inside lf_advlock */ + int lf_purging; /* vclean is destroying this lockf */ }; #ifdef _KERNEL int lf_advlock(struct vop_advlock_args *, struct lockf *, u_quad_t); +void lf_purgelocks(struct vnode *); void lf_count_adjust(struct proc *, int); extern int maxposixlocksperuid; diff --git a/sys/sys/vnode.h b/sys/sys/vnode.h index aeaa82339a..78e5303e61 100644 --- a/sys/sys/vnode.h +++ b/sys/sys/vnode.h @@ -75,6 +75,8 @@ #include #endif +struct lockf; + /* * The vnode is the focus of all file activity in UNIX. There is a * unique vnode allocated for each active file, each current directory, @@ -182,6 +184,7 @@ struct vnode { struct vm_object *v_object; /* Place to store VM object */ enum vtagtype v_tag; /* type of underlying data */ void *v_data; /* private data for fs */ + struct lockf *v_lockf; /* published by lf_advlock */ struct namecache_list v_namecache; /* (S) associated nc entries */ int v_namecache_count; /* (S) count of entries */ int v_auxrefs; /* vhold/vdrop refs */ ```