Home | History | Annotate | Line # | Download | only in common
      1 /*	$NetBSD: linux_file.c,v 1.137 2026/09/20 13:43:51 riastradh Exp $	*/
      2 
      3 /*-
      4  * Copyright (c) 1995, 1998, 2008 The NetBSD Foundation, Inc.
      5  * All rights reserved.
      6  *
      7  * This code is derived from software contributed to The NetBSD Foundation
      8  * by Frank van der Linden and Eric Haszlakiewicz.
      9  *
     10  * Redistribution and use in source and binary forms, with or without
     11  * modification, are permitted provided that the following conditions
     12  * are met:
     13  * 1. Redistributions of source code must retain the above copyright
     14  *    notice, this list of conditions and the following disclaimer.
     15  * 2. Redistributions in binary form must reproduce the above copyright
     16  *    notice, this list of conditions and the following disclaimer in the
     17  *    documentation and/or other materials provided with the distribution.
     18  *
     19  * THIS SOFTWARE IS PROVIDED BY THE NETBSD FOUNDATION, INC. AND CONTRIBUTORS
     20  * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
     21  * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
     22  * PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL THE FOUNDATION OR CONTRIBUTORS
     23  * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
     24  * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
     25  * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
     26  * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
     27  * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
     28  * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
     29  * POSSIBILITY OF SUCH DAMAGE.
     30  */
     31 
     32 /*
     33  * Functions in multiarch:
     34  *	linux_sys_llseek	: linux_llseek.c
     35  */
     36 
     37 #include <sys/cdefs.h>
     38 __KERNEL_RCSID(0, "$NetBSD: linux_file.c,v 1.137 2026/09/20 13:43:51 riastradh Exp $");
     39 
     40 #include <sys/types.h>
     41 #include <sys/param.h>
     42 #include <sys/systm.h>
     43 #include <sys/namei.h>
     44 #include <sys/proc.h>
     45 #include <sys/file.h>
     46 #include <sys/fcntl.h>
     47 #include <sys/stat.h>
     48 #include <sys/vfs_syscalls.h>
     49 #include <sys/filedesc.h>
     50 #include <sys/ioctl.h>
     51 #include <sys/kernel.h>
     52 #include <sys/mount.h>
     53 #include <sys/namei.h>
     54 #include <sys/vnode.h>
     55 #include <sys/tty.h>
     56 #include <sys/socketvar.h>
     57 #include <sys/conf.h>
     58 #include <sys/pipe.h>
     59 #include <sys/fstrans.h>
     60 #include <sys/syscallargs.h>
     61 #include <sys/vfs_syscalls.h>
     62 
     63 #include <compat/linux/common/linux_types.h>
     64 #include <compat/linux/common/linux_signal.h>
     65 #include <compat/linux/common/linux_fcntl.h>
     66 #include <compat/linux/common/linux_util.h>
     67 #include <compat/linux/common/linux_machdep.h>
     68 #include <compat/linux/common/linux_ipc.h>
     69 #include <compat/linux/common/linux_sem.h>
     70 
     71 #include <compat/linux/linux_syscallargs.h>
     72 
     73 #ifdef DEBUG_LINUX
     74 #define DPRINTF(a, ...)	uprintf(a, __VA_ARGS__)
     75 #else
     76 #define DPRINTF(a, ...)
     77 #endif
     78 
     79 #define LINUX_COPY_FILE_RANGE_MAX_CHUNK 8192
     80 
     81 static int bsd_to_linux_ioflags(int);
     82 #if !defined(__aarch64__) && !defined(__amd64__)
     83 static void bsd_to_linux_stat(struct stat *, struct linux_stat *);
     84 #endif
     85 
     86 conv_linux_flock(linux, flock)
     87 
     88 /*
     89  * Some file-related calls are handled here. The usual flag conversion
     90  * an structure conversion is done, and alternate emul path searching.
     91  */
     92 
     93 /*
     94  * The next two functions convert between the Linux and NetBSD values
     95  * of the flags used in open(2) and fcntl(2).
     96  */
     97 int
     98 linux_to_bsd_ioflags(int lflags)
     99 {
    100 	int res = 0;
    101 
    102 	res |= cvtto_bsd_mask(lflags, LINUX_O_WRONLY, O_WRONLY);
    103 	res |= cvtto_bsd_mask(lflags, LINUX_O_RDONLY, O_RDONLY);
    104 	res |= cvtto_bsd_mask(lflags, LINUX_O_RDWR, O_RDWR);
    105 
    106 	res |= cvtto_bsd_mask(lflags, LINUX_O_CREAT, O_CREAT);
    107 	res |= cvtto_bsd_mask(lflags, LINUX_O_EXCL, O_EXCL);
    108 	res |= cvtto_bsd_mask(lflags, LINUX_O_NOCTTY, O_NOCTTY);
    109 	res |= cvtto_bsd_mask(lflags, LINUX_O_TRUNC, O_TRUNC);
    110 	res |= cvtto_bsd_mask(lflags, LINUX_O_APPEND, O_APPEND);
    111 	res |= cvtto_bsd_mask(lflags, LINUX_O_NONBLOCK, O_NONBLOCK);
    112 	res |= cvtto_bsd_mask(lflags, LINUX_O_NDELAY, O_NDELAY);
    113 	res |= cvtto_bsd_mask(lflags, LINUX_O_SYNC, O_FSYNC);
    114 	res |= cvtto_bsd_mask(lflags, LINUX_FASYNC, O_ASYNC);
    115 	res |= cvtto_bsd_mask(lflags, LINUX_O_DIRECT, O_DIRECT);
    116 	res |= cvtto_bsd_mask(lflags, LINUX_O_DIRECTORY, O_DIRECTORY);
    117 	res |= cvtto_bsd_mask(lflags, LINUX_O_NOFOLLOW, O_NOFOLLOW);
    118 	res |= cvtto_bsd_mask(lflags, LINUX_O_CLOEXEC, O_CLOEXEC);
    119 
    120 	return res;
    121 }
    122 
    123 static int
    124 bsd_to_linux_ioflags(int bflags)
    125 {
    126 	int res = 0;
    127 
    128 	res |= cvtto_linux_mask(bflags, O_WRONLY, LINUX_O_WRONLY);
    129 	res |= cvtto_linux_mask(bflags, O_RDONLY, LINUX_O_RDONLY);
    130 	res |= cvtto_linux_mask(bflags, O_RDWR, LINUX_O_RDWR);
    131 
    132 	res |= cvtto_linux_mask(bflags, O_CREAT, LINUX_O_CREAT);
    133 	res |= cvtto_linux_mask(bflags, O_EXCL, LINUX_O_EXCL);
    134 	res |= cvtto_linux_mask(bflags, O_NOCTTY, LINUX_O_NOCTTY);
    135 	res |= cvtto_linux_mask(bflags, O_TRUNC, LINUX_O_TRUNC);
    136 	res |= cvtto_linux_mask(bflags, O_APPEND, LINUX_O_APPEND);
    137 	res |= cvtto_linux_mask(bflags, O_NONBLOCK, LINUX_O_NONBLOCK);
    138 	res |= cvtto_linux_mask(bflags, O_NDELAY, LINUX_O_NDELAY);
    139 	res |= cvtto_linux_mask(bflags, O_FSYNC, LINUX_O_SYNC);
    140 	res |= cvtto_linux_mask(bflags, O_ASYNC, LINUX_FASYNC);
    141 	res |= cvtto_linux_mask(bflags, O_DIRECT, LINUX_O_DIRECT);
    142 	res |= cvtto_linux_mask(bflags, O_DIRECTORY, LINUX_O_DIRECTORY);
    143 	res |= cvtto_linux_mask(bflags, O_NOFOLLOW, LINUX_O_NOFOLLOW);
    144 	res |= cvtto_linux_mask(bflags, O_CLOEXEC, LINUX_O_CLOEXEC);
    145 
    146 	return res;
    147 }
    148 
    149 static inline off_t
    150 linux_hilo_to_off_t(unsigned long hi, unsigned long lo)
    151 {
    152 #ifdef _LP64
    153 	/*
    154 	 * Linux discards the "hi" portion on LP64 platforms; even though
    155 	 * glibc puts of the upper 32-bits of the offset into the "hi"
    156 	 * argument regardless, the "lo" argument has all the bits in
    157 	 * this case.
    158 	 */
    159 	(void) hi;
    160 	return (off_t)lo;
    161 #else
    162 	return (((off_t)hi) << 32) | lo;
    163 #endif /* _LP64 */
    164 }
    165 
    166 #if !defined(__aarch64__)
    167 /*
    168  * creat(2) is an obsolete function, but it's present as a Linux
    169  * system call, so let's deal with it.
    170  *
    171  * Note: On the Alpha this doesn't really exist in Linux, but it's defined
    172  * in syscalls.master anyway so this doesn't have to be special cased.
    173  *
    174  * Just call open(2) with the TRUNC, CREAT and WRONLY flags.
    175  */
    176 int
    177 linux_sys_creat(struct lwp *l, const struct linux_sys_creat_args *uap,
    178     register_t *retval)
    179 {
    180 	/* {
    181 		syscallarg(const char *) path;
    182 		syscallarg(linux_umode_t) mode;
    183 	} */
    184 	struct sys_open_args oa;
    185 
    186 	memset(&oa, 0, sizeof(oa));
    187 	SCARG(&oa, path) = SCARG(uap, path);
    188 	SCARG(&oa, flags) = O_CREAT | O_TRUNC | O_WRONLY;
    189 	SCARG(&oa, mode) = SCARG(uap, mode);
    190 
    191 	return sys_open(l, &oa, retval);
    192 }
    193 #endif
    194 
    195 static void
    196 linux_open_ctty(struct lwp *l, int flags, int fd)
    197 {
    198 	struct proc *p = l->l_proc;
    199 
    200 	/*
    201 	 * this bit from sunos_misc.c (and svr4_fcntl.c).
    202 	 * If we are a session leader, and we don't have a controlling
    203 	 * terminal yet, and the O_NOCTTY flag is not set, try to make
    204 	 * this the controlling terminal.
    205 	 */
    206 	if (!(flags & O_NOCTTY) && SESS_LEADER(p) && !(p->p_lflag & PL_CONTROLT)) {
    207 		file_t *fp;
    208 
    209 		fp = fd_getfile(fd);
    210 
    211 		/* ignore any error, just give it a try */
    212 		if (fp != NULL) {
    213 			if (fp->f_type == DTYPE_VNODE) {
    214 				(fp->f_ops->fo_ioctl) (fp, TIOCSCTTY, NULL);
    215 			}
    216 			fd_putfile(fd);
    217 		}
    218 	}
    219 }
    220 
    221 /*
    222  * open(2). Take care of the different flag values, and let the
    223  * NetBSD syscall do the real work. See if this operation
    224  * gives the current process a controlling terminal.
    225  * (XXX is this necessary?)
    226  */
    227 int
    228 linux_sys_open(struct lwp *l, const struct linux_sys_open_args *uap,
    229     register_t *retval)
    230 {
    231 	/* {
    232 		syscallarg(const char *) path;
    233 		syscallarg(int) flags;
    234 		syscallarg(linux_umode_t) mode;
    235 	} */
    236 	int error, fl;
    237 	struct sys_open_args boa;
    238 
    239 	fl = linux_to_bsd_ioflags(SCARG(uap, flags));
    240 
    241 	memset(&boa, 0, sizeof(boa));
    242 	SCARG(&boa, path) = SCARG(uap, path);
    243 	SCARG(&boa, flags) = fl;
    244 	SCARG(&boa, mode) = SCARG(uap, mode);
    245 
    246 	if ((error = sys_open(l, &boa, retval)))
    247 		return (error == EFTYPE) ? ELOOP : error;
    248 
    249 	linux_open_ctty(l, fl, *retval);
    250 	return 0;
    251 }
    252 
    253 int
    254 linux_sys_openat(struct lwp *l, const struct linux_sys_openat_args *uap,
    255     register_t *retval)
    256 {
    257 	/* {
    258 		syscallarg(int) fd;
    259 		syscallarg(const char *) path;
    260 		syscallarg(int) flags;
    261 		syscallarg(linux_umode_t) mode;
    262 	} */
    263 	int error, fl;
    264 	struct sys_openat_args boa;
    265 
    266 	fl = linux_to_bsd_ioflags(SCARG(uap, flags));
    267 
    268 	memset(&boa, 0, sizeof(boa));
    269 	SCARG(&boa, fd) = SCARG(uap, fd);
    270 	SCARG(&boa, path) = SCARG(uap, path);
    271 	SCARG(&boa, oflags) = fl;
    272 	SCARG(&boa, mode) = SCARG(uap, mode);
    273 
    274 	if ((error = sys_openat(l, &boa, retval)))
    275 		return (error == EFTYPE) ? ELOOP : error;
    276 
    277 	linux_open_ctty(l, fl, *retval);
    278 	return 0;
    279 }
    280 
    281 /*
    282  * Most actions in the fcntl() call are straightforward; simply
    283  * pass control to the NetBSD system call. A few commands need
    284  * conversions after the actual system call has done its work,
    285  * because the flag values and lock structure are different.
    286  */
    287 int
    288 linux_sys_fcntl(struct lwp *l, const struct linux_sys_fcntl_args *uap,
    289     register_t *retval)
    290 {
    291 	/* {
    292 		syscallarg(int) fd;
    293 		syscallarg(int) cmd;
    294 		syscallarg(void *) arg;
    295 	} */
    296 	struct proc *p = l->l_proc;
    297 	int fd, cmd, error;
    298 	u_long val;
    299 	void *arg;
    300 	struct sys_fcntl_args fca;
    301 	file_t *fp;
    302 	struct vnode *vp;
    303 	struct vattr va;
    304 	long pgid;
    305 	struct pgrp *pgrp;
    306 	struct tty *tp;
    307 
    308 	fd = SCARG(uap, fd);
    309 	cmd = SCARG(uap, cmd);
    310 	arg = SCARG(uap, arg);
    311 
    312 	memset(&fca, 0, sizeof(fca));
    313 
    314 	switch (cmd) {
    315 
    316 	case LINUX_F_DUPFD:
    317 		cmd = F_DUPFD;
    318 		break;
    319 
    320 	case LINUX_F_GETFD:
    321 		cmd = F_GETFD;
    322 		break;
    323 
    324 	case LINUX_F_SETFD:
    325 		cmd = F_SETFD;
    326 		break;
    327 
    328 	case LINUX_F_GETFL:
    329 		SCARG(&fca, fd) = fd;
    330 		SCARG(&fca, cmd) = F_GETFL;
    331 		SCARG(&fca, arg) = arg;
    332 		if ((error = sys_fcntl(l, &fca, retval)))
    333 			return error;
    334 		retval[0] = bsd_to_linux_ioflags(retval[0]);
    335 		return 0;
    336 
    337 	case LINUX_F_SETFL: {
    338 		file_t	*fp1 = NULL;
    339 
    340 		val = linux_to_bsd_ioflags((unsigned long)SCARG(uap, arg));
    341 		/*
    342 		 * Linux seems to have same semantics for sending SIGIO to the
    343 		 * read side of socket, but slightly different semantics
    344 		 * for SIGIO to the write side.  Rather than sending the SIGIO
    345 		 * every time it's possible to write (directly) more data, it
    346 		 * only sends SIGIO if last write(2) failed due to insufficient
    347 		 * memory to hold the data. This is compatible enough
    348 		 * with NetBSD semantics to not do anything about the
    349 		 * difference.
    350 		 *
    351 		 * Linux does NOT send SIGIO for pipes. Deal with socketpair
    352 		 * ones and DTYPE_PIPE ones. For these, we don't set
    353 		 * the underlying flags (we don't pass O_ASYNC flag down
    354 		 * to sys_fcntl()), but set the FASYNC flag for file descriptor,
    355 		 * so that F_GETFL would report the ASYNC i/o is on.
    356 		 */
    357 		if (val & O_ASYNC) {
    358 			if (((fp1 = fd_getfile(fd)) == NULL))
    359 			    return (EBADF);
    360 			if (((fp1->f_type == DTYPE_SOCKET) && fp1->f_data
    361 			      && ((struct socket *)fp1->f_data)->so_state & SS_ISAPIPE)
    362 			    || (fp1->f_type == DTYPE_PIPE))
    363 				val &= ~O_ASYNC;
    364 			else {
    365 				/* not a pipe, do not modify anything */
    366 				fd_putfile(fd);
    367 				fp1 = NULL;
    368 			}
    369 		}
    370 
    371 		SCARG(&fca, fd) = fd;
    372 		SCARG(&fca, cmd) = F_SETFL;
    373 		SCARG(&fca, arg) = (void *) val;
    374 
    375 		error = sys_fcntl(l, &fca, retval);
    376 
    377 		/* Now set the FASYNC flag for pipes */
    378 		if (fp1) {
    379 			if (!error) {
    380 				mutex_enter(&fp1->f_lock);
    381 				fp1->f_flag |= FASYNC;
    382 				mutex_exit(&fp1->f_lock);
    383 			}
    384 			fd_putfile(fd);
    385 		}
    386 
    387 		return (error);
    388 	    }
    389 
    390 	case LINUX_F_GETLK:
    391 		do_linux_getlk(fd, cmd, arg, linux, flock);
    392 
    393 	case LINUX_F_SETLK:
    394 	case LINUX_F_SETLKW:
    395 		do_linux_setlk(fd, cmd, arg, linux, flock, LINUX_F_SETLK);
    396 
    397 	case LINUX_F_SETOWN:
    398 	case LINUX_F_GETOWN:
    399 		/*
    400 		 * We need to route fcntl() for tty descriptors around normal
    401 		 * fcntl(), since NetBSD tty TIOC{G,S}PGRP semantics is too
    402 		 * restrictive for Linux F_{G,S}ETOWN. For non-tty descriptors,
    403 		 * this is not a problem.
    404 		 */
    405 		if ((fp = fd_getfile(fd)) == NULL)
    406 			return EBADF;
    407 
    408 		/* Check it's a character device vnode */
    409 		if (fp->f_type != DTYPE_VNODE
    410 		    || (vp = (struct vnode *)fp->f_data) == NULL
    411 		    || vp->v_type != VCHR) {
    412 			fd_putfile(fd);
    413 
    414 	    not_tty:
    415 			/* Not a tty, proceed with common fcntl() */
    416 			cmd = cmd == LINUX_F_SETOWN ? F_SETOWN : F_GETOWN;
    417 			break;
    418 		}
    419 
    420 		vn_lock(vp, LK_SHARED | LK_RETRY);
    421 		error = VOP_GETATTR(vp, &va, l->l_cred);
    422 		VOP_UNLOCK(vp);
    423 
    424 		fd_putfile(fd);
    425 
    426 		if (error)
    427 			return error;
    428 
    429 		if ((tp = cdev_tty(va.va_rdev)) == NULL)
    430 			goto not_tty;
    431 
    432 		/* set tty pg_id appropriately */
    433 		mutex_enter(&proc_lock);
    434 		if (cmd == LINUX_F_GETOWN) {
    435 			retval[0] = tp->t_pgrp ? tp->t_pgrp->pg_id : NO_PGID;
    436 			mutex_exit(&proc_lock);
    437 			return 0;
    438 		}
    439 		if ((long)arg <= 0) {
    440 			pgid = -(long)arg;
    441 		} else {
    442 			struct proc *p1 = proc_find((long)arg);
    443 			if (p1 == NULL) {
    444 				mutex_exit(&proc_lock);
    445 				return (ESRCH);
    446 			}
    447 			pgid = (long)p1->p_pgrp->pg_id;
    448 		}
    449 		pgrp = pgrp_find(pgid);
    450 		if (pgrp == NULL || pgrp->pg_session != p->p_session) {
    451 			mutex_exit(&proc_lock);
    452 			return EPERM;
    453 		}
    454 		tp->t_pgrp = pgrp;
    455 		mutex_exit(&proc_lock);
    456 		return 0;
    457 
    458 	case LINUX_F_DUPFD_CLOEXEC:
    459 		cmd = F_DUPFD_CLOEXEC;
    460 		break;
    461 
    462 	case LINUX_F_ADD_SEALS:
    463 		cmd = F_ADD_SEALS;
    464 		break;
    465 
    466 	case LINUX_F_GET_SEALS:
    467 		cmd = F_GET_SEALS;
    468 		break;
    469 
    470 	default:
    471 		return EOPNOTSUPP;
    472 	}
    473 
    474 	SCARG(&fca, fd) = fd;
    475 	SCARG(&fca, cmd) = cmd;
    476 	SCARG(&fca, arg) = arg;
    477 
    478 	return sys_fcntl(l, &fca, retval);
    479 }
    480 
    481 #if !defined(__aarch64__) && !defined(__amd64__)
    482 /*
    483  * Convert a NetBSD stat structure to a Linux stat structure.
    484  * Only the order of the fields and the padding in the structure
    485  * is different. linux_fakedev is a machine-dependent function
    486  * which optionally converts device driver major/minor numbers
    487  * (XXX horrible, but what can you do against code that compares
    488  * things against constant major device numbers? sigh)
    489  */
    490 static void
    491 bsd_to_linux_stat(struct stat *bsp, struct linux_stat *lsp)
    492 {
    493 
    494 	memset(lsp, 0, sizeof(*lsp));
    495 	lsp->lst_dev     = linux_fakedev(bsp->st_dev, 0);
    496 	lsp->lst_ino     = bsp->st_ino;
    497 	lsp->lst_mode    = (linux_mode_t)bsp->st_mode;
    498 	if (bsp->st_nlink >= (1 << 15))
    499 		lsp->lst_nlink = (1 << 15) - 1;
    500 	else
    501 		lsp->lst_nlink = (linux_nlink_t)bsp->st_nlink;
    502 	lsp->lst_uid     = bsp->st_uid;
    503 	lsp->lst_gid     = bsp->st_gid;
    504 	lsp->lst_rdev    = linux_fakedev(bsp->st_rdev, 1);
    505 	lsp->lst_size    = bsp->st_size;
    506 	lsp->lst_blksize = bsp->st_blksize;
    507 	lsp->lst_blocks  = bsp->st_blocks;
    508 	lsp->lst_atime   = bsp->st_atime;
    509 	lsp->lst_mtime   = bsp->st_mtime;
    510 	lsp->lst_ctime   = bsp->st_ctime;
    511 #ifdef LINUX_STAT_HAS_NSEC
    512 	lsp->lst_atime_nsec   = bsp->st_atimensec;
    513 	lsp->lst_mtime_nsec   = bsp->st_mtimensec;
    514 	lsp->lst_ctime_nsec   = bsp->st_ctimensec;
    515 #endif
    516 }
    517 
    518 /*
    519  * The stat functions below are plain sailing. stat and lstat are handled
    520  * by one function to avoid code duplication.
    521  */
    522 int
    523 linux_sys_fstat(struct lwp *l, const struct linux_sys_fstat_args *uap,
    524     register_t *retval)
    525 {
    526 	/* {
    527 		syscallarg(int) fd;
    528 		syscallarg(linux_stat *) sp;
    529 	} */
    530 	struct linux_stat tmplst;
    531 	struct stat tmpst;
    532 	int error;
    533 
    534 	error = do_sys_fstat(SCARG(uap, fd), &tmpst);
    535 	if (error != 0)
    536 		return error;
    537 	bsd_to_linux_stat(&tmpst, &tmplst);
    538 
    539 	return copyout(&tmplst, SCARG(uap, sp), sizeof tmplst);
    540 }
    541 
    542 static int
    543 linux_stat1(const struct linux_sys_stat_args *uap, register_t *retval,
    544     int flags)
    545 {
    546 	struct linux_stat tmplst;
    547 	struct stat tmpst;
    548 	int error;
    549 
    550 	error = do_sys_stat(SCARG(uap, path), flags, &tmpst);
    551 	if (error != 0)
    552 		return error;
    553 
    554 	bsd_to_linux_stat(&tmpst, &tmplst);
    555 
    556 	return copyout(&tmplst, SCARG(uap, sp), sizeof tmplst);
    557 }
    558 
    559 int
    560 linux_sys_stat(struct lwp *l, const struct linux_sys_stat_args *uap,
    561     register_t *retval)
    562 {
    563 	/* {
    564 		syscallarg(const char *) path;
    565 		syscallarg(struct linux_stat *) sp;
    566 	} */
    567 
    568 	return linux_stat1(uap, retval, FOLLOW);
    569 }
    570 
    571 /* Note: this is "newlstat" in the Linux sources */
    572 /*	(we don't bother with the old lstat currently) */
    573 int
    574 linux_sys_lstat(struct lwp *l, const struct linux_sys_lstat_args *uap,
    575     register_t *retval)
    576 {
    577 	/* {
    578 		syscallarg(const char *) path;
    579 		syscallarg(struct linux_stat *) sp;
    580 	} */
    581 
    582 	return linux_stat1((const void *)uap, retval, NOFOLLOW);
    583 }
    584 #endif /* !__aarch64__ && !__amd64__ */
    585 
    586 /*
    587  * The following syscalls are mostly here because of the alternate path check.
    588  */
    589 
    590 int
    591 linux_sys_linkat(struct lwp *l, const struct linux_sys_linkat_args *uap,
    592     register_t *retval)
    593 {
    594 	/* {
    595 		syscallarg(int) fd1;
    596 		syscallarg(const char *) name1;
    597 		syscallarg(int) fd2;
    598 		syscallarg(const char *) name2;
    599 		syscallarg(int) flags;
    600 	} */
    601 	int fd1 = SCARG(uap, fd1);
    602 	const char *name1 = SCARG(uap, name1);
    603 	int fd2 = SCARG(uap, fd2);
    604 	const char *name2 = SCARG(uap, name2);
    605 	int follow;
    606 
    607 	follow = SCARG(uap, flags) & LINUX_AT_SYMLINK_FOLLOW;
    608 
    609 	return do_sys_linkat(l, fd1, name1, fd2, name2, follow, retval);
    610 }
    611 
    612 static int
    613 linux_unlink_dircheck(const char *path)
    614 {
    615 	struct nameidata nd;
    616 	struct pathbuf *pb;
    617 	int error;
    618 
    619 	/*
    620 	 * Linux returns EISDIR if unlink(2) is called on a directory.
    621 	 * We return EPERM in such cases. To emulate correct behaviour,
    622 	 * check if the path points to directory and return EISDIR if this
    623 	 * is the case.
    624 	 *
    625 	 * XXX this should really not copy in the path buffer twice...
    626 	 */
    627 	error = pathbuf_copyin(path, &pb);
    628 	if (error) {
    629 		return error;
    630 	}
    631 	NDINIT(&nd, LOOKUP, FOLLOW | LOCKLEAF | TRYEMULROOT, pb);
    632 	if (namei(&nd) == 0) {
    633 		struct stat sb;
    634 
    635 		if (vn_stat(nd.ni_vp, &sb) == 0
    636 		    && S_ISDIR(sb.st_mode))
    637 			error = EISDIR;
    638 
    639 		vput(nd.ni_vp);
    640 	}
    641 	pathbuf_destroy(pb);
    642 	return error ? error : EPERM;
    643 }
    644 
    645 int
    646 linux_sys_unlink(struct lwp *l, const struct linux_sys_unlink_args *uap,
    647     register_t *retval)
    648 {
    649 	/* {
    650 		syscallarg(const char *) path;
    651 	} */
    652 	int error;
    653 
    654 	error = sys_unlink(l, (const void *)uap, retval);
    655 	if (error == EPERM)
    656 		error = linux_unlink_dircheck(SCARG(uap, path));
    657 
    658 	return error;
    659 }
    660 
    661 int
    662 linux_sys_unlinkat(struct lwp *l, const struct linux_sys_unlinkat_args *uap,
    663     register_t *retval)
    664 {
    665 	/* {
    666 		syscallarg(int) fd;
    667 		syscallarg(const char *) path;
    668 		syscallarg(int) flag;
    669 	} */
    670 	struct sys_unlinkat_args ua;
    671 	int error;
    672 
    673 	memset(&ua, 0, sizeof(ua));
    674 	SCARG(&ua, fd) = SCARG(uap, fd);
    675 	SCARG(&ua, path) = SCARG(uap, path);
    676 	SCARG(&ua, flag) = linux_to_bsd_atflags(SCARG(uap, flag));
    677 
    678 	error = sys_unlinkat(l, &ua, retval);
    679 	if (error == EPERM)
    680 		error = linux_unlink_dircheck(SCARG(uap, path));
    681 
    682 	return error;
    683 }
    684 
    685 int
    686 linux_sys_mknod(struct lwp *l, const struct linux_sys_mknod_args *uap,
    687     register_t *retval)
    688 {
    689 	/* {
    690 		syscallarg(const char *) path;
    691 		syscallarg(linux_umode_t) mode;
    692 		syscallarg(unsigned) dev;
    693 	} */
    694 	struct linux_sys_mknodat_args ua;
    695 
    696 	memset(&ua, 0, sizeof(ua));
    697 	SCARG(&ua, fd) = LINUX_AT_FDCWD;
    698 	SCARG(&ua, path) = SCARG(uap, path);
    699 	SCARG(&ua, mode) = SCARG(uap, mode);
    700 	SCARG(&ua, dev) = SCARG(uap, dev);
    701 
    702 	return linux_sys_mknodat(l, &ua, retval);
    703 }
    704 
    705 int
    706 linux_sys_mknodat(struct lwp *l, const struct linux_sys_mknodat_args *uap,
    707     register_t *retval)
    708 {
    709 	/* {
    710 		syscallarg(int) fd;
    711 		syscallarg(const char *) path;
    712 		syscallarg(linux_umode_t) mode;
    713 		syscallarg(unsigned) dev;
    714 	} */
    715 
    716 	/*
    717 	 * BSD handles FIFOs separately
    718 	 */
    719 	if (S_ISFIFO(SCARG(uap, mode))) {
    720 		struct sys_mkfifoat_args bma;
    721 
    722 		memset(&bma, 0, sizeof(bma));
    723 		SCARG(&bma, fd) = SCARG(uap, fd);
    724 		SCARG(&bma, path) = SCARG(uap, path);
    725 		SCARG(&bma, mode) = SCARG(uap, mode);
    726 		return sys_mkfifoat(l, &bma, retval);
    727 	} else {
    728 
    729 		/*
    730 		 * Linux device numbers uses 8 bits for minor and 8 bits
    731 		 * for major. Due to how we map our major and minor,
    732 		 * this just fits into our dev_t. Just mask off the
    733 		 * upper 16bit to remove any random junk.
    734 		 */
    735 
    736 		return do_sys_mknodat(l, SCARG(uap, fd), SCARG(uap, path),
    737 		    SCARG(uap, mode), SCARG(uap, dev) & 0xffff, UIO_USERSPACE);
    738 	}
    739 }
    740 
    741 int
    742 linux_sys_fchmodat(struct lwp *l, const struct linux_sys_fchmodat_args *uap,
    743     register_t *retval)
    744 {
    745 	/* {
    746 		syscallarg(int) fd;
    747 		syscallarg(const char *) path;
    748 		syscallarg(linux_umode_t) mode;
    749 	} */
    750 
    751 	return do_sys_chmodat(l, SCARG(uap, fd), SCARG(uap, path),
    752 			      SCARG(uap, mode), AT_SYMLINK_FOLLOW);
    753 }
    754 
    755 int
    756 linux_sys_fchownat(struct lwp *l, const struct linux_sys_fchownat_args *uap,
    757     register_t *retval)
    758 {
    759 	/* {
    760 		syscallarg(int) fd;
    761 		syscallarg(const char *) path;
    762 		syscallarg(uid_t) owner;
    763 		syscallarg(gid_t) group;
    764 		syscallarg(int) flag;
    765 	} */
    766 	int flag;
    767 
    768 	flag = linux_to_bsd_atflags(SCARG(uap, flag));
    769 	return do_sys_chownat(l, SCARG(uap, fd), SCARG(uap, path),
    770 			      SCARG(uap, owner), SCARG(uap, group), flag);
    771 }
    772 
    773 int
    774 linux_sys_faccessat(struct lwp *l, const struct linux_sys_faccessat_args *uap,
    775     register_t *retval)
    776 {
    777 	/* {
    778 		syscallarg(int) fd;
    779 		syscallarg(const char *) path;
    780 		syscallarg(int) amode;
    781 	} */
    782 
    783 	return do_sys_accessat(l, SCARG(uap, fd), SCARG(uap, path),
    784 	     SCARG(uap, amode), AT_SYMLINK_FOLLOW);
    785 }
    786 
    787 /*
    788  * This is just fsync() for now (just as it is in the Linux kernel)
    789  * Note: this is not implemented under Linux on Alpha and Arm
    790  *	but should still be defined in our syscalls.master.
    791  *	(syscall #148 on the arm)
    792  */
    793 int
    794 linux_sys_fdatasync(struct lwp *l, const struct linux_sys_fdatasync_args *uap,
    795     register_t *retval)
    796 {
    797 	/* {
    798 		syscallarg(int) fd;
    799 	} */
    800 
    801 	return sys_fsync(l, (const void *)uap, retval);
    802 }
    803 
    804 /*
    805  * pread(2).
    806  */
    807 int
    808 linux_sys_pread(struct lwp *l, const struct linux_sys_pread_args *uap,
    809     register_t *retval)
    810 {
    811 	/* {
    812 		syscallarg(int) fd;
    813 		syscallarg(void *) buf;
    814 		syscallarg(size_t) nbyte;
    815 		syscallarg(off_t) offset;
    816 	} */
    817 	struct sys_pread_args pra;
    818 
    819 	memset(&pra, 0, sizeof(pra));
    820 	SCARG(&pra, fd) = SCARG(uap, fd);
    821 	SCARG(&pra, buf) = SCARG(uap, buf);
    822 	SCARG(&pra, nbyte) = SCARG(uap, nbyte);
    823 	SCARG(&pra, PAD) = 0;
    824 	SCARG(&pra, offset) = SCARG(uap, offset);
    825 
    826 	return sys_pread(l, &pra, retval);
    827 }
    828 
    829 /*
    830  * pwrite(2).
    831  */
    832 int
    833 linux_sys_pwrite(struct lwp *l, const struct linux_sys_pwrite_args *uap,
    834     register_t *retval)
    835 {
    836 	/* {
    837 		syscallarg(int) fd;
    838 		syscallarg(void *) buf;
    839 		syscallarg(size_t) nbyte;
    840 		syscallarg(off_t) offset;
    841 	} */
    842 	struct sys_pwrite_args pra;
    843 
    844 	memset(&pra, 0, sizeof(pra));
    845 	SCARG(&pra, fd) = SCARG(uap, fd);
    846 	SCARG(&pra, buf) = SCARG(uap, buf);
    847 	SCARG(&pra, nbyte) = SCARG(uap, nbyte);
    848 	SCARG(&pra, PAD) = 0;
    849 	SCARG(&pra, offset) = SCARG(uap, offset);
    850 
    851 	return sys_pwrite(l, &pra, retval);
    852 }
    853 
    854 /*
    855  * preadv(2)
    856  */
    857 int
    858 linux_sys_preadv(struct lwp *l, const struct linux_sys_preadv_args *uap,
    859     register_t *retval)
    860 {
    861 	/* {
    862 		syscallarg(int) fd;
    863 		syscallarg(const struct iovec *) iovp;
    864 		syscallarg(int) iovcnt;
    865 		syscallarg(unsigned long) off_lo;
    866 		syscallarg(unsigned long) off_hi;
    867 	} */
    868 	struct sys_preadv_args ua;
    869 
    870 	memset(&ua, 0, sizeof(ua));
    871 	SCARG(&ua, fd) = SCARG(uap, fd);
    872 	SCARG(&ua, iovp) = SCARG(uap, iovp);
    873 	SCARG(&ua, iovcnt) = SCARG(uap, iovcnt);
    874 	SCARG(&ua, PAD) = 0;
    875 	SCARG(&ua, offset) = linux_hilo_to_off_t(SCARG(uap, off_hi),
    876 						 SCARG(uap, off_lo));
    877 	return sys_preadv(l, &ua, retval);
    878 }
    879 
    880 /*
    881  * pwritev(2)
    882  */
    883 int
    884 linux_sys_pwritev(struct lwp *l, const struct linux_sys_pwritev_args *uap,
    885     register_t *retval)
    886 {
    887 	/* {
    888 		syscallarg(int) fd;
    889 		syscallarg(const struct iovec *) iovp;
    890 		syscallarg(int) iovcnt;
    891 		syscallarg(unsigned long) off_lo;
    892 		syscallarg(unsigned long) off_hi;
    893 	} */
    894 	struct sys_pwritev_args ua;
    895 
    896 	memset(&ua, 0, sizeof(ua));
    897 	SCARG(&ua, fd) = SCARG(uap, fd);
    898 	SCARG(&ua, iovp) = (const void *)SCARG(uap, iovp);
    899 	SCARG(&ua, iovcnt) = SCARG(uap, iovcnt);
    900 	SCARG(&ua, PAD) = 0;
    901 	SCARG(&ua, offset) = linux_hilo_to_off_t(SCARG(uap, off_hi),
    902 						 SCARG(uap, off_lo));
    903 	return sys_pwritev(l, &ua, retval);
    904 }
    905 
    906 /*
    907  * writev(2)
    908  */
    909 int
    910 linux_sys_writev(struct lwp *l, const struct linux_sys_writev_args *uap, register_t *retval)
    911 {
    912 	/* {
    913 		syscallarg(int) fd;
    914 		syscallarg(const struct iovec *) iovp;
    915 		syscallarg(int) iovcnt;
    916 	} */
    917 
    918 	int iovcnt = SCARG(uap, iovcnt);
    919 
    920 	if (iovcnt == 0) {
    921 		*retval = 0;
    922 		return 0;
    923 	}
    924 
    925 	struct sys_writev_args wra;
    926 
    927 	memset(&wra, 0, sizeof(wra));
    928 	SCARG(&wra, fd) = SCARG(uap, fd);
    929 	SCARG(&wra, iovp) = SCARG(uap, iovp);
    930 	SCARG(&wra, iovcnt) = iovcnt;
    931 
    932 	return sys_writev(l, &wra, retval);
    933 }
    934 
    935 int
    936 linux_sys_dup3(struct lwp *l, const struct linux_sys_dup3_args *uap,
    937     register_t *retval)
    938 {
    939 	/* {
    940 		syscallarg(int) from;
    941 		syscallarg(int) to;
    942 		syscallarg(int) flags;
    943 	} */
    944 	int flags;
    945 
    946 	flags = linux_to_bsd_ioflags(SCARG(uap, flags));
    947 	if ((flags & ~O_CLOEXEC) != 0)
    948 		return EINVAL;
    949 
    950 	if (SCARG(uap, from) == SCARG(uap, to))
    951 		return EINVAL;
    952 
    953 	return dodup(l, SCARG(uap, from), SCARG(uap, to), flags, retval);
    954 }
    955 
    956 int
    957 linux_to_bsd_atflags(int lflags)
    958 {
    959 	int bflags = 0;
    960 
    961 	if (lflags & LINUX_AT_SYMLINK_NOFOLLOW)
    962 		bflags |= AT_SYMLINK_NOFOLLOW;
    963 	if (lflags & LINUX_AT_REMOVEDIR)
    964 		bflags |= AT_REMOVEDIR;
    965 	if (lflags & LINUX_AT_SYMLINK_FOLLOW)
    966 		bflags |= AT_SYMLINK_FOLLOW;
    967 
    968 	return bflags;
    969 }
    970 
    971 int
    972 linux_sys_faccessat2(lwp_t *l, const struct linux_sys_faccessat2_args *uap,
    973     register_t *retval)
    974 {
    975 	/* {
    976 		syscallarg(int) fd;
    977 		syscallarg(const char *) path;
    978 		syscallarg(int) amode;
    979 		syscallarg(int) flags;
    980 	} */
    981 	int flag = linux_to_bsd_atflags(SCARG(uap, flags));
    982 	int mode = SCARG(uap, amode);
    983 	int fd = SCARG(uap, fd);
    984 	const char *path = SCARG(uap, path);
    985 
    986 	return do_sys_accessat(l, fd, path, mode, flag);
    987 }
    988 
    989 int
    990 linux_sys_sync_file_range(lwp_t *l,
    991     const struct linux_sys_sync_file_range_args *uap, register_t *retval)
    992 {
    993 	/* {
    994 		syscallarg(int) fd;
    995 		syscallarg(off_t) offset;
    996 		syscallarg(off_t) nbytes;
    997 		syscallarg(unsigned int) flags;
    998 	} */
    999 
   1000 	struct sys_fsync_range_args ua;
   1001 
   1002 	if (SCARG(uap, offset) < 0 || SCARG(uap, nbytes) < 0 ||
   1003 	    ((SCARG(uap, flags) & ~LINUX_SYNC_FILE_RANGE_ALL) != 0))
   1004 		return EINVAL;
   1005 
   1006 	memset(&ua, 0, sizeof(ua));
   1007 
   1008 	/* Fill ua with uap */
   1009 	SCARG(&ua, fd) = SCARG(uap, fd);
   1010 	SCARG(&ua, flags) = SCARG(uap, flags);
   1011 
   1012 	/* Round down offset to page boundary */
   1013 	SCARG(&ua, start) = rounddown(SCARG(uap, offset), PAGE_SIZE);
   1014 	SCARG(&ua, length) = SCARG(uap, nbytes);
   1015 	if (SCARG(&ua, length) != 0) {
   1016 		/* Round up length to nbytes+offset to page boundary */
   1017 		SCARG(&ua, length) = roundup(SCARG(uap, nbytes)
   1018 		    + SCARG(uap, offset) - SCARG(&ua, start), PAGE_SIZE);
   1019 	}
   1020 
   1021 	return sys_fsync_range(l, &ua, retval);
   1022 }
   1023 
   1024 int
   1025 linux_sys_syncfs(lwp_t *l, const struct linux_sys_syncfs_args *uap,
   1026     register_t *retval)
   1027 {
   1028 	/* {
   1029 		syscallarg(int) fd;
   1030 	} */
   1031 
   1032 	struct mount *mp;
   1033 	struct vnode *vp;
   1034 	file_t *fp;
   1035 	int error, fd;
   1036 	fd = SCARG(uap, fd);
   1037 
   1038 	/* Get file pointer */
   1039 	if ((error = fd_getvnode(fd, &fp)) != 0)
   1040 		return error;
   1041 
   1042 	/* Get vnode and mount point */
   1043 	vp = fp->f_vnode;
   1044 	mp = vp->v_mount;
   1045 
   1046 	mutex_enter(mp->mnt_updating);
   1047 	if ((mp->mnt_flag & MNT_RDONLY) == 0) {
   1048 		int asyncflag = mp->mnt_flag & MNT_ASYNC;
   1049 		mp->mnt_flag &= ~MNT_ASYNC;
   1050 		VFS_SYNC(mp, MNT_NOWAIT, l->l_cred);
   1051 		if (asyncflag)
   1052 			mp->mnt_flag |= MNT_ASYNC;
   1053 	}
   1054 	mutex_exit(mp->mnt_updating);
   1055 
   1056 	/* Cleanup vnode and file pointer */
   1057 	vrele(vp);
   1058 	fd_putfile(fd);
   1059 	return 0;
   1060 
   1061 }
   1062 
   1063 int
   1064 linux_sys_renameat2(struct lwp *l, const struct linux_sys_renameat2_args *uap,
   1065     register_t *retval)
   1066 {
   1067 	/* {
   1068 		syscallarg(int) fromfd;
   1069 		syscallarg(const char *) from;
   1070 		syscallarg(int) tofd;
   1071 		syscallarg(const char *) to;
   1072 		syscallarg(unsigned int) flags;
   1073 	} */
   1074 
   1075 	struct sys_renameat_args ua;
   1076 
   1077 	memset(&ua, 0, sizeof(ua));
   1078 	SCARG(&ua, fromfd) = SCARG(uap, fromfd);
   1079 	SCARG(&ua, from) = SCARG(uap, from);
   1080 	SCARG(&ua, tofd) = SCARG(uap, tofd);
   1081 	SCARG(&ua, to) = SCARG(uap, to);
   1082 
   1083 	unsigned int flags = SCARG(uap, flags);
   1084 	int error;
   1085 
   1086 	if (flags != 0) {
   1087 		if (flags & ~LINUX_RENAME_ALL)
   1088 			return EINVAL;
   1089 		if ((flags & LINUX_RENAME_EXCHANGE) != 0 &&
   1090 		    (flags & (LINUX_RENAME_NOREPLACE | LINUX_RENAME_WHITEOUT))
   1091 		    != 0)
   1092 			return EINVAL;
   1093 		/*
   1094 		 * Suppoting renameat2 flags without support from file systems
   1095 		 * becomes a messy affair cause of locks and how VOP_RENAME
   1096 		 * protocol is implemented. So, return EOPNOTSUPP for now.
   1097 		 */
   1098 		return EOPNOTSUPP;
   1099 	}
   1100 
   1101 	error = sys_renameat(l, &ua, retval);
   1102 	return error;
   1103 }
   1104 
   1105 int
   1106 linux_sys_copy_file_range(lwp_t *l,
   1107     const struct linux_sys_copy_file_range_args *uap, register_t *retval)
   1108 {
   1109 	/* {
   1110 		syscallarg(int) fd_in;
   1111 		syscallarg(unsigned long) off_in;
   1112 		syscallarg(int) fd_out;
   1113 		syscallarg(unsigned long) off_out;
   1114 		syscallarg(size_t) len;
   1115 		syscallarg(unsigned int) flags;
   1116 	} */
   1117 	const off_t OFF_MAX = __type_max(off_t);
   1118 	int fd_in, fd_out;
   1119 	file_t *fp_in, *fp_out;
   1120 	struct vnode *invp, *outvp;
   1121 	off_t off_in = 0, off_out = 0;
   1122 	struct vattr vattr_in, vattr_out;
   1123 	ssize_t total_copied = 0;
   1124 	size_t bytes_left, to_copy;
   1125 	bool have_off_in = false, have_off_out = false;
   1126 	int error = 0;
   1127 	size_t len = SCARG(uap, len);
   1128 	unsigned int flags = SCARG(uap, flags);
   1129 	/* Structures for actual copy */
   1130 	char *buffer = NULL;
   1131 	struct uio auio;
   1132 	struct iovec aiov;
   1133 
   1134 	if (len > SSIZE_MAX) {
   1135 		DPRINTF("%s: len is greater than SSIZE_MAX\n",
   1136 		    __func__);
   1137 		return EOVERFLOW;
   1138 	}
   1139 
   1140 	if (flags != 0) {
   1141 		DPRINTF("%s: unsupported flags %#x\n", __func__, flags);
   1142 		return EINVAL;
   1143 	}
   1144 
   1145 	fd_in = SCARG(uap, fd_in);
   1146 	fd_out = SCARG(uap, fd_out);
   1147 	error = fd_getvnode(fd_in, &fp_in);
   1148 	if (error) {
   1149 		return error;
   1150 	}
   1151 
   1152 	error = fd_getvnode(fd_out, &fp_out);
   1153 	if (error) {
   1154 		fd_putfile(fd_in);
   1155 		return error;
   1156 	}
   1157 
   1158 	invp = fp_in->f_vnode;
   1159 	outvp = fp_out->f_vnode;
   1160 
   1161 	/* Get attributes of input and output files */
   1162 	VOP_GETATTR(invp, &vattr_in, l->l_cred);
   1163 	VOP_GETATTR(outvp, &vattr_out, l->l_cred);
   1164 
   1165 	/* Check if input and output files are regular files */
   1166 	if (vattr_in.va_type == VDIR || vattr_out.va_type == VDIR) {
   1167 		error = EISDIR;
   1168 		DPRINTF("%s: Input or output is a directory\n", __func__);
   1169 		goto out;
   1170 	}
   1171 	if (vattr_in.va_type != VREG || vattr_out.va_type != VREG) {
   1172 		error = EINVAL;
   1173 		DPRINTF("%s: Invalid file type\n", __func__);
   1174 		goto out;
   1175 	}
   1176 
   1177 	if ((fp_in->f_flag & FREAD) == 0 ||
   1178 	    (fp_out->f_flag & FWRITE) == 0 ||
   1179 	    (fp_out->f_flag & FAPPEND) != 0) {
   1180 		DPRINTF("%s: input file can't be read or output file "
   1181 		    "can't be written\n", __func__);
   1182 		error = EBADF;
   1183 		goto out;
   1184 	}
   1185 	/* Retrieve and validate offsets if provided */
   1186 	if (SCARG(uap, off_in) != NULL) {
   1187 		error = copyin(SCARG(uap, off_in), &off_in, sizeof(off_in));
   1188 		if (error) {
   1189 			goto out;
   1190 		}
   1191 		have_off_in = true;
   1192 	}
   1193 
   1194 	if (SCARG(uap, off_out) != NULL) {
   1195 		error = copyin(SCARG(uap, off_out), &off_out, sizeof(off_out));
   1196 		if (error) {
   1197 			goto out;
   1198 		}
   1199 		have_off_out = true;
   1200 	}
   1201 
   1202 	if (off_out < 0 || len > OFF_MAX - off_out ||
   1203 	    off_in < 0 || len > OFF_MAX - off_in) {
   1204 		DPRINTF("%s: New size is greater than OFF_MAX\n", __func__);
   1205 		error = EFBIG;
   1206 		goto out;
   1207 	}
   1208 
   1209 	/* Identify overlapping ranges */
   1210 	if ((invp == outvp) &&
   1211 	    ((off_in <= off_out && off_in + (off_t)len > off_out) ||
   1212 		(off_in > off_out && off_out + (off_t)len > off_in))) {
   1213 		DPRINTF("%s: Ranges overlap\n", __func__);
   1214 		error = EINVAL;
   1215 		goto out;
   1216 	}
   1217 
   1218 	buffer = kmem_alloc(LINUX_COPY_FILE_RANGE_MAX_CHUNK, KM_SLEEP);
   1219 
   1220 	bytes_left = len;
   1221 
   1222 	while (bytes_left > 0) {
   1223 		to_copy = MIN(bytes_left, LINUX_COPY_FILE_RANGE_MAX_CHUNK);
   1224 
   1225 		/* Lock the input vnode for reading */
   1226 		vn_lock(fp_in->f_vnode, LK_SHARED | LK_RETRY);
   1227 		/* Set up iovec and uio for reading */
   1228 		aiov.iov_base = buffer;
   1229 		aiov.iov_len = to_copy;
   1230 		auio.uio_iov = &aiov;
   1231 		auio.uio_iovcnt = 1;
   1232 		auio.uio_offset = have_off_in ? off_in : fp_in->f_offset;
   1233 		auio.uio_resid = to_copy;
   1234 		auio.uio_rw = UIO_READ;
   1235 		auio.uio_vmspace = l->l_proc->p_vmspace;
   1236 		UIO_SETUP_SYSSPACE(&auio);
   1237 
   1238 		/* Perform read using vn_read */
   1239 		error = VOP_READ(fp_in->f_vnode, &auio, 0, l->l_cred);
   1240 		VOP_UNLOCK(fp_in->f_vnode);
   1241 		if (error) {
   1242 			DPRINTF("%s: Read error %d\n", __func__, error);
   1243 			break;
   1244 		}
   1245 
   1246 		size_t read_bytes = to_copy - auio.uio_resid;
   1247 		if (read_bytes == 0) {
   1248 			/* EOF reached */
   1249 			break;
   1250 		}
   1251 
   1252 		/* Lock the output vnode for writing */
   1253 		vn_lock(fp_out->f_vnode, LK_EXCLUSIVE | LK_RETRY);
   1254 		/* Set up iovec and uio for writing */
   1255 		aiov.iov_base = buffer;
   1256 		aiov.iov_len = read_bytes;
   1257 		auio.uio_iov = &aiov;
   1258 		auio.uio_iovcnt = 1;
   1259 		auio.uio_offset = have_off_out ? off_out : fp_out->f_offset;
   1260 		auio.uio_resid = read_bytes;
   1261 		auio.uio_rw = UIO_WRITE;
   1262 		auio.uio_vmspace = l->l_proc->p_vmspace;
   1263 		UIO_SETUP_SYSSPACE(&auio);
   1264 
   1265 		/* Perform the write */
   1266 		error = VOP_WRITE(fp_out->f_vnode, &auio, 0, l->l_cred);
   1267 		VOP_UNLOCK(fp_out->f_vnode);
   1268 		if (error) {
   1269 			DPRINTF("%s: Write error %d\n", __func__, error);
   1270 			break;
   1271 		}
   1272 		size_t written_bytes = read_bytes - auio.uio_resid;
   1273 		total_copied += written_bytes;
   1274 		bytes_left -= written_bytes;
   1275 
   1276 		/* Update offsets if provided */
   1277 		if (have_off_in) {
   1278 			off_in += written_bytes;
   1279 		} else {
   1280 			fp_in->f_offset += written_bytes;
   1281 		}
   1282 		if (have_off_out) {
   1283 			off_out += written_bytes;
   1284 		} else {
   1285 			fp_out->f_offset += written_bytes;
   1286 		}
   1287 	}
   1288 
   1289 	if (have_off_in) {
   1290 		/* Adjust user space offset */
   1291 		error = copyout(&off_in, SCARG(uap, off_in), sizeof(off_t));
   1292 		if (error) {
   1293 			DPRINTF("%s: Error adjusting user space offset\n",
   1294 			    __func__);
   1295 		}
   1296 		goto out;
   1297 	}
   1298 
   1299 	if (have_off_out) {
   1300 		/* Adjust user space offset */
   1301 		error = copyout(&off_out, SCARG(uap, off_out), sizeof(off_t));
   1302 		if (error) {
   1303 			DPRINTF("%s: Error adjusting user space offset\n",
   1304 			    __func__);
   1305 		}
   1306 	}
   1307 
   1308 	*retval = total_copied;
   1309 out:
   1310 	if (buffer) {
   1311 		kmem_free(buffer, LINUX_COPY_FILE_RANGE_MAX_CHUNK);
   1312 	}
   1313 	if (fp_out) {
   1314 		fd_putfile(fd_out);
   1315 	}
   1316 	if (fp_in) {
   1317 		fd_putfile(fd_in);
   1318 	}
   1319 	return error;
   1320 }
   1321 
   1322 #define LINUX_NOT_SUPPORTED(fun) \
   1323 int \
   1324 fun(struct lwp *l, const struct fun##_args *uap, register_t *retval) \
   1325 { \
   1326 	return EOPNOTSUPP; \
   1327 }
   1328 
   1329 LINUX_NOT_SUPPORTED(linux_sys_setxattr)
   1330 LINUX_NOT_SUPPORTED(linux_sys_lsetxattr)
   1331 LINUX_NOT_SUPPORTED(linux_sys_fsetxattr)
   1332 
   1333 LINUX_NOT_SUPPORTED(linux_sys_getxattr)
   1334 LINUX_NOT_SUPPORTED(linux_sys_lgetxattr)
   1335 LINUX_NOT_SUPPORTED(linux_sys_fgetxattr)
   1336 
   1337 LINUX_NOT_SUPPORTED(linux_sys_listxattr)
   1338 LINUX_NOT_SUPPORTED(linux_sys_llistxattr)
   1339 LINUX_NOT_SUPPORTED(linux_sys_flistxattr)
   1340 
   1341 LINUX_NOT_SUPPORTED(linux_sys_removexattr)
   1342 LINUX_NOT_SUPPORTED(linux_sys_lremovexattr)
   1343 LINUX_NOT_SUPPORTED(linux_sys_fremovexattr)
   1344 
   1345 /*
   1346  * For now just return EOPNOTSUPP, this makes glibc posix_fallocate()
   1347  * to fallback to emulation.
   1348  * XXX Right now no filesystem actually implements fallocate support,
   1349  * so no need for mapping.
   1350  */
   1351 LINUX_NOT_SUPPORTED(linux_sys_fallocate)
   1352