1 /* $NetBSD: linux_file64.c,v 1.69 2026/09/20 13:43:51 riastradh Exp $ */ 2 3 /*- 4 * Copyright (c) 1995, 1998, 2000, 2008 The NetBSD Foundation, Inc. 5 * All rights reserved. 6 * 7 * This code is derived from software contributed to The NetBSD Foundation 8 * by Frank van der Linden and Eric Haszlakiewicz. 9 * 10 * Redistribution and use in source and binary forms, with or without 11 * modification, are permitted provided that the following conditions 12 * are met: 13 * 1. Redistributions of source code must retain the above copyright 14 * notice, this list of conditions and the following disclaimer. 15 * 2. Redistributions in binary form must reproduce the above copyright 16 * notice, this list of conditions and the following disclaimer in the 17 * documentation and/or other materials provided with the distribution. 18 * 19 * THIS SOFTWARE IS PROVIDED BY THE NETBSD FOUNDATION, INC. AND CONTRIBUTORS 20 * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED 21 * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR 22 * PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR CONTRIBUTORS 23 * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR 24 * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF 25 * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS 26 * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN 27 * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) 28 * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE 29 * POSSIBILITY OF SUCH DAMAGE. 30 */ 31 32 /* 33 * Linux 64bit filesystem calls. Used on 32bit archs, not used on 64bit ones. 34 */ 35 36 #include <sys/cdefs.h> 37 __KERNEL_RCSID(0, "$NetBSD: linux_file64.c,v 1.69 2026/09/20 13:43:51 riastradh Exp $"); 38 39 #include <sys/param.h> 40 #include <sys/systm.h> 41 #include <sys/namei.h> 42 #include <sys/proc.h> 43 #include <sys/dirent.h> 44 #include <sys/file.h> 45 #include <sys/stat.h> 46 #include <sys/filedesc.h> 47 #include <sys/ioctl.h> 48 #include <sys/kernel.h> 49 #include <sys/mount.h> 50 #include <sys/malloc.h> 51 #include <sys/namei.h> 52 #include <sys/vfs_syscalls.h> 53 #include <sys/vnode.h> 54 #include <sys/tty.h> 55 #include <sys/conf.h> 56 57 #include <sys/syscallargs.h> 58 59 #include <compat/linux/common/linux_types.h> 60 #include <compat/linux/common/linux_signal.h> 61 #include <compat/linux/common/linux_fcntl.h> 62 #include <compat/linux/common/linux_util.h> 63 #include <compat/linux/common/linux_machdep.h> 64 #include <compat/linux/common/linux_dirent.h> 65 #include <compat/linux/common/linux_ipc.h> 66 #include <compat/linux/common/linux_sem.h> 67 68 #include <compat/linux/linux_syscall.h> 69 #include <compat/linux/linux_syscallargs.h> 70 71 static void bsd_to_linux_stat64(struct stat *, struct linux_stat64 *); 72 73 /* 74 * Convert a NetBSD stat structure to a Linux stat structure. 75 * Only the order of the fields and the padding in the structure 76 * is different. linux_fakedev is a machine-dependent function 77 * which optionally converts device driver major/minor numbers 78 * (XXX horrible, but what can you do against code that compares 79 * things against constant major device numbers? sigh) 80 */ 81 static void 82 bsd_to_linux_stat64(struct stat *bsp, struct linux_stat64 *lsp) 83 { 84 memset(lsp, 0, sizeof(*lsp)); 85 lsp->lst_dev = linux_fakedev(bsp->st_dev, 0); 86 lsp->lst_ino = bsp->st_ino; 87 lsp->lst_mode = (linux_mode_t)bsp->st_mode; 88 if (bsp->st_nlink >= (1 << 15)) 89 lsp->lst_nlink = (1 << 15) - 1; 90 else 91 lsp->lst_nlink = (linux_nlink_t)bsp->st_nlink; 92 lsp->lst_uid = bsp->st_uid; 93 lsp->lst_gid = bsp->st_gid; 94 lsp->lst_rdev = linux_fakedev(bsp->st_rdev, 1); 95 lsp->lst_size = bsp->st_size; 96 lsp->lst_blksize = bsp->st_blksize; 97 lsp->lst_blocks = bsp->st_blocks; 98 lsp->lst_atime = bsp->st_atime; 99 lsp->lst_mtime = bsp->st_mtime; 100 lsp->lst_ctime = bsp->st_ctime; 101 # ifdef LINUX_STAT64_HAS_NSEC 102 lsp->lst_atime_nsec = bsp->st_atimensec; 103 lsp->lst_mtime_nsec = bsp->st_mtimensec; 104 lsp->lst_ctime_nsec = bsp->st_ctimensec; 105 # endif 106 # if LINUX_STAT64_HAS_BROKEN_ST_INO 107 lsp->__lst_ino = (linux_ino_t) bsp->st_ino; 108 # endif 109 } 110 111 int 112 bsd_to_linux_statx(struct stat *st, struct linux_statx *stx, 113 unsigned int mask) 114 { 115 if (mask & STATX__RESERVED) 116 return EINVAL; 117 118 /* XXX: STATX_MNT_ID is not supported */ 119 unsigned int rmask = STATX_TYPE | STATX_MODE | STATX_NLINK | 120 STATX_UID | STATX_GID | STATX_ATIME | STATX_MTIME | STATX_CTIME | 121 STATX_INO | STATX_SIZE | STATX_BLOCKS | STATX_BTIME; 122 123 memset(stx, 0, sizeof(*stx)); 124 125 if ((st->st_flags & UF_NODUMP) != 0) 126 stx->stx_attributes |= STATX_ATTR_NODUMP; 127 if ((st->st_flags & (UF_IMMUTABLE|SF_IMMUTABLE)) != 0) 128 stx->stx_attributes |= STATX_ATTR_IMMUTABLE; 129 if ((st->st_flags & (UF_APPEND|SF_APPEND)) != 0) 130 stx->stx_attributes |= STATX_ATTR_APPEND; 131 132 stx->stx_attributes_mask = 133 STATX_ATTR_NODUMP | STATX_ATTR_IMMUTABLE | STATX_ATTR_APPEND; 134 135 stx->stx_blksize = st->st_blksize; 136 137 stx->stx_nlink = st->st_nlink; 138 stx->stx_uid = st->st_uid; 139 stx->stx_gid = st->st_gid; 140 stx->stx_mode |= st->st_mode & S_IFMT; 141 stx->stx_mode |= st->st_mode & ~S_IFMT; 142 stx->stx_ino = st->st_ino; 143 stx->stx_size = st->st_size; 144 stx->stx_blocks = st->st_blocks; 145 146 stx->stx_atime.tv_sec = st->st_atime; 147 stx->stx_atime.tv_nsec = st->st_atimensec; 148 149 /* some filesystem has no birthtime returns 0 or -1 */ 150 if ((st->st_birthtime == 0 && st->st_birthtimensec == 0) || 151 (st->st_birthtime == (time_t)-1 && 152 st->st_birthtimensec == (long)-1)) { 153 rmask &= ~STATX_BTIME; 154 } else { 155 stx->stx_btime.tv_sec = st->st_birthtime; 156 stx->stx_btime.tv_nsec = st->st_birthtimensec; 157 } 158 159 stx->stx_ctime.tv_sec = st->st_ctime; 160 stx->stx_ctime.tv_nsec = st->st_ctimensec; 161 162 stx->stx_mtime.tv_sec = st->st_mtime; 163 stx->stx_mtime.tv_nsec = st->st_mtimensec; 164 165 if (S_ISCHR(st->st_mode) || S_ISBLK(st->st_mode)) { 166 stx->stx_rdev_major = major(st->st_rdev); 167 stx->stx_rdev_minor = minor(st->st_rdev); 168 } else { 169 stx->stx_dev_major = major(st->st_rdev); 170 stx->stx_dev_minor = minor(st->st_rdev); 171 } 172 173 stx->stx_mask = rmask; 174 175 return 0; 176 } 177 178 /* 179 * The stat functions below are plain sailing. stat and lstat are handled 180 * by one function to avoid code duplication. 181 */ 182 int 183 linux_sys_fstat64(struct lwp *l, const struct linux_sys_fstat64_args *uap, register_t *retval) 184 { 185 /* { 186 syscallarg(int) fd; 187 syscallarg(struct linux_stat64 *) sp; 188 } */ 189 struct linux_stat64 tmplst; 190 struct stat tmpst; 191 int error; 192 193 error = do_sys_fstat(SCARG(uap, fd), &tmpst); 194 if (error != 0) 195 return error; 196 197 bsd_to_linux_stat64(&tmpst, &tmplst); 198 199 return copyout(&tmplst, SCARG(uap, sp), sizeof tmplst); 200 } 201 202 #if !defined(__aarch64__) 203 static int 204 linux_do_stat64(struct lwp *l, const struct linux_sys_stat64_args *uap, register_t *retval, int flags) 205 { 206 struct linux_stat64 tmplst; 207 struct stat tmpst; 208 int error; 209 210 error = do_sys_stat(SCARG(uap, path), flags, &tmpst); 211 if (error != 0) 212 return error; 213 214 bsd_to_linux_stat64(&tmpst, &tmplst); 215 216 return copyout(&tmplst, SCARG(uap, sp), sizeof tmplst); 217 } 218 219 int 220 linux_sys_stat64(struct lwp *l, const struct linux_sys_stat64_args *uap, register_t *retval) 221 { 222 /* { 223 syscallarg(const char *) path; 224 syscallarg(struct linux_stat64 *) sp; 225 } */ 226 227 return linux_do_stat64(l, uap, retval, FOLLOW); 228 } 229 230 int 231 linux_sys_lstat64(struct lwp *l, const struct linux_sys_lstat64_args *uap, register_t *retval) 232 { 233 /* { 234 syscallarg(const char *) path; 235 syscallarg(struct linux_stat64 *) sp; 236 } */ 237 238 return linux_do_stat64(l, (const void *)uap, retval, NOFOLLOW); 239 } 240 #endif 241 242 /* 243 * This is an internal function for the *statat() variant of linux, 244 * which returns struct stat, but flags and other handling are 245 * the same as in linux. 246 */ 247 int 248 linux_statat(struct lwp *l, int fd, const char *path, int lflag, 249 struct stat *st) 250 { 251 struct vnode *vp; 252 int error, nd_flag; 253 uint8_t c; 254 255 if (lflag & ~(LINUX_AT_EMPTY_PATH|LINUX_AT_NO_AUTOMOUNT 256 |LINUX_AT_SYMLINK_NOFOLLOW)) 257 return EINVAL; 258 259 if (lflag & LINUX_AT_EMPTY_PATH) { 260 /* 261 * If path is null string: 262 */ 263 error = ufetch_8(path, &c); 264 if (error != 0) 265 return error; 266 if (c == '\0') { 267 if (fd == LINUX_AT_FDCWD) { 268 /* 269 * operate on current directory 270 */ 271 vp = l->l_proc->p_cwdi->cwdi_cdir; 272 vn_lock(vp, LK_EXCLUSIVE | LK_RETRY); 273 error = vn_stat(vp, st); 274 VOP_UNLOCK(vp); 275 } else { 276 /* 277 * operate on fd 278 */ 279 error = do_sys_fstat(fd, st); 280 } 281 return error; 282 } 283 } 284 285 if (lflag & LINUX_AT_SYMLINK_NOFOLLOW) 286 nd_flag = NOFOLLOW; 287 else 288 nd_flag = FOLLOW; 289 290 return do_sys_statat(l, fd, path, nd_flag, st); 291 } 292 293 int 294 linux_sys_fstatat64(struct lwp *l, const struct linux_sys_fstatat64_args *uap, register_t *retval) 295 { 296 /* { 297 syscallarg(int) fd; 298 syscallarg(const char *) path; 299 syscallarg(struct linux_stat64 *) sp; 300 syscallarg(int) flag; 301 } */ 302 struct linux_stat64 tmplst; 303 struct stat tmpst; 304 int error; 305 306 error = linux_statat(l, SCARG(uap, fd), SCARG(uap, path), 307 SCARG(uap, flag), &tmpst); 308 if (error != 0) 309 return error; 310 311 bsd_to_linux_stat64(&tmpst, &tmplst); 312 313 return copyout(&tmplst, SCARG(uap, sp), sizeof tmplst); 314 } 315 316 #ifdef LINUX_SYS_statx 317 int 318 linux_sys_statx(struct lwp *l, const struct linux_sys_statx_args *uap, 319 register_t *retval) 320 { 321 /* { 322 syscallarg(int) fd; 323 syscallarg(const char *) path; 324 syscallarg(int) flag; 325 syscallarg(unsigned int) mask; 326 syscallarg(struct linux_statx *) sp; 327 } */ 328 struct linux_statx stx; 329 struct stat st; 330 int error; 331 332 error = linux_statat(l, SCARG(uap, fd), SCARG(uap, path), 333 SCARG(uap, flag), &st); 334 if (error != 0) 335 return error; 336 337 error = bsd_to_linux_statx(&st, &stx, SCARG(uap, mask)); 338 if (error != 0) 339 return error; 340 341 return copyout(&stx, SCARG(uap, sp), sizeof stx); 342 } 343 #endif /* LINUX_SYS_statx */ 344 345 #ifndef __alpha__ 346 int 347 linux_sys_truncate64(struct lwp *l, const struct linux_sys_truncate64_args *uap, register_t *retval) 348 { 349 /* { 350 syscallarg(const char *) path; 351 syscallarg(off_t) length; 352 } */ 353 struct sys_truncate_args ta; 354 355 /* Linux doesn't have the 'pad' pseudo-parameter */ 356 memset(&ta, 0, sizeof(ta)); 357 SCARG(&ta, path) = SCARG(uap, path); 358 SCARG(&ta, PAD) = 0; 359 SCARG(&ta, length) = SCARG(uap, length); 360 361 return sys_truncate(l, &ta, retval); 362 } 363 364 int 365 linux_sys_ftruncate64(struct lwp *l, const struct linux_sys_ftruncate64_args *uap, register_t *retval) 366 { 367 /* { 368 syscallarg(unsigned int) fd; 369 syscallarg(off_t) length; 370 } */ 371 struct sys_ftruncate_args ta; 372 373 /* Linux doesn't have the 'pad' pseudo-parameter */ 374 memset(&ta, 0, sizeof(ta)); 375 SCARG(&ta, fd) = SCARG(uap, fd); 376 SCARG(&ta, PAD) = 0; 377 SCARG(&ta, length) = SCARG(uap, length); 378 379 return sys_ftruncate(l, &ta, retval); 380 } 381 #endif /* __alpha__ */ 382 383 /* 384 * Linux 'readdir' call. This code is mostly taken from the 385 * SunOS getdents call (see compat/sunos/sunos_misc.c), though 386 * an attempt has been made to keep it a little cleaner. 387 * 388 * The d_off field contains the offset of the next valid entry, 389 * unless the older Linux getdents(2), which used to have it set 390 * to the offset of the entry itself. This function also doesn't 391 * need to deal with the old count == 1 glibc problem. 392 * 393 * Read in BSD-style entries, convert them, and copy them out. 394 * 395 * Note that this doesn't handle union-mounted filesystems. 396 */ 397 int 398 linux_sys_getdents64(struct lwp *l, const struct linux_sys_getdents64_args *uap, register_t *retval) 399 { 400 /* { 401 syscallarg(int) fd; 402 syscallarg(struct linux_dirent64 *) dent; 403 syscallarg(unsigned int) count; 404 } */ 405 struct dirent *bdp; 406 struct vnode *vp; 407 char *inp, *tbuf; /* BSD-format */ 408 int len, reclen; /* BSD-format */ 409 char *outp; /* Linux-format */ 410 int resid, linux_reclen = 0; /* Linux-format */ 411 file_t *fp; 412 struct uio auio; 413 struct iovec aiov; 414 struct linux_dirent64 idb; 415 off_t off; /* true file offset */ 416 int buflen, error, eofflag, nbytes; 417 struct vattr va; 418 off_t *cookiebuf = NULL, *cookie; 419 int ncookies; 420 421 /* fd_getvnode() will use the descriptor for us */ 422 if ((error = fd_getvnode(SCARG(uap, fd), &fp)) != 0) 423 return (error); 424 425 if ((fp->f_flag & FREAD) == 0) { 426 error = EBADF; 427 goto out1; 428 } 429 430 vp = (struct vnode *)fp->f_data; 431 if (vp->v_type != VDIR) { 432 error = ENOTDIR; 433 goto out1; 434 } 435 436 vn_lock(vp, LK_SHARED | LK_RETRY); 437 error = VOP_GETATTR(vp, &va, l->l_cred); 438 VOP_UNLOCK(vp); 439 if (error) 440 goto out1; 441 442 nbytes = SCARG(uap, count); 443 buflen = uimin(MAXBSIZE, nbytes); 444 if (buflen < va.va_blocksize) 445 buflen = va.va_blocksize; 446 tbuf = malloc(buflen, M_TEMP, M_WAITOK); 447 448 vn_lock(vp, LK_EXCLUSIVE | LK_RETRY); 449 off = fp->f_offset; 450 again: 451 aiov.iov_base = tbuf; 452 aiov.iov_len = buflen; 453 auio.uio_iov = &aiov; 454 auio.uio_iovcnt = 1; 455 auio.uio_rw = UIO_READ; 456 auio.uio_resid = buflen; 457 auio.uio_offset = off; 458 UIO_SETUP_SYSSPACE(&auio); 459 /* 460 * First we read into the malloc'ed buffer, then 461 * we massage it into user space, one record at a time. 462 */ 463 error = VOP_READDIR(vp, &auio, fp->f_cred, &eofflag, &cookiebuf, 464 &ncookies); 465 if (error) 466 goto out; 467 468 inp = tbuf; 469 outp = (void *)SCARG(uap, dent); 470 resid = nbytes; 471 if ((len = buflen - auio.uio_resid) == 0) 472 goto eof; 473 474 for (cookie = cookiebuf; len > 0; len -= reclen) { 475 bdp = (struct dirent *)inp; 476 reclen = bdp->d_reclen; 477 if (reclen & 3) { 478 error = EIO; 479 goto out; 480 } 481 if (bdp->d_fileno == 0) { 482 inp += reclen; /* it is a hole; squish it out */ 483 if (cookie) 484 off = *cookie++; 485 else 486 off += reclen; 487 continue; 488 } 489 linux_reclen = LINUX_RECLEN(&idb, bdp->d_namlen); 490 if (reclen > len || resid < linux_reclen) { 491 /* entry too big for buffer, so just stop */ 492 outp++; 493 break; 494 } 495 if (cookie) 496 off = *cookie++; /* each entry points to next */ 497 else 498 off += reclen; 499 /* 500 * Massage in place to make a Linux-shaped dirent (otherwise 501 * we have to worry about touching user memory outside of 502 * the copyout() call). 503 */ 504 memset(&idb, 0, sizeof(idb)); 505 idb.d_ino = bdp->d_fileno; 506 idb.d_type = bdp->d_type; 507 idb.d_off = off; 508 idb.d_reclen = (u_short)linux_reclen; 509 memcpy(idb.d_name, bdp->d_name, MIN(sizeof(idb.d_name), 510 bdp->d_namlen + 1)); 511 if ((error = copyout((void *)&idb, outp, linux_reclen))) 512 goto out; 513 /* advance past this real entry */ 514 inp += reclen; 515 /* advance output past Linux-shaped entry */ 516 outp += linux_reclen; 517 resid -= linux_reclen; 518 } 519 520 /* if we squished out the whole block, try again */ 521 if (outp == (void *)SCARG(uap, dent)) { 522 if (cookiebuf) 523 free(cookiebuf, M_TEMP); 524 cookiebuf = NULL; 525 goto again; 526 } 527 fp->f_offset = off; /* update the vnode offset */ 528 529 eof: 530 *retval = nbytes - resid; 531 out: 532 VOP_UNLOCK(vp); 533 if (cookiebuf) 534 free(cookiebuf, M_TEMP); 535 free(tbuf, M_TEMP); 536 out1: 537 fd_putfile(SCARG(uap, fd)); 538 return error; 539 } 540