Home | History | Annotate | Line # | Download | only in kernel
      1 /*	$NetBSD: t_fdpass.c,v 1.2 2026/10/01 22:24:25 riastradh Exp $	*/
      2 
      3 /*-
      4  * Copyright (c) 2026 The NetBSD Foundation, Inc.
      5  * All rights reserved.
      6  *
      7  * Redistribution and use in source and binary forms, with or without
      8  * modification, are permitted provided that the following conditions
      9  * are met:
     10  * 1. Redistributions of source code must retain the above copyright
     11  *    notice, this list of conditions and the following disclaimer.
     12  * 2. Redistributions in binary form must reproduce the above copyright
     13  *    notice, this list of conditions and the following disclaimer in the
     14  *    documentation and/or other materials provided with the distribution.
     15  *
     16  * THIS SOFTWARE IS PROVIDED BY THE NETBSD FOUNDATION, INC. AND CONTRIBUTORS
     17  * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
     18  * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
     19  * PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL THE FOUNDATION OR CONTRIBUTORS
     20  * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
     21  * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
     22  * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
     23  * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
     24  * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
     25  * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
     26  * POSSIBILITY OF SUCH DAMAGE.
     27  */
     28 
     29 #include <sys/cdefs.h>
     30 __RCSID("$NetBSD: t_fdpass.c,v 1.2 2026/10/01 22:24:25 riastradh Exp $");
     31 
     32 #include <sys/param.h>		/* needed by sys/mbuf.h */
     33 
     34 #include <sys/socket.h>
     35 
     36 #include <atf-c.h>
     37 #include <errno.h>
     38 #include <fcntl.h>
     39 
     40 #include <rump/rump.h>
     41 #include <rump/rump_syscalls.h>
     42 
     43 #include "h_macros.h"
     44 
     45 /*
     46  * XXX We include <sys/mbuf.h> last to avoid conflict with atf-c.h over
     47  * m_type.
     48  */
     49 #include <sys/mbuf.h>		/* MHLEN */
     50 
     51 static unsigned
     52 countfds(void)
     53 {
     54 	int fd, nfds, maxfd;
     55 
     56 	RL(maxfd = rump_sys_fcntl(-1, F_MAXFD));
     57 	for (fd = nfds = 0; fd <= maxfd; fd++) {
     58 		if (rump_sys_fcntl(fd, F_GETFL) != -1)
     59 			nfds++;
     60 	}
     61 
     62 	return nfds;
     63 }
     64 
     65 enum {
     66 	MAXNFDS = 32,
     67 };
     68 
     69 static void
     70 test_pr60832(const int fds[static 0], unsigned nfds)
     71 {
     72 	static char buf[4096];
     73 	int sock[2];
     74 	socklen_t optlen;
     75 	int sndbuf;
     76 	unsigned resid, clen, nfds_before, nfds_after;
     77 	ssize_t nsent, nrcvd;
     78 	size_t total = 0;
     79 	bool sentfds;
     80 
     81 	/*
     82 	 * Create a socket pair, nonblocking so that we fail promptly
     83 	 * when buffers are full instead of hanging until timeout.
     84 	 */
     85 	RL(rump_sys_socketpair(AF_LOCAL, SOCK_STREAM|SOCK_NONBLOCK, 0, sock));
     86 
     87 	/*
     88 	 * Count the number of file descriptors in this process so we
     89 	 * make sure we receive the right total number of them at the
     90 	 * end.
     91 	 */
     92 	nfds_before = countfds();
     93 
     94 	/*
     95 	 * Find how much space is required for the control message.
     96 	 *
     97 	 * Note: We use CMSG_SPACE for struct msghdr::msg_controllen;
     98 	 * for struct cmsghdr::cmsg_len below, we will use CMSG_LEN.
     99 	 * Confused?  Thank the cmsg(3) API designers...
    100 	 */
    101 	ATF_REQUIRE(nfds <= MAXNFDS);
    102 	clen = CMSG_SPACE(nfds * sizeof(int));
    103 
    104 	/*
    105 	 * Find out how much we can write to the socket (SO_SNDBUF)
    106 	 * without blocking.
    107 	 *
    108 	 * XXX Why can we still write after sndbuf - sndlowat bytes?
    109 	 */
    110 	optlen = sizeof(sndbuf);
    111 	RL(rump_sys_getsockopt(sock[0], SOL_SOCKET, SO_SNDBUF, &sndbuf,
    112 		&optlen));
    113 	ATF_REQUIRE_MSG(optlen == sizeof(sndbuf),
    114 	    "optlen=%zu sizeof(sndbuf)=%zu",
    115 	    (size_t)optlen, sizeof(sndbuf));
    116 
    117 	printf("sndbuf=%d\n", sndbuf);
    118 	ATF_REQUIRE(sndbuf > 0);
    119 	ATF_REQUIRE((unsigned)sndbuf <= __type_max(__typeof(resid)));
    120 
    121 	/*
    122 	 * Fill the buffer until we have only MHLEN bytes left.  Count
    123 	 * how many bytes we have sent total as we go.
    124 	 *
    125 	 * XXX Seems like this should really go until we have sndlowat
    126 	 * bytes left.
    127 	 */
    128 	total = 0;
    129 	for (resid = sndbuf;
    130 	     resid > (size_t)MHLEN + clen;
    131 	     resid -= (size_t)nsent) {
    132 		const ssize_t nsend = MIN(resid - ((size_t)MHLEN + clen),
    133 		    sizeof(buf));
    134 		struct iovec iov[] = {
    135 			{ .iov_base = buf, .iov_len = nsend },
    136 		};
    137 		const struct msghdr msg = {
    138 			.msg_name = NULL,
    139 			.msg_namelen = 0,
    140 			.msg_iov = iov,
    141 			.msg_iovlen = __arraycount(iov),
    142 			.msg_control = NULL,
    143 			.msg_controllen = 0,
    144 			.msg_flags = 0,
    145 		};
    146 
    147 		printf("sendmsg, resid=%u nsend=%zd\n",
    148 		    resid, nsend);
    149 		ATF_REQUIRE(nsend > 0);
    150 		RL(nsent = rump_sys_sendmsg(sock[0], &msg, 0));
    151 		printf("sent %zd of %zd\n", nsent, nsend);
    152 		ATF_REQUIRE_MSG(nsent == nsend, "nsent=%zd nsend=%zd",
    153 		    nsent, nsend);
    154 		total += (size_t)nsent;
    155 	}
    156 	ATF_REQUIRE_MSG(resid == (size_t)MHLEN + clen, "resid=%u", resid);
    157 
    158 	/*
    159 	 * Try to fill the remainder of the buffer, but with file
    160 	 * descriptors as well.  The control message length should
    161 	 * count toward sndbuf as well as the data length.
    162 	 */
    163     {
    164 	union {
    165 		struct cmsghdr	hdr;
    166 		unsigned char	buf[CMSG_SPACE(MAXNFDS * sizeof(int))];
    167 	} cmsgbuf;
    168 	const ssize_t nsend = resid - clen;
    169 	struct iovec iov[] = {
    170 		{ .iov_base = buf, .iov_len = nsend }
    171 	};
    172 	const struct msghdr msg = {
    173 		.msg_name = NULL,
    174 		.msg_namelen = 0,
    175 		.msg_iov = iov,
    176 		.msg_iovlen = __arraycount(iov),
    177 		.msg_control = &cmsgbuf,
    178 		.msg_controllen = clen,
    179 		.msg_flags = 0,
    180 	};
    181 	struct cmsghdr *const cmsg = CMSG_FIRSTHDR(&msg);
    182 
    183 	cmsg->cmsg_len = CMSG_LEN(nfds * sizeof(int));
    184 	cmsg->cmsg_level = SOL_SOCKET;
    185 	cmsg->cmsg_type = SCM_RIGHTS;
    186 	memcpy(CMSG_DATA(cmsg), fds, nfds * sizeof(int));
    187 
    188 	/*
    189 	 * If the kernel uses an intermediate data structure for the
    190 	 * control message that is larger than the user's buffer, we
    191 	 * might pass the blocking check (so no EAGAIN) but then fail
    192 	 * with ENOBUFS a little downstream.
    193 	 *
    194 	 * XXX This is suboptimal!  For example, there is no way to
    195 	 * block until the buffer space is available.  Perhaps the
    196 	 * kernel should account only the user's buffer size, not the
    197 	 * kernel's intermediate buffer size, in the path that fails
    198 	 * with ENOBUFS?
    199 	 *
    200 	 * The obvious alternative, of counting the kernel's
    201 	 * intermediate buffer size in the path that decides whether to
    202 	 * block, has two problems:
    203 	 *
    204 	 * 1. it crosses abstraction layers (blocking path is in
    205 	 *    AF-generic logic in uipc_socket.c, treats control message
    206 	 *    as opaque),
    207 	 *
    208 	 * 2. it would mean that the caller can't count up to sndbuf by
    209 	 *    adding the data and control byte counts, because the
    210 	 *    user's control byte count might be lower than the
    211 	 *    kernel's control byte count.
    212 	 *
    213 	 * In any case, if we fix this so that the backpressure is
    214 	 * applied in the blocking path and cannot lead to ENOBUFS,
    215 	 * then we can tighten this to simply
    216 	 *
    217 	 *	RL(nsent = rump_sys_sendmsg(sock[0], &msg, 0));
    218 	 *
    219 	 * and assert that we always sent the fds.
    220 	 */
    221 	printf("send %zd data bytes with %u control bytes\n", nsend, clen);
    222 	nsent = rump_sys_sendmsg(sock[0], &msg, 0);
    223 	if (nsent == -1) {
    224 		int error = errno;
    225 
    226 		ATF_CHECK_EQ_MSG(error, ENOBUFS, "error=%d (%s)", error,
    227 		    strerror(errno));
    228 		sentfds = false;
    229 	} else {
    230 		sentfds = true;
    231 		total += (size_t)nsent;
    232 	}
    233     }
    234 
    235 	/*
    236 	 * Receive the data and file descriptors.  In some chunk
    237 	 * (though it may not be the last chunk, because the last chunk
    238 	 * may be split into smaller chunks), we should receive the
    239 	 * fds.
    240 	 */
    241 	for (resid = total; resid > 0; resid -= (size_t)nrcvd) {
    242 		union {
    243 			struct cmsghdr	hdr;
    244 			unsigned char	buf[CMSG_SPACE(MAXNFDS * sizeof(int))];
    245 		} cmsgbuf;
    246 		const ssize_t nrecv = sizeof(buf);
    247 		struct iovec iov[] = {
    248 			{ .iov_base = buf, .iov_len = nrecv }
    249 		};
    250 		struct msghdr msg = {
    251 			.msg_name = NULL,
    252 			.msg_namelen = 0,
    253 			.msg_iov = iov,
    254 			.msg_iovlen = __arraycount(iov),
    255 			.msg_control = &cmsgbuf,
    256 			.msg_controllen = sizeof(cmsgbuf),
    257 			.msg_flags = 0,
    258 		};
    259 		struct cmsghdr *cmsg;
    260 
    261 		printf("recvmsg, resid=%u nrecv=%zu\n", resid, nrecv);
    262 		nrcvd = rump_sys_recvmsg(sock[1], &msg, 0);
    263 		if (nrcvd == -1) {
    264 			int error = errno;
    265 
    266 			atf_tc_fail_nonfatal("recvmsg: %d (%s)", error,
    267 			    strerror(error));
    268 			break;
    269 		}
    270 		printf("nrcvd=%zu\n", nrcvd);
    271 		ATF_REQUIRE_MSG((size_t)nrcvd <= resid, "nrcvd=%zd resid=%u",
    272 		    nrcvd, resid);
    273 		for (cmsg = CMSG_FIRSTHDR(&msg);
    274 		     cmsg != NULL;
    275 		     cmsg = CMSG_NXTHDR(&msg, cmsg)) {
    276 			unsigned i, n;
    277 			const int *fdptr;
    278 
    279 			if (cmsg->cmsg_level != SOL_SOCKET) {
    280 				atf_tc_fail_nonfatal("cmsg_level=%d,"
    281 				    " expected %d\n",
    282 				    cmsg->cmsg_level, SOL_SOCKET);
    283 				continue;
    284 			}
    285 			if (cmsg->cmsg_type != SCM_RIGHTS) {
    286 				atf_tc_fail_nonfatal("cmsg_type=%d,"
    287 				    " expected %d\n",
    288 				    cmsg->cmsg_type, SCM_RIGHTS);
    289 				continue;
    290 			}
    291 			if (cmsg->cmsg_len < CMSG_LEN(sizeof(int))) {
    292 				atf_tc_fail_nonfatal("cmsg_len=%zu,"
    293 				    " expected >=%zu\n",
    294 				    (size_t)cmsg->cmsg_len,
    295 				    (size_t)CMSG_LEN(sizeof(int)));
    296 				continue;
    297 			}
    298 			if ((cmsg->cmsg_len - CMSG_LEN(0)) % sizeof(int)) {
    299 				atf_tc_fail_nonfatal("cmsg_len=%zu,"
    300 				    " expected 0 mod %zu after %zu\n",
    301 				    (size_t)cmsg->cmsg_len,
    302 				    (size_t)sizeof(int),
    303 				    (size_t)CMSG_LEN(0));
    304 				continue;
    305 			}
    306 			n = (cmsg->cmsg_len - CMSG_LEN(0))/sizeof(int);
    307 			if (sentfds)
    308 				ATF_CHECK(n > 0);
    309 			else
    310 				ATF_CHECK_MSG(n == 0, "n=%u", n);
    311 			fdptr = (const int *)CMSG_DATA(cmsg);
    312 			for (i = 0; i < n; i++)
    313 				RL(rump_sys_close(fdptr[i]));
    314 		}
    315 	}
    316 
    317 	/*
    318 	 * For one final recvmsg, we should block.
    319 	 */
    320     {
    321 	union {
    322 		struct cmsghdr	hdr;
    323 		unsigned char	buf[CMSG_SPACE(MAXNFDS * sizeof(int))];
    324 	} cmsgbuf;
    325 	struct iovec iov[] = {
    326 		{ .iov_base = buf, .iov_len = sizeof(buf) }
    327 	};
    328 	struct msghdr msg = {
    329 		.msg_name = NULL,
    330 		.msg_namelen = 0,
    331 		.msg_iov = iov,
    332 		.msg_iovlen = __arraycount(iov),
    333 		.msg_control = &cmsgbuf,
    334 		.msg_controllen = sizeof(cmsgbuf),
    335 		.msg_flags = 0,
    336 	};
    337 
    338 	ATF_CHECK_ERRNO(EAGAIN, rump_sys_recvmsg(sock[1], &msg, 0) == -1);
    339     }
    340 
    341 	/*
    342 	 * Verify the total number of file descriptors has not changed.
    343 	 * (We closed all the copies we sent ourself in the recvmsg
    344 	 * loop above.)
    345 	 */
    346 	nfds_after = countfds();
    347 	ATF_CHECK_MSG(nfds_before == nfds_after,
    348 	    "nfds_before=%u nfds_after=%u", nfds_before, nfds_after);
    349 
    350 	RL(rump_sys_close(sock[0]));
    351 	RL(rump_sys_close(sock[1]));
    352 }
    353 
    354 ATF_TC(pr60832);
    355 ATF_TC_HEAD(pr60832, tc)
    356 {
    357 	atf_tc_set_md_var(tc, "descr", "Test sendmsg fails on partial data");
    358 }
    359 ATF_TC_BODY(pr60832, tc)
    360 {
    361 	int fds[MAXNFDS];
    362 	unsigned i;
    363 
    364 	REQUIRE_LIBC(setlinebuf(stdout), EOF);
    365 	printf("MHLEN=%d\n", MHLEN);
    366 
    367 	/*
    368 	 * Initialize rump before we try using it.
    369 	 */
    370 	RL(rump_init());
    371 
    372 	/*
    373 	 * Create some spare file descriptors.
    374 	 */
    375 	for (i = 0; i < __arraycount(fds); i++)
    376 		RL(fds[i] = rump_sys_socket(AF_LOCAL, SOCK_STREAM, 0));
    377 
    378 	/*
    379 	 * Run the test with different numbers of file descriptors.
    380 	 */
    381 	for (i = 0; i < __arraycount(fds); i++) {
    382 		const unsigned nfds = i + 1;
    383 
    384 		printf("test %u fd%s\n", nfds, nfds == 1 ? "" : "s");
    385 		test_pr60832(fds, nfds);
    386 	}
    387 }
    388 
    389 ATF_TP_ADD_TCS(tp)
    390 {
    391 
    392 	ATF_TP_ADD_TC(tp, pr60832);
    393 
    394 	return atf_no_error();
    395 }
    396