Home | History | Annotate | Line # | Download | only in kern
sys_eventfd.c revision 1.11
      1  1.11  riastrad /*	$NetBSD: sys_eventfd.c,v 1.11 2023/11/19 17:16:00 riastradh Exp $	*/
      2   1.2   thorpej 
      3   1.2   thorpej /*-
      4   1.2   thorpej  * Copyright (c) 2020 The NetBSD Foundation, Inc.
      5   1.2   thorpej  * All rights reserved.
      6   1.2   thorpej  *
      7   1.2   thorpej  * This code is derived from software contributed to The NetBSD Foundation
      8   1.2   thorpej  * by Jason R. Thorpe.
      9   1.2   thorpej  *
     10   1.2   thorpej  * Redistribution and use in source and binary forms, with or without
     11   1.2   thorpej  * modification, are permitted provided that the following conditions
     12   1.2   thorpej  * are met:
     13   1.2   thorpej  * 1. Redistributions of source code must retain the above copyright
     14   1.2   thorpej  *    notice, this list of conditions and the following disclaimer.
     15   1.2   thorpej  * 2. Redistributions in binary form must reproduce the above copyright
     16   1.2   thorpej  *    notice, this list of conditions and the following disclaimer in the
     17   1.2   thorpej  *    documentation and/or other materials provided with the distribution.
     18   1.2   thorpej  *
     19   1.2   thorpej  * THIS SOFTWARE IS PROVIDED BY THE NETBSD FOUNDATION, INC. AND CONTRIBUTORS
     20   1.2   thorpej  * ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
     21   1.2   thorpej  * TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
     22   1.2   thorpej  * PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL THE FOUNDATION OR CONTRIBUTORS
     23   1.2   thorpej  * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
     24   1.2   thorpej  * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
     25   1.2   thorpej  * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
     26   1.2   thorpej  * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
     27   1.2   thorpej  * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
     28   1.2   thorpej  * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
     29   1.2   thorpej  * POSSIBILITY OF SUCH DAMAGE.
     30   1.2   thorpej  */
     31   1.2   thorpej 
     32   1.2   thorpej #include <sys/cdefs.h>
     33  1.11  riastrad __KERNEL_RCSID(0, "$NetBSD: sys_eventfd.c,v 1.11 2023/11/19 17:16:00 riastradh Exp $");
     34   1.2   thorpej 
     35   1.2   thorpej /*
     36   1.2   thorpej  * eventfd
     37   1.2   thorpej  *
     38   1.2   thorpej  * Eventfd objects present a simple counting object associated with a
     39   1.2   thorpej  * file descriptor.  Writes and reads to this file descriptor increment
     40   1.2   thorpej  * and decrement the count, respectively.  When the count is non-zero,
     41   1.2   thorpej  * the descriptor is considered "readable", and when less than the max
     42   1.2   thorpej  * value (EVENTFD_MAXVAL), is considered "writable".
     43   1.2   thorpej  *
     44   1.2   thorpej  * This implementation is API compatible with the Linux eventfd(2)
     45   1.2   thorpej  * interface.
     46   1.2   thorpej  */
     47   1.2   thorpej 
     48   1.3     skrll #include <sys/param.h>
     49   1.2   thorpej #include <sys/types.h>
     50   1.2   thorpej #include <sys/condvar.h>
     51   1.2   thorpej #include <sys/eventfd.h>
     52   1.2   thorpej #include <sys/file.h>
     53   1.2   thorpej #include <sys/filedesc.h>
     54   1.2   thorpej #include <sys/kauth.h>
     55   1.2   thorpej #include <sys/mutex.h>
     56   1.2   thorpej #include <sys/poll.h>
     57   1.2   thorpej #include <sys/proc.h>
     58   1.2   thorpej #include <sys/select.h>
     59   1.2   thorpej #include <sys/stat.h>
     60   1.2   thorpej #include <sys/syscallargs.h>
     61   1.2   thorpej #include <sys/uio.h>
     62   1.2   thorpej 
     63   1.2   thorpej struct eventfd {
     64   1.2   thorpej 	kmutex_t	efd_lock;
     65   1.2   thorpej 	kcondvar_t	efd_read_wait;
     66   1.2   thorpej 	kcondvar_t	efd_write_wait;
     67   1.2   thorpej 	struct selinfo	efd_read_sel;
     68   1.2   thorpej 	struct selinfo	efd_write_sel;
     69   1.2   thorpej 	eventfd_t	efd_val;
     70   1.2   thorpej 	int64_t		efd_nwaiters;
     71   1.2   thorpej 	bool		efd_restarting;
     72   1.2   thorpej 	bool		efd_is_semaphore;
     73   1.2   thorpej 
     74   1.2   thorpej 	/*
     75   1.2   thorpej 	 * Information kept for stat(2).
     76   1.2   thorpej 	 */
     77   1.2   thorpej 	struct timespec efd_btime;	/* time created */
     78   1.2   thorpej 	struct timespec	efd_mtime;	/* last write */
     79   1.2   thorpej 	struct timespec	efd_atime;	/* last read */
     80   1.2   thorpej };
     81   1.2   thorpej 
     82   1.2   thorpej #define	EVENTFD_MAXVAL	(UINT64_MAX - 1)
     83   1.2   thorpej 
     84   1.2   thorpej /*
     85   1.2   thorpej  * eventfd_create:
     86   1.2   thorpej  *
     87   1.2   thorpej  *	Create an eventfd object.
     88   1.2   thorpej  */
     89   1.2   thorpej static struct eventfd *
     90   1.2   thorpej eventfd_create(unsigned int const val, int const flags)
     91   1.2   thorpej {
     92   1.2   thorpej 	struct eventfd * const efd = kmem_zalloc(sizeof(*efd), KM_SLEEP);
     93   1.2   thorpej 
     94   1.2   thorpej 	mutex_init(&efd->efd_lock, MUTEX_DEFAULT, IPL_NONE);
     95   1.2   thorpej 	cv_init(&efd->efd_read_wait, "efdread");
     96   1.2   thorpej 	cv_init(&efd->efd_write_wait, "efdwrite");
     97   1.2   thorpej 	selinit(&efd->efd_read_sel);
     98   1.2   thorpej 	selinit(&efd->efd_write_sel);
     99   1.2   thorpej 	efd->efd_val = val;
    100   1.2   thorpej 	efd->efd_is_semaphore = !!(flags & EFD_SEMAPHORE);
    101   1.2   thorpej 	getnanotime(&efd->efd_btime);
    102   1.2   thorpej 
    103   1.2   thorpej 	/* Caller deals with EFD_CLOEXEC and EFD_NONBLOCK. */
    104   1.2   thorpej 
    105   1.2   thorpej 	return efd;
    106   1.2   thorpej }
    107   1.2   thorpej 
    108   1.2   thorpej /*
    109   1.2   thorpej  * eventfd_destroy:
    110   1.2   thorpej  *
    111   1.2   thorpej  *	Destroy an eventfd object.
    112   1.2   thorpej  */
    113   1.2   thorpej static void
    114   1.2   thorpej eventfd_destroy(struct eventfd * const efd)
    115   1.2   thorpej {
    116   1.2   thorpej 
    117   1.2   thorpej 	KASSERT(efd->efd_nwaiters == 0);
    118   1.2   thorpej 
    119   1.2   thorpej 	cv_destroy(&efd->efd_read_wait);
    120   1.2   thorpej 	cv_destroy(&efd->efd_write_wait);
    121   1.2   thorpej 
    122   1.2   thorpej 	seldestroy(&efd->efd_read_sel);
    123   1.2   thorpej 	seldestroy(&efd->efd_write_sel);
    124   1.2   thorpej 
    125   1.2   thorpej 	mutex_destroy(&efd->efd_lock);
    126   1.4   thorpej 
    127   1.4   thorpej 	kmem_free(efd, sizeof(*efd));
    128   1.2   thorpej }
    129   1.2   thorpej 
    130   1.2   thorpej /*
    131   1.2   thorpej  * eventfd_wait:
    132   1.2   thorpej  *
    133   1.2   thorpej  *	Block on an eventfd.  Handles non-blocking, as well as
    134   1.2   thorpej  *	the restart cases.
    135   1.2   thorpej  */
    136   1.2   thorpej static int
    137   1.2   thorpej eventfd_wait(struct eventfd * const efd, int const fflag, bool const is_write)
    138   1.2   thorpej {
    139   1.2   thorpej 	kcondvar_t *waitcv;
    140   1.2   thorpej 	int error;
    141   1.2   thorpej 
    142   1.2   thorpej 	if (fflag & FNONBLOCK) {
    143   1.2   thorpej 		return EAGAIN;
    144   1.2   thorpej 	}
    145   1.2   thorpej 
    146   1.2   thorpej 	/*
    147   1.8   thorpej 	 * We're going to block.  Check if we need to return ERESTART.
    148   1.2   thorpej 	 */
    149   1.8   thorpej 	if (efd->efd_restarting) {
    150   1.8   thorpej 		return ERESTART;
    151   1.2   thorpej 	}
    152   1.2   thorpej 
    153   1.2   thorpej 	if (is_write) {
    154   1.2   thorpej 		waitcv = &efd->efd_write_wait;
    155   1.2   thorpej 	} else {
    156   1.2   thorpej 		waitcv = &efd->efd_read_wait;
    157   1.2   thorpej 	}
    158   1.2   thorpej 
    159   1.2   thorpej 	efd->efd_nwaiters++;
    160   1.2   thorpej 	KASSERT(efd->efd_nwaiters > 0);
    161   1.2   thorpej 	error = cv_wait_sig(waitcv, &efd->efd_lock);
    162   1.2   thorpej 	efd->efd_nwaiters--;
    163   1.2   thorpej 	KASSERT(efd->efd_nwaiters >= 0);
    164   1.2   thorpej 
    165   1.2   thorpej 	/*
    166   1.2   thorpej 	 * If a restart was triggered while we were asleep, we need
    167   1.8   thorpej 	 * to return ERESTART if no other error was returned.
    168   1.2   thorpej 	 */
    169   1.2   thorpej 	if (efd->efd_restarting) {
    170   1.2   thorpej 		if (error == 0) {
    171   1.2   thorpej 			error = ERESTART;
    172   1.2   thorpej 		}
    173   1.2   thorpej 	}
    174   1.2   thorpej 
    175   1.2   thorpej 	return error;
    176   1.2   thorpej }
    177   1.2   thorpej 
    178   1.2   thorpej /*
    179   1.2   thorpej  * eventfd_wake:
    180   1.2   thorpej  *
    181   1.2   thorpej  *	Wake LWPs block on an eventfd.
    182   1.2   thorpej  */
    183   1.2   thorpej static void
    184   1.2   thorpej eventfd_wake(struct eventfd * const efd, bool const is_write)
    185   1.2   thorpej {
    186   1.2   thorpej 	kcondvar_t *waitcv = NULL;
    187   1.2   thorpej 	struct selinfo *sel;
    188   1.2   thorpej 	int pollev;
    189   1.2   thorpej 
    190   1.2   thorpej 	if (is_write) {
    191  1.10  riastrad 		waitcv = &efd->efd_read_wait;
    192   1.2   thorpej 		sel = &efd->efd_read_sel;
    193   1.2   thorpej 		pollev = POLLIN | POLLRDNORM;
    194   1.2   thorpej 	} else {
    195  1.10  riastrad 		waitcv = &efd->efd_write_wait;
    196   1.2   thorpej 		sel = &efd->efd_write_sel;
    197   1.2   thorpej 		pollev = POLLOUT | POLLWRNORM;
    198   1.2   thorpej 	}
    199  1.11  riastrad 	cv_broadcast(waitcv);
    200   1.2   thorpej 	selnotify(sel, pollev, NOTE_SUBMIT);
    201   1.2   thorpej }
    202   1.2   thorpej 
    203   1.2   thorpej /*
    204   1.2   thorpej  * eventfd file operations
    205   1.2   thorpej  */
    206   1.2   thorpej 
    207   1.2   thorpej static int
    208   1.2   thorpej eventfd_fop_read(file_t * const fp, off_t * const offset,
    209   1.2   thorpej     struct uio * const uio, kauth_cred_t const cred, int const flags)
    210   1.2   thorpej {
    211   1.2   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    212   1.2   thorpej 	int const fflag = fp->f_flag;
    213   1.2   thorpej 	eventfd_t return_value;
    214   1.2   thorpej 	int error;
    215   1.2   thorpej 
    216   1.2   thorpej 	if (uio->uio_resid < sizeof(eventfd_t)) {
    217   1.2   thorpej 		return EINVAL;
    218   1.2   thorpej 	}
    219   1.2   thorpej 
    220   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    221   1.2   thorpej 
    222   1.2   thorpej 	while (efd->efd_val == 0) {
    223   1.2   thorpej 		if ((error = eventfd_wait(efd, fflag, false)) != 0) {
    224   1.2   thorpej 			mutex_exit(&efd->efd_lock);
    225   1.2   thorpej 			return error;
    226   1.2   thorpej 		}
    227   1.2   thorpej 	}
    228   1.2   thorpej 
    229   1.2   thorpej 	if (efd->efd_is_semaphore) {
    230   1.2   thorpej 		return_value = 1;
    231   1.2   thorpej 		efd->efd_val--;
    232   1.2   thorpej 	} else {
    233   1.2   thorpej 		return_value = efd->efd_val;
    234   1.2   thorpej 		efd->efd_val = 0;
    235   1.2   thorpej 	}
    236   1.2   thorpej 
    237   1.2   thorpej 	getnanotime(&efd->efd_atime);
    238   1.2   thorpej 	eventfd_wake(efd, false);
    239   1.2   thorpej 
    240   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    241   1.2   thorpej 
    242   1.2   thorpej 	error = uiomove(&return_value, sizeof(return_value), uio);
    243   1.2   thorpej 
    244   1.2   thorpej 	return error;
    245   1.2   thorpej }
    246   1.2   thorpej 
    247   1.2   thorpej static int
    248   1.2   thorpej eventfd_fop_write(file_t * const fp, off_t * const offset,
    249   1.2   thorpej     struct uio * const uio, kauth_cred_t const cred, int const flags)
    250   1.2   thorpej {
    251   1.2   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    252   1.2   thorpej 	int const fflag = fp->f_flag;
    253   1.2   thorpej 	eventfd_t write_value;
    254   1.2   thorpej 	int error;
    255   1.2   thorpej 
    256   1.2   thorpej 	if (uio->uio_resid < sizeof(eventfd_t)) {
    257   1.2   thorpej 		return EINVAL;
    258   1.2   thorpej 	}
    259   1.2   thorpej 
    260   1.2   thorpej 	if ((error = uiomove(&write_value, sizeof(write_value), uio)) != 0) {
    261   1.2   thorpej 		return error;
    262   1.2   thorpej 	}
    263   1.2   thorpej 
    264   1.2   thorpej 	if (write_value > EVENTFD_MAXVAL) {
    265   1.2   thorpej 		error = EINVAL;
    266   1.2   thorpej 		goto out;
    267   1.2   thorpej 	}
    268   1.2   thorpej 
    269   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    270   1.2   thorpej 
    271   1.2   thorpej 	KASSERT(efd->efd_val <= EVENTFD_MAXVAL);
    272   1.2   thorpej 	while ((EVENTFD_MAXVAL - efd->efd_val) < write_value) {
    273   1.2   thorpej 		if ((error = eventfd_wait(efd, fflag, true)) != 0) {
    274   1.2   thorpej 			mutex_exit(&efd->efd_lock);
    275   1.2   thorpej 			goto out;
    276   1.2   thorpej 		}
    277   1.2   thorpej 	}
    278   1.2   thorpej 
    279   1.2   thorpej 	efd->efd_val += write_value;
    280   1.2   thorpej 	KASSERT(efd->efd_val <= EVENTFD_MAXVAL);
    281   1.2   thorpej 
    282   1.2   thorpej 	getnanotime(&efd->efd_mtime);
    283   1.2   thorpej 	eventfd_wake(efd, true);
    284   1.2   thorpej 
    285   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    286   1.2   thorpej 
    287   1.2   thorpej  out:
    288   1.2   thorpej 	if (error) {
    289   1.2   thorpej 		/*
    290   1.2   thorpej 		 * Undo the effect of uiomove() so that the error
    291   1.2   thorpej 		 * gets reported correctly; see dofilewrite().
    292   1.2   thorpej 		 */
    293   1.2   thorpej 		uio->uio_resid += sizeof(write_value);
    294   1.2   thorpej 	}
    295   1.2   thorpej 	return error;
    296   1.2   thorpej }
    297   1.2   thorpej 
    298   1.2   thorpej static int
    299   1.9   thorpej eventfd_ioctl(file_t * const fp, u_long const cmd, void * const data)
    300   1.9   thorpej {
    301   1.9   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    302   1.9   thorpej 
    303   1.9   thorpej 	switch (cmd) {
    304   1.9   thorpej 	case FIONBIO:
    305   1.9   thorpej 		return 0;
    306   1.9   thorpej 
    307   1.9   thorpej 	case FIONREAD:
    308   1.9   thorpej 		mutex_enter(&efd->efd_lock);
    309   1.9   thorpej 		*(int *)data = efd->efd_val != 0 ? sizeof(eventfd_t) : 0;
    310   1.9   thorpej 		mutex_exit(&efd->efd_lock);
    311   1.9   thorpej 		return 0;
    312   1.9   thorpej 
    313   1.9   thorpej 	case FIONWRITE:
    314   1.9   thorpej 		*(int *)data = 0;
    315   1.9   thorpej 		return 0;
    316   1.9   thorpej 
    317   1.9   thorpej 	case FIONSPACE:
    318   1.9   thorpej 		/*
    319   1.9   thorpej 		 * FIONSPACE doesn't really work for eventfd, because the
    320   1.9   thorpej 		 * writability depends on the contents (value) being written.
    321   1.9   thorpej 		 */
    322   1.9   thorpej 		break;
    323   1.9   thorpej 
    324   1.9   thorpej 	default:
    325   1.9   thorpej 		break;
    326   1.9   thorpej 	}
    327   1.9   thorpej 
    328   1.9   thorpej 	return EPASSTHROUGH;
    329   1.9   thorpej }
    330   1.9   thorpej 
    331   1.9   thorpej static int
    332   1.2   thorpej eventfd_fop_poll(file_t * const fp, int const events)
    333   1.2   thorpej {
    334   1.2   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    335   1.2   thorpej 	int revents = 0;
    336   1.2   thorpej 
    337   1.2   thorpej 	/*
    338   1.2   thorpej 	 * Note that Linux will return POLLERR if the eventfd count
    339   1.2   thorpej 	 * overflows, but that is not possible in the normal read/write
    340   1.2   thorpej 	 * API, only with Linux kernel-internal interfaces.  So, this
    341   1.2   thorpej 	 * implementation never returns POLLERR.
    342   1.2   thorpej 	 *
    343   1.2   thorpej 	 * Also note that the Linux eventfd(2) man page does not
    344   1.2   thorpej 	 * specifically discuss returning POLLRDNORM, but we check
    345   1.2   thorpej 	 * for that event in addition to POLLIN.
    346   1.2   thorpej 	 */
    347   1.2   thorpej 
    348   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    349   1.2   thorpej 
    350   1.2   thorpej 	if (events & (POLLIN | POLLRDNORM)) {
    351   1.2   thorpej 		if (efd->efd_val != 0) {
    352   1.2   thorpej 			revents |= events & (POLLIN | POLLRDNORM);
    353   1.2   thorpej 		} else {
    354   1.2   thorpej 			selrecord(curlwp, &efd->efd_read_sel);
    355   1.2   thorpej 		}
    356   1.2   thorpej 	}
    357   1.2   thorpej 
    358   1.2   thorpej 	if (events & (POLLOUT | POLLWRNORM)) {
    359   1.2   thorpej 		if (efd->efd_val < EVENTFD_MAXVAL) {
    360   1.2   thorpej 			revents |= events & (POLLOUT | POLLWRNORM);
    361   1.2   thorpej 		} else {
    362   1.2   thorpej 			selrecord(curlwp, &efd->efd_write_sel);
    363   1.2   thorpej 		}
    364   1.2   thorpej 	}
    365   1.2   thorpej 
    366   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    367   1.2   thorpej 
    368   1.2   thorpej 	return revents;
    369   1.2   thorpej }
    370   1.2   thorpej 
    371   1.2   thorpej static int
    372   1.2   thorpej eventfd_fop_stat(file_t * const fp, struct stat * const st)
    373   1.2   thorpej {
    374   1.2   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    375   1.2   thorpej 
    376   1.2   thorpej 	memset(st, 0, sizeof(*st));
    377   1.2   thorpej 
    378   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    379   1.2   thorpej 	st->st_size = (off_t)efd->efd_val;
    380   1.2   thorpej 	st->st_blksize = sizeof(eventfd_t);
    381   1.2   thorpej 	st->st_mode = S_IFIFO | S_IRUSR | S_IWUSR;
    382   1.2   thorpej 	st->st_blocks = 1;
    383   1.2   thorpej 	st->st_birthtimespec = st->st_ctimespec = efd->efd_btime;
    384   1.2   thorpej 	st->st_atimespec = efd->efd_atime;
    385   1.2   thorpej 	st->st_mtimespec = efd->efd_mtime;
    386   1.2   thorpej 	st->st_uid = kauth_cred_geteuid(fp->f_cred);
    387   1.2   thorpej 	st->st_gid = kauth_cred_getegid(fp->f_cred);
    388   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    389   1.2   thorpej 
    390   1.2   thorpej 	return 0;
    391   1.2   thorpej }
    392   1.2   thorpej 
    393   1.2   thorpej static int
    394   1.2   thorpej eventfd_fop_close(file_t * const fp)
    395   1.2   thorpej {
    396   1.2   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    397   1.2   thorpej 
    398   1.2   thorpej 	fp->f_eventfd = NULL;
    399   1.2   thorpej 	eventfd_destroy(efd);
    400   1.2   thorpej 
    401   1.2   thorpej 	return 0;
    402   1.2   thorpej }
    403   1.2   thorpej 
    404   1.2   thorpej static void
    405   1.2   thorpej eventfd_filt_read_detach(struct knote * const kn)
    406   1.2   thorpej {
    407   1.2   thorpej 	struct eventfd * const efd = ((file_t *)kn->kn_obj)->f_eventfd;
    408   1.2   thorpej 
    409   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    410   1.2   thorpej 	KASSERT(kn->kn_hook == efd);
    411   1.2   thorpej 	selremove_knote(&efd->efd_read_sel, kn);
    412   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    413   1.2   thorpej }
    414   1.2   thorpej 
    415   1.2   thorpej static int
    416   1.2   thorpej eventfd_filt_read(struct knote * const kn, long const hint)
    417   1.2   thorpej {
    418   1.2   thorpej 	struct eventfd * const efd = ((file_t *)kn->kn_obj)->f_eventfd;
    419   1.7   thorpej 	int rv;
    420   1.2   thorpej 
    421   1.2   thorpej 	if (hint & NOTE_SUBMIT) {
    422   1.2   thorpej 		KASSERT(mutex_owned(&efd->efd_lock));
    423   1.2   thorpej 	} else {
    424   1.2   thorpej 		mutex_enter(&efd->efd_lock);
    425   1.2   thorpej 	}
    426   1.2   thorpej 
    427   1.2   thorpej 	kn->kn_data = (int64_t)efd->efd_val;
    428   1.7   thorpej 	rv = (eventfd_t)kn->kn_data > 0;
    429   1.2   thorpej 
    430   1.2   thorpej 	if ((hint & NOTE_SUBMIT) == 0) {
    431   1.2   thorpej 		mutex_exit(&efd->efd_lock);
    432   1.2   thorpej 	}
    433   1.2   thorpej 
    434   1.7   thorpej 	return rv;
    435   1.2   thorpej }
    436   1.2   thorpej 
    437   1.2   thorpej static const struct filterops eventfd_read_filterops = {
    438   1.6   thorpej 	.f_flags = FILTEROP_ISFD | FILTEROP_MPSAFE,
    439   1.2   thorpej 	.f_detach = eventfd_filt_read_detach,
    440   1.2   thorpej 	.f_event = eventfd_filt_read,
    441   1.2   thorpej };
    442   1.2   thorpej 
    443   1.2   thorpej static void
    444   1.2   thorpej eventfd_filt_write_detach(struct knote * const kn)
    445   1.2   thorpej {
    446   1.2   thorpej 	struct eventfd * const efd = ((file_t *)kn->kn_obj)->f_eventfd;
    447   1.2   thorpej 
    448   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    449   1.2   thorpej 	KASSERT(kn->kn_hook == efd);
    450   1.2   thorpej 	selremove_knote(&efd->efd_write_sel, kn);
    451   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    452   1.2   thorpej }
    453   1.2   thorpej 
    454   1.2   thorpej static int
    455   1.2   thorpej eventfd_filt_write(struct knote * const kn, long const hint)
    456   1.2   thorpej {
    457   1.2   thorpej 	struct eventfd * const efd = ((file_t *)kn->kn_obj)->f_eventfd;
    458   1.7   thorpej 	int rv;
    459   1.2   thorpej 
    460   1.2   thorpej 	if (hint & NOTE_SUBMIT) {
    461   1.2   thorpej 		KASSERT(mutex_owned(&efd->efd_lock));
    462   1.2   thorpej 	} else {
    463   1.2   thorpej 		mutex_enter(&efd->efd_lock);
    464   1.2   thorpej 	}
    465   1.2   thorpej 
    466   1.2   thorpej 	kn->kn_data = (int64_t)efd->efd_val;
    467   1.7   thorpej 	rv = (eventfd_t)kn->kn_data < EVENTFD_MAXVAL;
    468   1.2   thorpej 
    469   1.2   thorpej 	if ((hint & NOTE_SUBMIT) == 0) {
    470   1.2   thorpej 		mutex_exit(&efd->efd_lock);
    471   1.2   thorpej 	}
    472   1.2   thorpej 
    473   1.7   thorpej 	return rv;
    474   1.2   thorpej }
    475   1.2   thorpej 
    476   1.2   thorpej static const struct filterops eventfd_write_filterops = {
    477   1.6   thorpej 	.f_flags = FILTEROP_ISFD | FILTEROP_MPSAFE,
    478   1.2   thorpej 	.f_detach = eventfd_filt_write_detach,
    479   1.2   thorpej 	.f_event = eventfd_filt_write,
    480   1.2   thorpej };
    481   1.2   thorpej 
    482   1.2   thorpej static int
    483   1.2   thorpej eventfd_fop_kqfilter(file_t * const fp, struct knote * const kn)
    484   1.2   thorpej {
    485   1.2   thorpej 	struct eventfd * const efd = ((file_t *)kn->kn_obj)->f_eventfd;
    486   1.2   thorpej 	struct selinfo *sel;
    487   1.2   thorpej 
    488   1.2   thorpej 	switch (kn->kn_filter) {
    489   1.2   thorpej 	case EVFILT_READ:
    490   1.2   thorpej 		sel = &efd->efd_read_sel;
    491   1.2   thorpej 		kn->kn_fop = &eventfd_read_filterops;
    492   1.2   thorpej 		break;
    493   1.2   thorpej 
    494   1.2   thorpej 	case EVFILT_WRITE:
    495   1.2   thorpej 		sel = &efd->efd_write_sel;
    496   1.2   thorpej 		kn->kn_fop = &eventfd_write_filterops;
    497   1.2   thorpej 		break;
    498   1.2   thorpej 
    499   1.2   thorpej 	default:
    500   1.2   thorpej 		return EINVAL;
    501   1.2   thorpej 	}
    502   1.2   thorpej 
    503   1.2   thorpej 	kn->kn_hook = efd;
    504   1.2   thorpej 
    505   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    506   1.2   thorpej 	selrecord_knote(sel, kn);
    507   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    508   1.2   thorpej 
    509   1.2   thorpej 	return 0;
    510   1.2   thorpej }
    511   1.2   thorpej 
    512   1.2   thorpej static void
    513   1.2   thorpej eventfd_fop_restart(file_t * const fp)
    514   1.2   thorpej {
    515   1.2   thorpej 	struct eventfd * const efd = fp->f_eventfd;
    516   1.2   thorpej 
    517   1.2   thorpej 	/*
    518   1.2   thorpej 	 * Unblock blocked reads/writes in order to allow close() to complete.
    519   1.2   thorpej 	 * System calls return ERESTART so that the fd is revalidated.
    520   1.2   thorpej 	 */
    521   1.2   thorpej 
    522   1.2   thorpej 	mutex_enter(&efd->efd_lock);
    523   1.2   thorpej 
    524   1.2   thorpej 	if (efd->efd_nwaiters != 0) {
    525   1.2   thorpej 		efd->efd_restarting = true;
    526  1.10  riastrad 		cv_broadcast(&efd->efd_read_wait);
    527  1.10  riastrad 		cv_broadcast(&efd->efd_write_wait);
    528   1.2   thorpej 	}
    529   1.2   thorpej 
    530   1.2   thorpej 	mutex_exit(&efd->efd_lock);
    531   1.2   thorpej }
    532   1.2   thorpej 
    533   1.2   thorpej static const struct fileops eventfd_fileops = {
    534   1.2   thorpej 	.fo_name = "eventfd",
    535   1.2   thorpej 	.fo_read = eventfd_fop_read,
    536   1.2   thorpej 	.fo_write = eventfd_fop_write,
    537   1.9   thorpej 	.fo_ioctl = eventfd_ioctl,
    538   1.2   thorpej 	.fo_fcntl = fnullop_fcntl,
    539   1.2   thorpej 	.fo_poll = eventfd_fop_poll,
    540   1.2   thorpej 	.fo_stat = eventfd_fop_stat,
    541   1.2   thorpej 	.fo_close = eventfd_fop_close,
    542   1.2   thorpej 	.fo_kqfilter = eventfd_fop_kqfilter,
    543   1.2   thorpej 	.fo_restart = eventfd_fop_restart,
    544   1.2   thorpej };
    545   1.2   thorpej 
    546   1.2   thorpej /*
    547   1.2   thorpej  * eventfd(2) system call
    548   1.2   thorpej  */
    549   1.2   thorpej int
    550   1.2   thorpej do_eventfd(struct lwp * const l, unsigned int const val, int const flags,
    551   1.2   thorpej     register_t *retval)
    552   1.2   thorpej {
    553   1.2   thorpej 	file_t *fp;
    554   1.2   thorpej 	int fd, error;
    555   1.2   thorpej 
    556   1.2   thorpej 	if (flags & ~(EFD_CLOEXEC | EFD_NONBLOCK | EFD_SEMAPHORE)) {
    557   1.2   thorpej 		return EINVAL;
    558   1.2   thorpej 	}
    559   1.2   thorpej 
    560   1.2   thorpej 	if ((error = fd_allocfile(&fp, &fd)) != 0) {
    561   1.2   thorpej 		return error;
    562   1.2   thorpej 	}
    563   1.2   thorpej 
    564   1.2   thorpej 	fp->f_flag = FREAD | FWRITE;
    565   1.2   thorpej 	if (flags & EFD_NONBLOCK) {
    566   1.2   thorpej 		fp->f_flag |= FNONBLOCK;
    567   1.2   thorpej 	}
    568   1.2   thorpej 	fp->f_type = DTYPE_EVENTFD;
    569   1.2   thorpej 	fp->f_ops = &eventfd_fileops;
    570   1.2   thorpej 	fp->f_eventfd = eventfd_create(val, flags);
    571   1.2   thorpej 	fd_set_exclose(l, fd, !!(flags & EFD_CLOEXEC));
    572   1.2   thorpej 	fd_affix(curproc, fp, fd);
    573   1.2   thorpej 
    574   1.2   thorpej 	*retval = fd;
    575   1.2   thorpej 	return 0;
    576   1.2   thorpej }
    577   1.2   thorpej 
    578   1.2   thorpej int
    579   1.2   thorpej sys_eventfd(struct lwp *l, const struct sys_eventfd_args *uap,
    580   1.2   thorpej     register_t *retval)
    581   1.2   thorpej {
    582   1.2   thorpej 	/* {
    583   1.2   thorpej 		syscallarg(unsigned int) val;
    584   1.2   thorpej 		syscallarg(int) flags;
    585   1.2   thorpej 	} */
    586   1.2   thorpej 
    587   1.2   thorpej 	return do_eventfd(l, SCARG(uap, val), SCARG(uap, flags), retval);
    588   1.2   thorpej }
    589