2 * fs/inotify_user.c - inotify support for userspace
5 * John McCutchan <ttb@tentacle.dhs.org>
6 * Robert Love <rml@novell.com>
8 * Copyright (C) 2005 John McCutchan
9 * Copyright 2006 Hewlett-Packard Development Company, L.P.
11 * This program is free software; you can redistribute it and/or modify it
12 * under the terms of the GNU General Public License as published by the
13 * Free Software Foundation; either version 2, or (at your option) any
16 * This program is distributed in the hope that it will be useful, but
17 * WITHOUT ANY WARRANTY; without even the implied warranty of
18 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
19 * General Public License for more details.
22 #include <linux/kernel.h>
23 #include <linux/sched.h>
24 #include <linux/slab.h>
26 #include <linux/file.h>
27 #include <linux/mount.h>
28 #include <linux/namei.h>
29 #include <linux/poll.h>
30 #include <linux/init.h>
31 #include <linux/list.h>
32 #include <linux/inotify.h>
33 #include <linux/syscalls.h>
34 #include <linux/magic.h>
36 #include <asm/ioctls.h>
38 static struct kmem_cache
*watch_cachep __read_mostly
;
39 static struct kmem_cache
*event_cachep __read_mostly
;
41 static struct vfsmount
*inotify_mnt __read_mostly
;
43 /* these are configurable via /proc/sys/fs/inotify/ */
44 static int inotify_max_user_instances __read_mostly
;
45 static int inotify_max_user_watches __read_mostly
;
46 static int inotify_max_queued_events __read_mostly
;
51 * inotify_dev->up_mutex (ensures we don't re-add the same watch)
52 * inode->inotify_mutex (protects inode's watch list)
53 * inotify_handle->mutex (protects inotify_handle's watch list)
54 * inotify_dev->ev_mutex (protects device's event queue)
58 * Lifetimes of the main data structures:
60 * inotify_device: Lifetime is managed by reference count, from
61 * sys_inotify_init() until release. Additional references can bump the count
62 * via get_inotify_dev() and drop the count via put_inotify_dev().
64 * inotify_user_watch: Lifetime is from create_watch() to the receipt of an
65 * IN_IGNORED event from inotify, or when using IN_ONESHOT, to receipt of the
66 * first event, or to inotify_destroy().
70 * struct inotify_device - represents an inotify instance
72 * This structure is protected by the mutex 'mutex'.
74 struct inotify_device
{
75 wait_queue_head_t wq
; /* wait queue for i/o */
76 struct mutex ev_mutex
; /* protects event queue */
77 struct mutex up_mutex
; /* synchronizes watch updates */
78 struct list_head events
; /* list of queued events */
79 atomic_t count
; /* reference count */
80 struct user_struct
*user
; /* user who opened this dev */
81 struct inotify_handle
*ih
; /* inotify handle */
82 struct fasync_struct
*fa
; /* async notification */
83 unsigned int queue_size
; /* size of the queue (bytes) */
84 unsigned int event_count
; /* number of pending events */
85 unsigned int max_events
; /* maximum number of events */
89 * struct inotify_kernel_event - An inotify event, originating from a watch and
90 * queued for user-space. A list of these is attached to each instance of the
91 * device. In read(), this list is walked and all events that can fit in the
92 * buffer are returned.
94 * Protected by dev->ev_mutex of the device in which we are queued.
96 struct inotify_kernel_event
{
97 struct inotify_event event
; /* the user-space event */
98 struct list_head list
; /* entry in inotify_device's list */
99 char *name
; /* filename, if any */
103 * struct inotify_user_watch - our version of an inotify_watch, we add
104 * a reference to the associated inotify_device.
106 struct inotify_user_watch
{
107 struct inotify_device
*dev
; /* associated device */
108 struct inotify_watch wdata
; /* inotify watch data */
113 #include <linux/sysctl.h>
117 ctl_table inotify_table
[] = {
119 .ctl_name
= INOTIFY_MAX_USER_INSTANCES
,
120 .procname
= "max_user_instances",
121 .data
= &inotify_max_user_instances
,
122 .maxlen
= sizeof(int),
124 .proc_handler
= &proc_dointvec_minmax
,
125 .strategy
= &sysctl_intvec
,
129 .ctl_name
= INOTIFY_MAX_USER_WATCHES
,
130 .procname
= "max_user_watches",
131 .data
= &inotify_max_user_watches
,
132 .maxlen
= sizeof(int),
134 .proc_handler
= &proc_dointvec_minmax
,
135 .strategy
= &sysctl_intvec
,
139 .ctl_name
= INOTIFY_MAX_QUEUED_EVENTS
,
140 .procname
= "max_queued_events",
141 .data
= &inotify_max_queued_events
,
142 .maxlen
= sizeof(int),
144 .proc_handler
= &proc_dointvec_minmax
,
145 .strategy
= &sysctl_intvec
,
150 #endif /* CONFIG_SYSCTL */
152 static inline void get_inotify_dev(struct inotify_device
*dev
)
154 atomic_inc(&dev
->count
);
157 static inline void put_inotify_dev(struct inotify_device
*dev
)
159 if (atomic_dec_and_test(&dev
->count
)) {
160 atomic_dec(&dev
->user
->inotify_devs
);
167 * free_inotify_user_watch - cleans up the watch and its references
169 static void free_inotify_user_watch(struct inotify_watch
*w
)
171 struct inotify_user_watch
*watch
;
172 struct inotify_device
*dev
;
174 watch
= container_of(w
, struct inotify_user_watch
, wdata
);
177 atomic_dec(&dev
->user
->inotify_watches
);
178 put_inotify_dev(dev
);
179 kmem_cache_free(watch_cachep
, watch
);
183 * kernel_event - create a new kernel event with the given parameters
185 * This function can sleep.
187 static struct inotify_kernel_event
* kernel_event(s32 wd
, u32 mask
, u32 cookie
,
190 struct inotify_kernel_event
*kevent
;
192 kevent
= kmem_cache_alloc(event_cachep
, GFP_NOFS
);
193 if (unlikely(!kevent
))
196 /* we hand this out to user-space, so zero it just in case */
197 memset(&kevent
->event
, 0, sizeof(struct inotify_event
));
199 kevent
->event
.wd
= wd
;
200 kevent
->event
.mask
= mask
;
201 kevent
->event
.cookie
= cookie
;
203 INIT_LIST_HEAD(&kevent
->list
);
206 size_t len
, rem
, event_size
= sizeof(struct inotify_event
);
209 * We need to pad the filename so as to properly align an
210 * array of inotify_event structures. Because the structure is
211 * small and the common case is a small filename, we just round
212 * up to the next multiple of the structure's sizeof. This is
213 * simple and safe for all architectures.
215 len
= strlen(name
) + 1;
216 rem
= event_size
- len
;
217 if (len
> event_size
) {
218 rem
= event_size
- (len
% event_size
);
219 if (len
% event_size
== 0)
223 kevent
->name
= kmalloc(len
+ rem
, GFP_KERNEL
);
224 if (unlikely(!kevent
->name
)) {
225 kmem_cache_free(event_cachep
, kevent
);
228 memcpy(kevent
->name
, name
, len
);
230 memset(kevent
->name
+ len
, 0, rem
);
231 kevent
->event
.len
= len
+ rem
;
233 kevent
->event
.len
= 0;
241 * inotify_dev_get_event - return the next event in the given dev's queue
243 * Caller must hold dev->ev_mutex.
245 static inline struct inotify_kernel_event
*
246 inotify_dev_get_event(struct inotify_device
*dev
)
248 return list_entry(dev
->events
.next
, struct inotify_kernel_event
, list
);
252 * inotify_dev_get_last_event - return the last event in the given dev's queue
254 * Caller must hold dev->ev_mutex.
256 static inline struct inotify_kernel_event
*
257 inotify_dev_get_last_event(struct inotify_device
*dev
)
259 if (list_empty(&dev
->events
))
261 return list_entry(dev
->events
.prev
, struct inotify_kernel_event
, list
);
265 * inotify_dev_queue_event - event handler registered with core inotify, adds
266 * a new event to the given device
268 * Can sleep (calls kernel_event()).
270 static void inotify_dev_queue_event(struct inotify_watch
*w
, u32 wd
, u32 mask
,
271 u32 cookie
, const char *name
,
272 struct inode
*ignored
)
274 struct inotify_user_watch
*watch
;
275 struct inotify_device
*dev
;
276 struct inotify_kernel_event
*kevent
, *last
;
278 watch
= container_of(w
, struct inotify_user_watch
, wdata
);
281 mutex_lock(&dev
->ev_mutex
);
283 /* we can safely put the watch as we don't reference it while
284 * generating the event
286 if (mask
& IN_IGNORED
|| w
->mask
& IN_ONESHOT
)
287 put_inotify_watch(w
); /* final put */
289 /* coalescing: drop this event if it is a dupe of the previous */
290 last
= inotify_dev_get_last_event(dev
);
291 if (last
&& last
->event
.mask
== mask
&& last
->event
.wd
== wd
&&
292 last
->event
.cookie
== cookie
) {
293 const char *lastname
= last
->name
;
295 if (!name
&& !lastname
)
297 if (name
&& lastname
&& !strcmp(lastname
, name
))
301 /* the queue overflowed and we already sent the Q_OVERFLOW event */
302 if (unlikely(dev
->event_count
> dev
->max_events
))
305 /* if the queue overflows, we need to notify user space */
306 if (unlikely(dev
->event_count
== dev
->max_events
))
307 kevent
= kernel_event(-1, IN_Q_OVERFLOW
, cookie
, NULL
);
309 kevent
= kernel_event(wd
, mask
, cookie
, name
);
311 if (unlikely(!kevent
))
314 /* queue the event and wake up anyone waiting */
316 dev
->queue_size
+= sizeof(struct inotify_event
) + kevent
->event
.len
;
317 list_add_tail(&kevent
->list
, &dev
->events
);
318 wake_up_interruptible(&dev
->wq
);
319 kill_fasync(&dev
->fa
, SIGIO
, POLL_IN
);
322 mutex_unlock(&dev
->ev_mutex
);
326 * remove_kevent - cleans up and ultimately frees the given kevent
328 * Caller must hold dev->ev_mutex.
330 static void remove_kevent(struct inotify_device
*dev
,
331 struct inotify_kernel_event
*kevent
)
333 list_del(&kevent
->list
);
336 dev
->queue_size
-= sizeof(struct inotify_event
) + kevent
->event
.len
;
339 kmem_cache_free(event_cachep
, kevent
);
343 * inotify_dev_event_dequeue - destroy an event on the given device
345 * Caller must hold dev->ev_mutex.
347 static void inotify_dev_event_dequeue(struct inotify_device
*dev
)
349 if (!list_empty(&dev
->events
)) {
350 struct inotify_kernel_event
*kevent
;
351 kevent
= inotify_dev_get_event(dev
);
352 remove_kevent(dev
, kevent
);
357 * find_inode - resolve a user-given path to a specific inode and return a nd
359 static int find_inode(const char __user
*dirname
, struct nameidata
*nd
,
364 error
= __user_walk(dirname
, flags
, nd
);
367 /* you can only watch an inode if you have read permissions on it */
368 error
= vfs_permission(nd
, MAY_READ
);
375 * create_watch - creates a watch on the given device.
377 * Callers must hold dev->up_mutex.
379 static int create_watch(struct inotify_device
*dev
, struct inode
*inode
,
382 struct inotify_user_watch
*watch
;
385 if (atomic_read(&dev
->user
->inotify_watches
) >=
386 inotify_max_user_watches
)
389 watch
= kmem_cache_alloc(watch_cachep
, GFP_KERNEL
);
390 if (unlikely(!watch
))
393 /* save a reference to device and bump the count to make it official */
394 get_inotify_dev(dev
);
397 atomic_inc(&dev
->user
->inotify_watches
);
399 inotify_init_watch(&watch
->wdata
);
400 ret
= inotify_add_watch(dev
->ih
, &watch
->wdata
, inode
, mask
);
402 free_inotify_user_watch(&watch
->wdata
);
407 /* Device Interface */
409 static unsigned int inotify_poll(struct file
*file
, poll_table
*wait
)
411 struct inotify_device
*dev
= file
->private_data
;
414 poll_wait(file
, &dev
->wq
, wait
);
415 mutex_lock(&dev
->ev_mutex
);
416 if (!list_empty(&dev
->events
))
417 ret
= POLLIN
| POLLRDNORM
;
418 mutex_unlock(&dev
->ev_mutex
);
423 static ssize_t
inotify_read(struct file
*file
, char __user
*buf
,
424 size_t count
, loff_t
*pos
)
426 size_t event_size
= sizeof (struct inotify_event
);
427 struct inotify_device
*dev
;
433 dev
= file
->private_data
;
438 prepare_to_wait(&dev
->wq
, &wait
, TASK_INTERRUPTIBLE
);
440 mutex_lock(&dev
->ev_mutex
);
441 events
= !list_empty(&dev
->events
);
442 mutex_unlock(&dev
->ev_mutex
);
448 if (file
->f_flags
& O_NONBLOCK
) {
453 if (signal_pending(current
)) {
461 finish_wait(&dev
->wq
, &wait
);
465 mutex_lock(&dev
->ev_mutex
);
467 struct inotify_kernel_event
*kevent
;
470 if (list_empty(&dev
->events
))
473 kevent
= inotify_dev_get_event(dev
);
474 if (event_size
+ kevent
->event
.len
> count
) {
475 if (ret
== 0 && count
> 0) {
477 * could not get a single event because we
478 * didn't have enough buffer space.
485 if (copy_to_user(buf
, &kevent
->event
, event_size
)) {
493 if (copy_to_user(buf
, kevent
->name
, kevent
->event
.len
)){
497 buf
+= kevent
->event
.len
;
498 count
-= kevent
->event
.len
;
501 remove_kevent(dev
, kevent
);
503 mutex_unlock(&dev
->ev_mutex
);
508 static int inotify_fasync(int fd
, struct file
*file
, int on
)
510 struct inotify_device
*dev
= file
->private_data
;
512 return fasync_helper(fd
, file
, on
, &dev
->fa
) >= 0 ? 0 : -EIO
;
515 static int inotify_release(struct inode
*ignored
, struct file
*file
)
517 struct inotify_device
*dev
= file
->private_data
;
519 inotify_destroy(dev
->ih
);
521 /* destroy all of the events on this device */
522 mutex_lock(&dev
->ev_mutex
);
523 while (!list_empty(&dev
->events
))
524 inotify_dev_event_dequeue(dev
);
525 mutex_unlock(&dev
->ev_mutex
);
527 if (file
->f_flags
& FASYNC
)
528 inotify_fasync(-1, file
, 0);
530 /* free this device: the put matching the get in inotify_init() */
531 put_inotify_dev(dev
);
536 static long inotify_ioctl(struct file
*file
, unsigned int cmd
,
539 struct inotify_device
*dev
;
543 dev
= file
->private_data
;
544 p
= (void __user
*) arg
;
548 ret
= put_user(dev
->queue_size
, (int __user
*) p
);
555 static const struct file_operations inotify_fops
= {
556 .poll
= inotify_poll
,
557 .read
= inotify_read
,
558 .fasync
= inotify_fasync
,
559 .release
= inotify_release
,
560 .unlocked_ioctl
= inotify_ioctl
,
561 .compat_ioctl
= inotify_ioctl
,
564 static const struct inotify_operations inotify_user_ops
= {
565 .handle_event
= inotify_dev_queue_event
,
566 .destroy_watch
= free_inotify_user_watch
,
569 asmlinkage
long sys_inotify_init(void)
571 struct inotify_device
*dev
;
572 struct inotify_handle
*ih
;
573 struct user_struct
*user
;
577 fd
= get_unused_fd();
581 filp
= get_empty_filp();
587 user
= get_uid(current
->user
);
588 if (unlikely(atomic_read(&user
->inotify_devs
) >=
589 inotify_max_user_instances
)) {
594 dev
= kmalloc(sizeof(struct inotify_device
), GFP_KERNEL
);
595 if (unlikely(!dev
)) {
600 ih
= inotify_init(&inotify_user_ops
);
608 filp
->f_op
= &inotify_fops
;
609 filp
->f_path
.mnt
= mntget(inotify_mnt
);
610 filp
->f_path
.dentry
= dget(inotify_mnt
->mnt_root
);
611 filp
->f_mapping
= filp
->f_path
.dentry
->d_inode
->i_mapping
;
612 filp
->f_mode
= FMODE_READ
;
613 filp
->f_flags
= O_RDONLY
;
614 filp
->private_data
= dev
;
616 INIT_LIST_HEAD(&dev
->events
);
617 init_waitqueue_head(&dev
->wq
);
618 mutex_init(&dev
->ev_mutex
);
619 mutex_init(&dev
->up_mutex
);
620 dev
->event_count
= 0;
622 dev
->max_events
= inotify_max_queued_events
;
624 atomic_set(&dev
->count
, 0);
626 get_inotify_dev(dev
);
627 atomic_inc(&user
->inotify_devs
);
628 fd_install(fd
, filp
);
641 asmlinkage
long sys_inotify_add_watch(int fd
, const char __user
*path
, u32 mask
)
644 struct inotify_device
*dev
;
647 int ret
, fput_needed
;
650 filp
= fget_light(fd
, &fput_needed
);
654 /* verify that this is indeed an inotify instance */
655 if (unlikely(filp
->f_op
!= &inotify_fops
)) {
660 if (!(mask
& IN_DONT_FOLLOW
))
661 flags
|= LOOKUP_FOLLOW
;
662 if (mask
& IN_ONLYDIR
)
663 flags
|= LOOKUP_DIRECTORY
;
665 ret
= find_inode(path
, &nd
, flags
);
669 /* inode held in place by reference to nd; dev by fget on fd */
670 inode
= nd
.path
.dentry
->d_inode
;
671 dev
= filp
->private_data
;
673 mutex_lock(&dev
->up_mutex
);
674 ret
= inotify_find_update_watch(dev
->ih
, inode
, mask
);
676 ret
= create_watch(dev
, inode
, mask
);
677 mutex_unlock(&dev
->up_mutex
);
681 fput_light(filp
, fput_needed
);
685 asmlinkage
long sys_inotify_rm_watch(int fd
, u32 wd
)
688 struct inotify_device
*dev
;
689 int ret
, fput_needed
;
691 filp
= fget_light(fd
, &fput_needed
);
695 /* verify that this is indeed an inotify instance */
696 if (unlikely(filp
->f_op
!= &inotify_fops
)) {
701 dev
= filp
->private_data
;
703 /* we free our watch data when we get IN_IGNORED */
704 ret
= inotify_rm_wd(dev
->ih
, wd
);
707 fput_light(filp
, fput_needed
);
712 inotify_get_sb(struct file_system_type
*fs_type
, int flags
,
713 const char *dev_name
, void *data
, struct vfsmount
*mnt
)
715 return get_sb_pseudo(fs_type
, "inotify", NULL
,
716 INOTIFYFS_SUPER_MAGIC
, mnt
);
719 static struct file_system_type inotify_fs_type
= {
721 .get_sb
= inotify_get_sb
,
722 .kill_sb
= kill_anon_super
,
726 * inotify_user_setup - Our initialization function. Note that we cannnot return
727 * error because we have compiled-in VFS hooks. So an (unlikely) failure here
728 * must result in panic().
730 static int __init
inotify_user_setup(void)
734 ret
= register_filesystem(&inotify_fs_type
);
736 panic("inotify: register_filesystem returned %d!\n", ret
);
738 inotify_mnt
= kern_mount(&inotify_fs_type
);
739 if (IS_ERR(inotify_mnt
))
740 panic("inotify: kern_mount ret %ld!\n", PTR_ERR(inotify_mnt
));
742 inotify_max_queued_events
= 16384;
743 inotify_max_user_instances
= 128;
744 inotify_max_user_watches
= 8192;
746 watch_cachep
= kmem_cache_create("inotify_watch_cache",
747 sizeof(struct inotify_user_watch
),
748 0, SLAB_PANIC
, NULL
);
749 event_cachep
= kmem_cache_create("inotify_event_cache",
750 sizeof(struct inotify_kernel_event
),
751 0, SLAB_PANIC
, NULL
);
756 module_init(inotify_user_setup
);