1======= 2Locking 3======= 4 5The text below describes the locking rules for VFS-related methods. 6It is (believed to be) up-to-date. *Please*, if you change anything in 7prototypes or locking protocols - update this file. And update the relevant 8instances in the tree, don't leave that to maintainers of filesystems/devices/ 9etc. At the very least, put the list of dubious cases in the end of this file. 10Don't turn it into log - maintainers of out-of-the-tree code are supposed to 11be able to use diff(1). 12 13Thing currently missing here: socket operations. Alexey? 14 15dentry_operations 16================= 17 18prototypes:: 19 20 int (*d_revalidate)(struct inode *, const struct qstr *, 21 struct dentry *, unsigned int); 22 int (*d_weak_revalidate)(struct dentry *, unsigned int); 23 int (*d_hash)(const struct dentry *, struct qstr *); 24 int (*d_compare)(const struct dentry *, 25 unsigned int, const char *, const struct qstr *); 26 int (*d_delete)(struct dentry *); 27 int (*d_init)(struct dentry *); 28 void (*d_release)(struct dentry *); 29 void (*d_iput)(struct dentry *, struct inode *); 30 char *(*d_dname)((struct dentry *dentry, char *buffer, int buflen); 31 struct vfsmount *(*d_automount)(struct path *path); 32 int (*d_manage)(const struct path *, bool); 33 struct dentry *(*d_real)(struct dentry *, enum d_real_type type); 34 bool (*d_unalias_trylock)(const struct dentry *); 35 void (*d_unalias_unlock)(const struct dentry *); 36 37locking rules: 38 39================== =========== ======== ============== ======== 40ops rename_lock ->d_lock may block rcu-walk 41================== =========== ======== ============== ======== 42d_revalidate: no no yes (ref-walk) maybe 43d_weak_revalidate: no no yes no 44d_hash no no no maybe 45d_compare: yes no no maybe 46d_delete: no yes no no 47d_init: no no yes no 48d_release: no no yes no 49d_prune: no yes no no 50d_iput: no no yes no 51d_dname: no no no no 52d_automount: no no yes no 53d_manage: no no yes (ref-walk) maybe 54d_real no no yes no 55d_unalias_trylock yes no no no 56d_unalias_unlock yes no no no 57================== =========== ======== ============== ======== 58 59inode_operations 60================ 61 62prototypes:: 63 64 int (*create) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t); 65 struct dentry * (*lookup) (struct inode *,struct dentry *, unsigned int); 66 int (*link) (struct dentry *,struct inode *,struct dentry *); 67 int (*unlink) (struct inode *,struct dentry *); 68 int (*symlink) (struct mnt_idmap *, struct inode *,struct dentry *,const char *); 69 struct dentry *(*mkdir) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t); 70 int (*rmdir) (struct inode *,struct dentry *); 71 int (*mknod) (struct mnt_idmap *, struct inode *,struct dentry *,umode_t,dev_t); 72 int (*rename) (struct mnt_idmap *, struct inode *, struct dentry *, 73 struct inode *, struct dentry *, unsigned int); 74 int (*readlink) (struct dentry *, char __user *,int); 75 const char *(*get_link) (struct dentry *, struct inode *, struct delayed_call *); 76 void (*truncate) (struct inode *); 77 int (*permission) (struct mnt_idmap *, struct inode *, int, unsigned int); 78 struct posix_acl * (*get_inode_acl)(struct inode *, int, bool); 79 int (*setattr) (struct mnt_idmap *, struct dentry *, struct iattr *); 80 int (*getattr) (struct mnt_idmap *, const struct path *, struct kstat *, u32, unsigned int); 81 ssize_t (*listxattr) (struct dentry *, char *, size_t); 82 int (*fiemap)(struct inode *, struct fiemap_extent_info *, u64 start, u64 len); 83 void (*update_time)(struct inode *inode, enum fs_update_time type, 84 int flags); 85 void (*sync_lazytime)(struct inode *inode); 86 int (*atomic_open)(struct inode *, struct dentry *, 87 struct file *, unsigned open_flag, 88 umode_t create_mode); 89 int (*tmpfile) (struct mnt_idmap *, struct inode *, 90 struct file *, umode_t); 91 int (*fileattr_set)(struct mnt_idmap *idmap, 92 struct dentry *dentry, struct file_kattr *fa); 93 int (*fileattr_get)(struct dentry *dentry, struct file_kattr *fa); 94 struct posix_acl * (*get_acl)(struct mnt_idmap *, struct dentry *, int); 95 struct offset_ctx *(*get_offset_ctx)(struct inode *inode); 96 97locking rules: 98 all may block 99 100============== ================================================== 101ops i_rwsem(inode) 102============== ================================================== 103lookup: shared 104create: exclusive 105link: exclusive (both) 106mknod: exclusive 107symlink: exclusive 108mkdir: exclusive 109unlink: exclusive (both) 110rmdir: exclusive (both)(see below) 111rename: exclusive (both parents, some children) (see below) 112readlink: no 113get_link: no 114setattr: exclusive 115permission: no (may not block if called in rcu-walk mode) 116get_inode_acl: no 117get_acl: no 118getattr: no 119listxattr: no 120fiemap: no 121update_time: no 122sync_lazytime: no 123atomic_open: shared (exclusive if O_CREAT is set in open flags) 124tmpfile: no 125fileattr_get: no or exclusive 126fileattr_set: exclusive 127get_offset_ctx no 128============== ================================================== 129 130 131 Additionally, ->rmdir(), ->unlink() and ->rename() have ->i_rwsem 132 exclusive on victim. 133 cross-directory ->rename() has (per-superblock) ->s_vfs_rename_sem. 134 ->unlink() and ->rename() have ->i_rwsem exclusive on all non-directories 135 involved. 136 ->rename() has ->i_rwsem exclusive on any subdirectory that changes parent. 137 138See Documentation/filesystems/directory-locking.rst for more detailed discussion 139of the locking scheme for directory operations. 140 141xattr_handler operations 142======================== 143 144prototypes:: 145 146 bool (*list)(struct dentry *dentry); 147 int (*get)(const struct xattr_handler *handler, struct dentry *dentry, 148 struct inode *inode, const char *name, void *buffer, 149 size_t size); 150 int (*set)(const struct xattr_handler *handler, 151 struct mnt_idmap *idmap, 152 struct dentry *dentry, struct inode *inode, const char *name, 153 const void *buffer, size_t size, int flags); 154 155locking rules: 156 all may block 157 158===== ============== 159ops i_rwsem(inode) 160===== ============== 161list: no 162get: no 163set: exclusive 164===== ============== 165 166super_operations 167================ 168 169prototypes:: 170 171 struct inode *(*alloc_inode)(struct super_block *sb); 172 void (*free_inode)(struct inode *); 173 void (*destroy_inode)(struct inode *); 174 void (*dirty_inode) (struct inode *, int flags); 175 int (*write_inode) (struct inode *, struct writeback_control *wbc); 176 int (*drop_inode) (struct inode *); 177 void (*evict_inode) (struct inode *); 178 void (*put_super) (struct super_block *); 179 int (*sync_fs)(struct super_block *sb, int wait); 180 int (*freeze_fs) (struct super_block *); 181 int (*unfreeze_fs) (struct super_block *); 182 int (*statfs) (struct dentry *, struct kstatfs *); 183 void (*umount_begin) (struct super_block *); 184 int (*show_options)(struct seq_file *, struct dentry *); 185 ssize_t (*quota_read)(struct super_block *, int, char *, size_t, loff_t); 186 ssize_t (*quota_write)(struct super_block *, int, const char *, size_t, loff_t); 187 188locking rules: 189 All may block [not true, see below] 190 191====================== ============ ======================== 192ops s_umount note 193====================== ============ ======================== 194alloc_inode: 195free_inode: called from RCU callback 196destroy_inode: 197dirty_inode: 198write_inode: 199drop_inode: !!!inode->i_lock!!! 200evict_inode: 201put_super: write 202sync_fs: read 203freeze_fs: write 204unfreeze_fs: write 205statfs: maybe(read) (see below) 206umount_begin: no 207show_options: no (namespace_sem) 208quota_read: no (see below) 209quota_write: no (see below) 210====================== ============ ======================== 211 212->statfs() has s_umount (shared) when called by ustat(2) (native or 213compat), but that's an accident of bad API; s_umount is used to pin 214the superblock down when we only have dev_t given us by userland to 215identify the superblock. Everything else (statfs(), fstatfs(), etc.) 216doesn't hold it when calling ->statfs() - superblock is pinned down 217by resolving the pathname passed to syscall. 218 219->quota_read() and ->quota_write() functions are both guaranteed to 220be the only ones operating on the quota file by the quota code (via 221dqio_sem) (unless an admin really wants to screw up something and 222writes to quota files with quotas on). For other details about locking 223see also dquot_operations section. 224 225file_system_type 226================ 227 228prototypes:: 229 230 void (*kill_sb) (struct super_block *); 231 232locking rules: 233 234======= ========= 235ops may block 236======= ========= 237kill_sb yes 238======= ========= 239 240->kill_sb() takes a write-locked superblock, does all shutdown work on it, 241unlocks and drops the reference. 242 243address_space_operations 244======================== 245prototypes:: 246 247 int (*read_folio)(struct file *, struct folio *); 248 int (*writepages)(struct address_space *, struct writeback_control *); 249 bool (*dirty_folio)(struct address_space *, struct folio *folio); 250 void (*readahead)(struct readahead_control *); 251 int (*write_begin)(const struct kiocb *, struct address_space *mapping, 252 loff_t pos, unsigned len, 253 struct folio **foliop, void **fsdata); 254 int (*write_end)(const struct kiocb *, struct address_space *mapping, 255 loff_t pos, unsigned len, unsigned copied, 256 struct folio *folio, void *fsdata); 257 sector_t (*bmap)(struct address_space *, sector_t); 258 void (*invalidate_folio) (struct folio *, size_t start, size_t len); 259 bool (*release_folio)(struct folio *, gfp_t); 260 void (*free_folio)(struct folio *); 261 int (*direct_IO)(struct kiocb *, struct iov_iter *iter); 262 int (*migrate_folio)(struct address_space *, struct folio *dst, 263 struct folio *src, enum migrate_mode); 264 int (*launder_folio)(struct folio *); 265 bool (*is_partially_uptodate)(struct folio *, size_t from, size_t count); 266 int (*error_remove_folio)(struct address_space *, struct folio *); 267 int (*swap_activate)(struct swap_info_struct *sis, struct file *f, sector_t *span) 268 int (*swap_deactivate)(struct file *); 269 270locking rules: 271 All except dirty_folio and free_folio may block 272 273====================== ======================== ========= =============== 274ops folio locked i_rwsem invalidate_lock 275====================== ======================== ========= =============== 276read_folio: yes, unlocks shared 277writepages: 278dirty_folio: maybe 279readahead: yes, unlocks shared 280write_begin: locks the folio exclusive 281write_end: yes, unlocks exclusive 282bmap: 283invalidate_folio: yes exclusive 284release_folio: yes 285free_folio: yes 286direct_IO: 287migrate_folio: yes (both) 288launder_folio: yes 289is_partially_uptodate: yes 290error_remove_folio: yes 291swap_activate: no 292swap_deactivate: no 293====================== ======================== ========= =============== 294 295->write_begin(), ->write_end() and ->read_folio() may be called from 296the request handler (/dev/loop). 297 298->read_folio() unlocks the folio, either synchronously or via I/O 299completion. 300 301->readahead() unlocks the folios that I/O is attempted on like ->read_folio(). 302 303->writepages() is used for periodic writeback and for syscall-initiated 304sync operations. The address_space should start I/O against at least 305``*nr_to_write`` pages. ``*nr_to_write`` must be decremented for each page 306which is written. The address_space implementation may write more (or less) 307pages than ``*nr_to_write`` asks for, but it should try to be reasonably close. 308If nr_to_write is NULL, all dirty pages must be written. 309 310writepages should _only_ write pages which are present in 311mapping->i_pages. 312 313->dirty_folio() is called from various places in the kernel when 314the target folio is marked as needing writeback. The folio cannot be 315truncated because either the caller holds the folio lock, or the caller 316has found the folio while holding the page table lock which will block 317truncation. 318 319->bmap() is currently used by legacy ioctl() (FIBMAP) provided by some 320filesystems and by the swapper. The latter will eventually go away. Please, 321keep it that way and don't breed new callers. 322 323->invalidate_folio() is called when the filesystem must attempt to drop 324some or all of the buffers from the page when it is being truncated. It 325returns zero on success. The filesystem must exclusively acquire 326invalidate_lock before invalidating page cache in truncate / hole punch 327path (and thus calling into ->invalidate_folio) to block races between page 328cache invalidation and page cache filling functions (fault, read, ...). 329 330->release_folio() is called when the MM wants to make a change to the 331folio that would invalidate the filesystem's private data. For example, 332it may be about to be removed from the address_space or split. The folio 333is locked and not under writeback. It may be dirty. The gfp parameter 334is not usually used for allocation, but rather to indicate what the 335filesystem may do to attempt to free the private data. The filesystem may 336return false to indicate that the folio's private data cannot be freed. 337If it returns true, it should have already removed the private data from 338the folio. If a filesystem does not provide a ->release_folio method, 339the pagecache will assume that private data is buffer_heads and call 340try_to_free_buffers(). 341 342->free_folio() is called when the kernel has dropped the folio 343from the page cache. 344 345->launder_folio() may be called prior to releasing a folio if 346it is still found to be dirty. It returns zero if the folio was successfully 347cleaned, or an error value if not. Note that in order to prevent the folio 348getting mapped back in and redirtied, it needs to be kept locked 349across the entire operation. 350 351->swap_activate() will be called to prepare the given file for swap. It 352should perform any validation and preparation necessary to ensure that 353writes can be performed with minimal memory allocation. It should call 354add_swap_extent(), or the helper iomap_swapfile_activate(), and return 355the number of extents added. If IO should be submitted through 356the file system it should call swap_fs_activate, otherwise IO will be 357submitted directly to the block device ``sis->bdev``. 358 359->swap_deactivate() will be called in the sys_swapoff() 360path after ->swap_activate() returned success. 361 362file_lock_operations 363==================== 364 365prototypes:: 366 367 void (*fl_copy_lock)(struct file_lock *, struct file_lock *); 368 void (*fl_release_private)(struct file_lock *); 369 370 371locking rules: 372 373=================== ============= ========= 374ops inode->i_lock may block 375=================== ============= ========= 376fl_copy_lock: yes no 377fl_release_private: maybe maybe[1]_ 378=================== ============= ========= 379 380.. [1]: 381 ->fl_release_private for flock or POSIX locks is currently allowed 382 to block. Leases however can still be freed while the i_lock is held and 383 so fl_release_private called on a lease should not block. 384 385lock_manager_operations 386======================= 387 388prototypes:: 389 390 void (*lm_notify)(struct file_lock *); /* unblock callback */ 391 int (*lm_grant)(struct file_lock *, struct file_lock *, int); 392 void (*lm_break)(struct file_lock *); /* break_lease callback */ 393 int (*lm_change)(struct file_lock **, int); 394 bool (*lm_breaker_owns_lease)(struct file_lock *); 395 bool (*lm_lock_expirable)(struct file_lock *); 396 void (*lm_expire_lock)(void); 397 bool (*lm_breaker_timedout)(struct file_lease *); 398 399locking rules: 400 401====================== ============= ================= ========= 402ops flc_lock blocked_lock_lock may block 403====================== ============= ================= ========= 404lm_notify: no yes no 405lm_grant: no no no 406lm_break: yes no no 407lm_change yes no no 408lm_breaker_owns_lease: yes no no 409lm_lock_expirable yes no no 410lm_expire_lock no no yes 411lm_open_conflict yes no no 412lm_breaker_timedout yes no no 413====================== ============= ================= ========= 414 415block_device_operations 416======================= 417prototypes:: 418 419 int (*open) (struct block_device *, fmode_t); 420 int (*release) (struct gendisk *, fmode_t); 421 int (*ioctl) (struct block_device *, fmode_t, unsigned, unsigned long); 422 int (*compat_ioctl) (struct block_device *, fmode_t, unsigned, unsigned long); 423 int (*direct_access) (struct block_device *, sector_t, void **, 424 unsigned long *); 425 void (*unlock_native_capacity) (struct gendisk *); 426 int (*getgeo)(struct gendisk *, struct hd_geometry *); 427 void (*swap_slot_free_notify) (struct block_device *, unsigned long); 428 429locking rules: 430 431======================= =================== 432ops open_mutex 433======================= =================== 434open: yes 435release: yes 436ioctl: no 437compat_ioctl: no 438direct_access: no 439unlock_native_capacity: no 440getgeo: no 441swap_slot_free_notify: no (see below) 442======================= =================== 443 444swap_slot_free_notify is called with swap_lock and sometimes the page lock 445held. 446 447 448file_operations 449=============== 450 451prototypes:: 452 453 loff_t (*llseek) (struct file *, loff_t, int); 454 ssize_t (*read) (struct file *, char __user *, size_t, loff_t *); 455 ssize_t (*write) (struct file *, const char __user *, size_t, loff_t *); 456 ssize_t (*read_iter) (struct kiocb *, struct iov_iter *); 457 ssize_t (*write_iter) (struct kiocb *, struct iov_iter *); 458 int (*iopoll) (struct kiocb *kiocb, bool spin); 459 int (*iterate_shared) (struct file *, struct dir_context *); 460 __poll_t (*poll) (struct file *, struct poll_table_struct *); 461 long (*unlocked_ioctl) (struct file *, unsigned int, unsigned long); 462 long (*compat_ioctl) (struct file *, unsigned int, unsigned long); 463 int (*mmap) (struct file *, struct vm_area_struct *); 464 int (*open) (struct inode *, struct file *); 465 int (*flush) (struct file *); 466 int (*release) (struct inode *, struct file *); 467 int (*fsync) (struct file *, loff_t start, loff_t end, int datasync); 468 int (*fasync) (int, struct file *, int); 469 int (*lock) (struct file *, int, struct file_lock *); 470 unsigned long (*get_unmapped_area)(struct file *, unsigned long, 471 unsigned long, unsigned long, unsigned long); 472 int (*check_flags)(int); 473 int (*flock) (struct file *, int, struct file_lock *); 474 ssize_t (*splice_write)(struct pipe_inode_info *, struct file *, loff_t *, 475 size_t, unsigned int); 476 ssize_t (*splice_read)(struct file *, loff_t *, struct pipe_inode_info *, 477 size_t, unsigned int); 478 int (*setlease)(struct file *, long, struct file_lock **, void **); 479 long (*fallocate)(struct file *, int, loff_t, loff_t); 480 void (*show_fdinfo)(struct seq_file *m, struct file *f); 481 unsigned (*mmap_capabilities)(struct file *); 482 ssize_t (*copy_file_range)(struct file *, loff_t, struct file *, 483 loff_t, size_t, unsigned int); 484 loff_t (*remap_file_range)(struct file *file_in, loff_t pos_in, 485 struct file *file_out, loff_t pos_out, 486 loff_t len, unsigned int remap_flags); 487 int (*fadvise)(struct file *, loff_t, loff_t, int); 488 489locking rules: 490 All may block. 491 492->llseek() locking has moved from llseek to the individual llseek 493implementations. If your fs is not using generic_file_llseek, you 494need to acquire and release the appropriate locks in your ->llseek(). 495For many filesystems, it is probably safe to acquire the inode 496mutex or just to use i_size_read() instead. 497Note: this does not protect the file->f_pos against concurrent modifications 498since this is something the userspace has to take care about. 499 500->iterate_shared() is called with i_rwsem held for reading, and with the 501file f_pos_lock held exclusively 502 503->fasync() is responsible for maintaining the FASYNC bit in filp->f_flags. 504Most instances call fasync_helper(), which does that maintenance, so it's 505not normally something one needs to worry about. Return values > 0 will be 506mapped to zero in the VFS layer. 507 508->readdir() and ->ioctl() on directories must be changed. Ideally we would 509move ->readdir() to inode_operations and use a separate method for directory 510->ioctl() or kill the latter completely. One of the problems is that for 511anything that resembles union-mount we won't have a struct file for all 512components. And there are other reasons why the current interface is a mess... 513 514->read on directories probably must go away - we should just enforce -EISDIR 515in sys_read() and friends. 516 517->setlease operations should call generic_setlease() before or after setting 518the lease within the individual filesystem to record the result of the 519operation 520 521->fallocate implementation must be really careful to maintain page cache 522consistency when punching holes or performing other operations that invalidate 523page cache contents. Usually the filesystem needs to call 524truncate_inode_pages_range() to invalidate relevant range of the page cache. 525However the filesystem usually also needs to update its internal (and on disk) 526view of file offset -> disk block mapping. Until this update is finished, the 527filesystem needs to block page faults and reads from reloading now-stale page 528cache contents from the disk. Since VFS acquires mapping->invalidate_lock in 529shared mode when loading pages from disk (filemap_fault(), filemap_read(), 530readahead paths), the fallocate implementation must take the invalidate_lock to 531prevent reloading. 532 533->copy_file_range and ->remap_file_range implementations need to serialize 534against modifications of file data while the operation is running. For 535blocking changes through write(2) and similar operations inode->i_rwsem can be 536used. To block changes to file contents via a memory mapping during the 537operation, the filesystem must take mapping->invalidate_lock to coordinate 538with ->page_mkwrite. 539 540dquot_operations 541================ 542 543prototypes:: 544 545 int (*write_dquot) (struct dquot *); 546 int (*acquire_dquot) (struct dquot *); 547 int (*release_dquot) (struct dquot *); 548 int (*mark_dirty) (struct dquot *); 549 int (*write_info) (struct super_block *, int); 550 551These operations are intended to be more or less wrapping functions that ensure 552a proper locking wrt the filesystem and call the generic quota operations. 553 554What filesystem should expect from the generic quota functions: 555 556============== ============ ========================= 557ops FS recursion Held locks when called 558============== ============ ========================= 559write_dquot: yes dqonoff_sem or dqptr_sem 560acquire_dquot: yes dqonoff_sem or dqptr_sem 561release_dquot: yes dqonoff_sem or dqptr_sem 562mark_dirty: no - 563write_info: yes dqonoff_sem 564============== ============ ========================= 565 566FS recursion means calling ->quota_read() and ->quota_write() from superblock 567operations. 568 569More details about quota locking can be found in fs/quota/dquot.c. 570 571vm_operations_struct 572==================== 573 574prototypes:: 575 576 void (*open)(struct vm_area_struct *); 577 void (*close)(struct vm_area_struct *); 578 vm_fault_t (*fault)(struct vm_fault *); 579 vm_fault_t (*huge_fault)(struct vm_fault *, unsigned int order); 580 vm_fault_t (*map_pages)(struct vm_fault *, pgoff_t start, pgoff_t end); 581 vm_fault_t (*page_mkwrite)(struct vm_area_struct *, struct vm_fault *); 582 vm_fault_t (*pfn_mkwrite)(struct vm_area_struct *, struct vm_fault *); 583 int (*access)(struct vm_area_struct *, unsigned long, void*, int, int); 584 585locking rules: 586 587============= ========== =========================== 588ops mmap_lock PageLocked(page) 589============= ========== =========================== 590open: write 591close: read/write 592fault: read can return with page locked 593huge_fault: maybe-read 594map_pages: maybe-read 595page_mkwrite: read can return with page locked 596pfn_mkwrite: read 597access: read 598============= ========== =========================== 599 600->fault() is called when a previously not present pte is about to be faulted 601in. The filesystem must find and return the page associated with the passed in 602"pgoff" in the vm_fault structure. If it is possible that the page may be 603truncated and/or invalidated, then the filesystem must lock invalidate_lock, 604then ensure the page is not already truncated (invalidate_lock will block 605subsequent truncate), and then return with VM_FAULT_LOCKED, and the page 606locked. The VM will unlock the page. 607 608->huge_fault() is called when there is no PUD or PMD entry present. This 609gives the filesystem the opportunity to install a PUD or PMD sized page. 610Filesystems can also use the ->fault method to return a PMD sized page, 611so implementing this function may not be necessary. In particular, 612filesystems should not call filemap_fault() from ->huge_fault(). 613The mmap_lock may not be held when this method is called. 614 615->map_pages() is called when VM asks to map easy accessible pages. 616Filesystem should find and map pages associated with offsets from "start_pgoff" 617till "end_pgoff". ->map_pages() is called with the RCU lock held and must 618not block. If it's not possible to reach a page without blocking, 619filesystem should skip it. Filesystem should use set_pte_range() to setup 620page table entry. Pointer to entry associated with the page is passed in 621"pte" field in vm_fault structure. Pointers to entries for other offsets 622should be calculated relative to "pte". 623 624->page_mkwrite() is called when a previously read-only pte is about to become 625writeable. The filesystem again must ensure that there are no 626truncate/invalidate races or races with operations such as ->remap_file_range 627or ->copy_file_range, and then return with the page locked. Usually 628mapping->invalidate_lock is suitable for proper serialization. If the page has 629been truncated, the filesystem should not look up a new page like the ->fault() 630handler, but simply return with VM_FAULT_NOPAGE, which will cause the VM to 631retry the fault. 632 633->pfn_mkwrite() is the same as page_mkwrite but when the pte is 634VM_PFNMAP or VM_MIXEDMAP with a page-less entry. Expected return is 635VM_FAULT_NOPAGE. Or one of the VM_FAULT_ERROR types. The default behavior 636after this call is to make the pte read-write, unless pfn_mkwrite returns 637an error. 638 639->access() is called when get_user_pages() fails in 640access_process_vm(), typically used to debug a process through 641/proc/pid/mem or ptrace. This function is needed only for 642VM_IO | VM_PFNMAP VMAs. 643 644-------------------------------------------------------------------------------- 645 646 Dubious stuff 647 648(if you break something or notice that it is broken and do not fix it yourself 649- at least put it here) 650