1 // SPDX-License-Identifier: GPL-2.0
2 #include <linux/ceph/ceph_debug.h>
3
4 #include <linux/file.h>
5 #include <linux/namei.h>
6 #include <linux/random.h>
7
8 #include "super.h"
9 #include "mds_client.h"
10 #include <linux/filelock.h>
11 #include <linux/ceph/pagelist.h>
12
13 static u64 lock_secret;
14 static int ceph_lock_wait_for_completion(struct ceph_mds_client *mdsc,
15 struct ceph_mds_request *req);
16
secure_addr(void * addr)17 static inline u64 secure_addr(void *addr)
18 {
19 u64 v = lock_secret ^ (u64)(unsigned long)addr;
20 /*
21 * Set the most significant bit, so that MDS knows the 'owner'
22 * is sufficient to identify the owner of lock. (old code uses
23 * both 'owner' and 'pid')
24 */
25 v |= (1ULL << 63);
26 return v;
27 }
28
ceph_flock_init(void)29 void __init ceph_flock_init(void)
30 {
31 get_random_bytes(&lock_secret, sizeof(lock_secret));
32 }
33
ceph_fl_copy_lock(struct file_lock * dst,struct file_lock * src)34 static void ceph_fl_copy_lock(struct file_lock *dst, struct file_lock *src)
35 {
36 struct inode *inode = file_inode(dst->c.flc_file);
37 atomic_inc(&ceph_inode(inode)->i_filelock_ref);
38 dst->fl_u.ceph.inode = igrab(inode);
39 }
40
41 /*
42 * Do not use the 'fl->fl_file' in release function, which
43 * is possibly already released by another thread.
44 */
ceph_fl_release_lock(struct file_lock * fl)45 static void ceph_fl_release_lock(struct file_lock *fl)
46 {
47 struct inode *inode = fl->fl_u.ceph.inode;
48 struct ceph_inode_info *ci;
49
50 /*
51 * If inode is NULL it should be a request file_lock,
52 * nothing we can do.
53 */
54 if (!inode)
55 return;
56
57 ci = ceph_inode(inode);
58 if (atomic_dec_and_test(&ci->i_filelock_ref)) {
59 /* clear error when all locks are released */
60 clear_bit(CEPH_I_ERROR_FILELOCK_BIT, &ci->i_ceph_flags);
61 }
62 fl->fl_u.ceph.inode = NULL;
63 iput(inode);
64 }
65
66 static const struct file_lock_operations ceph_fl_lock_ops = {
67 .fl_copy_lock = ceph_fl_copy_lock,
68 .fl_release_private = ceph_fl_release_lock,
69 };
70
71 /*
72 * Implement fcntl and flock locking functions.
73 */
ceph_lock_message(u8 lock_type,u16 operation,struct inode * inode,int cmd,u8 wait,struct file_lock * fl)74 static int ceph_lock_message(u8 lock_type, u16 operation, struct inode *inode,
75 int cmd, u8 wait, struct file_lock *fl)
76 {
77 struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(inode->i_sb);
78 struct ceph_client *cl = mdsc->fsc->client;
79 struct ceph_mds_request *req;
80 int err;
81 u64 length = 0;
82 u64 owner;
83
84 if (operation == CEPH_MDS_OP_SETFILELOCK) {
85 /*
86 * increasing i_filelock_ref closes race window between
87 * handling request reply and adding file_lock struct to
88 * inode. Otherwise, auth caps may get trimmed in the
89 * window. Caller function will decrease the counter.
90 */
91 fl->fl_ops = &ceph_fl_lock_ops;
92 fl->fl_ops->fl_copy_lock(fl, NULL);
93 }
94
95 if (operation != CEPH_MDS_OP_SETFILELOCK || cmd == CEPH_LOCK_UNLOCK)
96 wait = 0;
97
98 req = ceph_mdsc_create_request(mdsc, operation, USE_AUTH_MDS);
99 if (IS_ERR(req))
100 return PTR_ERR(req);
101 req->r_inode = inode;
102 ihold(inode);
103 req->r_num_caps = 1;
104
105 /* mds requires start and length rather than start and end */
106 if (LLONG_MAX == fl->fl_end)
107 length = 0;
108 else
109 length = fl->fl_end - fl->fl_start + 1;
110
111 owner = secure_addr(fl->c.flc_owner);
112
113 doutc(cl, "rule: %d, op: %d, owner: %llx, pid: %llu, "
114 "start: %llu, length: %llu, wait: %d, type: %d\n",
115 (int)lock_type, (int)operation, owner,
116 (u64) fl->c.flc_pid,
117 fl->fl_start, length, wait, fl->c.flc_type);
118
119 req->r_args.filelock_change.rule = lock_type;
120 req->r_args.filelock_change.type = cmd;
121 req->r_args.filelock_change.owner = cpu_to_le64(owner);
122 req->r_args.filelock_change.pid = cpu_to_le64((u64) fl->c.flc_pid);
123 req->r_args.filelock_change.start = cpu_to_le64(fl->fl_start);
124 req->r_args.filelock_change.length = cpu_to_le64(length);
125 req->r_args.filelock_change.wait = wait;
126
127 err = ceph_mdsc_submit_request(mdsc, inode, req);
128 if (!err)
129 err = ceph_mdsc_wait_request(mdsc, req, wait ?
130 ceph_lock_wait_for_completion : NULL);
131 if (!err && operation == CEPH_MDS_OP_GETFILELOCK) {
132 fl->c.flc_pid = -le64_to_cpu(req->r_reply_info.filelock_reply->pid);
133 if (CEPH_LOCK_SHARED == req->r_reply_info.filelock_reply->type)
134 fl->c.flc_type = F_RDLCK;
135 else if (CEPH_LOCK_EXCL == req->r_reply_info.filelock_reply->type)
136 fl->c.flc_type = F_WRLCK;
137 else
138 fl->c.flc_type = F_UNLCK;
139
140 fl->fl_start = le64_to_cpu(req->r_reply_info.filelock_reply->start);
141 length = le64_to_cpu(req->r_reply_info.filelock_reply->start) +
142 le64_to_cpu(req->r_reply_info.filelock_reply->length);
143 if (length >= 1)
144 fl->fl_end = length -1;
145 else
146 fl->fl_end = 0;
147
148 }
149 ceph_mdsc_put_request(req);
150 doutc(cl, "rule: %d, op: %d, pid: %llu, start: %llu, "
151 "length: %llu, wait: %d, type: %d, err code %d\n",
152 (int)lock_type, (int)operation, (u64) fl->c.flc_pid,
153 fl->fl_start, length, wait, fl->c.flc_type, err);
154 return err;
155 }
156
ceph_lock_wait_for_completion(struct ceph_mds_client * mdsc,struct ceph_mds_request * req)157 static int ceph_lock_wait_for_completion(struct ceph_mds_client *mdsc,
158 struct ceph_mds_request *req)
159 {
160 struct ceph_client *cl = mdsc->fsc->client;
161 struct ceph_mds_request *intr_req;
162 struct inode *inode = req->r_inode;
163 int err, lock_type;
164
165 BUG_ON(req->r_op != CEPH_MDS_OP_SETFILELOCK);
166 if (req->r_args.filelock_change.rule == CEPH_LOCK_FCNTL)
167 lock_type = CEPH_LOCK_FCNTL_INTR;
168 else if (req->r_args.filelock_change.rule == CEPH_LOCK_FLOCK)
169 lock_type = CEPH_LOCK_FLOCK_INTR;
170 else
171 BUG_ON(1);
172 BUG_ON(req->r_args.filelock_change.type == CEPH_LOCK_UNLOCK);
173
174 err = wait_for_completion_interruptible(&req->r_completion);
175 if (!err)
176 return 0;
177
178 doutc(cl, "request %llu was interrupted\n", req->r_tid);
179
180 mutex_lock(&mdsc->mutex);
181 if (test_bit(CEPH_MDS_R_GOT_RESULT, &req->r_req_flags)) {
182 err = 0;
183 } else {
184 /*
185 * ensure we aren't running concurrently with
186 * ceph_fill_trace or ceph_readdir_prepopulate, which
187 * rely on locks (dir mutex) held by our caller.
188 */
189 mutex_lock(&req->r_fill_mutex);
190 req->r_err = err;
191 set_bit(CEPH_MDS_R_ABORTED, &req->r_req_flags);
192 mutex_unlock(&req->r_fill_mutex);
193
194 if (!req->r_session) {
195 // haven't sent the request
196 err = 0;
197 }
198 }
199 mutex_unlock(&mdsc->mutex);
200 if (!err)
201 return 0;
202
203 intr_req = ceph_mdsc_create_request(mdsc, CEPH_MDS_OP_SETFILELOCK,
204 USE_AUTH_MDS);
205 if (IS_ERR(intr_req))
206 return PTR_ERR(intr_req);
207
208 intr_req->r_inode = inode;
209 ihold(inode);
210 intr_req->r_num_caps = 1;
211
212 intr_req->r_args.filelock_change = req->r_args.filelock_change;
213 intr_req->r_args.filelock_change.rule = lock_type;
214 intr_req->r_args.filelock_change.type = CEPH_LOCK_UNLOCK;
215
216 err = ceph_mdsc_do_request(mdsc, inode, intr_req);
217 ceph_mdsc_put_request(intr_req);
218
219 if (err && err != -ERESTARTSYS)
220 return err;
221
222 err = wait_for_completion_killable(&req->r_safe_completion);
223 if (err)
224 return err;
225
226 return 0;
227 }
228
try_unlock_file(struct file * file,struct file_lock * fl)229 static int try_unlock_file(struct file *file, struct file_lock *fl)
230 {
231 int err;
232 unsigned int orig_flags = fl->c.flc_flags;
233 fl->c.flc_flags |= FL_EXISTS;
234 err = locks_lock_file_wait(file, fl);
235 fl->c.flc_flags = orig_flags;
236 if (err == -ENOENT) {
237 if (!(orig_flags & FL_EXISTS))
238 err = 0;
239 return err;
240 }
241 return 1;
242 }
243
244 /*
245 * Attempt to set an fcntl lock.
246 * For now, this just goes away to the server. Later it may be more awesome.
247 */
ceph_lock(struct file * file,int cmd,struct file_lock * fl)248 int ceph_lock(struct file *file, int cmd, struct file_lock *fl)
249 {
250 struct inode *inode = file_inode(file);
251 struct ceph_inode_info *ci = ceph_inode(inode);
252 struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(inode->i_sb);
253 struct ceph_client *cl = ceph_inode_to_client(inode);
254 int err = 0;
255 u16 op = CEPH_MDS_OP_SETFILELOCK;
256 u8 wait = 0;
257 u8 lock_cmd;
258
259 if (!(fl->c.flc_flags & FL_POSIX))
260 return -ENOLCK;
261
262 if (ceph_inode_is_shutdown(inode))
263 return -ESTALE;
264
265 doutc(cl, "fl_owner: %p\n", fl->c.flc_owner);
266
267 /* set wait bit as appropriate, then make command as Ceph expects it*/
268 if (IS_GETLK(cmd))
269 op = CEPH_MDS_OP_GETFILELOCK;
270 else if (IS_SETLKW(cmd))
271 wait = 1;
272
273 if (test_bit(CEPH_I_ERROR_FILELOCK_BIT, &ci->i_ceph_flags)) {
274 if (op == CEPH_MDS_OP_SETFILELOCK && lock_is_unlock(fl))
275 posix_lock_file(file, fl, NULL);
276 return -EIO;
277 }
278
279 /* Wait for reset to complete before acquiring new locks */
280 if (op == CEPH_MDS_OP_SETFILELOCK && !lock_is_unlock(fl)) {
281 err = ceph_mdsc_wait_for_reset(mdsc);
282 if (err)
283 return err;
284 }
285
286 if (lock_is_read(fl))
287 lock_cmd = CEPH_LOCK_SHARED;
288 else if (lock_is_write(fl))
289 lock_cmd = CEPH_LOCK_EXCL;
290 else
291 lock_cmd = CEPH_LOCK_UNLOCK;
292
293 if (op == CEPH_MDS_OP_SETFILELOCK && lock_is_unlock(fl)) {
294 err = try_unlock_file(file, fl);
295 if (err <= 0)
296 return err;
297 }
298
299 err = ceph_lock_message(CEPH_LOCK_FCNTL, op, inode, lock_cmd, wait, fl);
300 if (!err) {
301 if (op == CEPH_MDS_OP_SETFILELOCK && F_UNLCK != fl->c.flc_type) {
302 doutc(cl, "locking locally\n");
303 err = posix_lock_file(file, fl, NULL);
304 if (err) {
305 /* undo! This should only happen if
306 * the kernel detects local
307 * deadlock. */
308 ceph_lock_message(CEPH_LOCK_FCNTL, op, inode,
309 CEPH_LOCK_UNLOCK, 0, fl);
310 doutc(cl, "got %d on posix_lock_file, undid lock\n",
311 err);
312 }
313 }
314 }
315 return err;
316 }
317
ceph_flock(struct file * file,int cmd,struct file_lock * fl)318 int ceph_flock(struct file *file, int cmd, struct file_lock *fl)
319 {
320 struct inode *inode = file_inode(file);
321 struct ceph_inode_info *ci = ceph_inode(inode);
322 struct ceph_mds_client *mdsc = ceph_sb_to_mdsc(inode->i_sb);
323 struct ceph_client *cl = ceph_inode_to_client(inode);
324 int err = 0;
325 u8 wait = 0;
326 u8 lock_cmd;
327
328 if (!(fl->c.flc_flags & FL_FLOCK))
329 return -ENOLCK;
330
331 if (ceph_inode_is_shutdown(inode))
332 return -ESTALE;
333
334 doutc(cl, "fl_file: %p\n", fl->c.flc_file);
335
336 if (test_bit(CEPH_I_ERROR_FILELOCK_BIT, &ci->i_ceph_flags)) {
337 if (lock_is_unlock(fl))
338 locks_lock_file_wait(file, fl);
339 return -EIO;
340 }
341
342 /* Wait for reset to complete before acquiring new locks */
343 if (!lock_is_unlock(fl)) {
344 err = ceph_mdsc_wait_for_reset(mdsc);
345 if (err)
346 return err;
347 }
348
349 if (IS_SETLKW(cmd))
350 wait = 1;
351
352 if (lock_is_read(fl))
353 lock_cmd = CEPH_LOCK_SHARED;
354 else if (lock_is_write(fl))
355 lock_cmd = CEPH_LOCK_EXCL;
356 else
357 lock_cmd = CEPH_LOCK_UNLOCK;
358
359 if (lock_is_unlock(fl)) {
360 err = try_unlock_file(file, fl);
361 if (err <= 0)
362 return err;
363 }
364
365 err = ceph_lock_message(CEPH_LOCK_FLOCK, CEPH_MDS_OP_SETFILELOCK,
366 inode, lock_cmd, wait, fl);
367 if (!err && F_UNLCK != fl->c.flc_type) {
368 err = locks_lock_file_wait(file, fl);
369 if (err) {
370 ceph_lock_message(CEPH_LOCK_FLOCK,
371 CEPH_MDS_OP_SETFILELOCK,
372 inode, CEPH_LOCK_UNLOCK, 0, fl);
373 doutc(cl, "got %d on locks_lock_file_wait, undid lock\n",
374 err);
375 }
376 }
377 return err;
378 }
379
380 /*
381 * Fills in the passed counter variables, so you can prepare pagelist metadata
382 * before calling ceph_encode_locks.
383 */
ceph_count_locks(struct inode * inode,int * fcntl_count,int * flock_count)384 void ceph_count_locks(struct inode *inode, int *fcntl_count, int *flock_count)
385 {
386 struct ceph_client *cl = ceph_inode_to_client(inode);
387 struct file_lock *lock;
388 struct file_lock_context *ctx;
389
390 *fcntl_count = 0;
391 *flock_count = 0;
392
393 ctx = locks_inode_context(inode);
394 if (ctx) {
395 spin_lock(&ctx->flc_lock);
396 for_each_file_lock(lock, &ctx->flc_posix)
397 ++(*fcntl_count);
398 for_each_file_lock(lock, &ctx->flc_flock)
399 ++(*flock_count);
400 spin_unlock(&ctx->flc_lock);
401 }
402 doutc(cl, "counted %d flock locks and %d fcntl locks\n",
403 *flock_count, *fcntl_count);
404 }
405
406 /*
407 * Given a pointer to a lock, convert it to a ceph filelock
408 */
lock_to_ceph_filelock(struct inode * inode,struct file_lock * lock,struct ceph_filelock * cephlock)409 static int lock_to_ceph_filelock(struct inode *inode,
410 struct file_lock *lock,
411 struct ceph_filelock *cephlock)
412 {
413 struct ceph_client *cl = ceph_inode_to_client(inode);
414 int err = 0;
415
416 cephlock->start = cpu_to_le64(lock->fl_start);
417 cephlock->length = cpu_to_le64(lock->fl_end - lock->fl_start + 1);
418 cephlock->client = cpu_to_le64(0);
419 cephlock->pid = cpu_to_le64((u64) lock->c.flc_pid);
420 cephlock->owner = cpu_to_le64(secure_addr(lock->c.flc_owner));
421
422 switch (lock->c.flc_type) {
423 case F_RDLCK:
424 cephlock->type = CEPH_LOCK_SHARED;
425 break;
426 case F_WRLCK:
427 cephlock->type = CEPH_LOCK_EXCL;
428 break;
429 case F_UNLCK:
430 cephlock->type = CEPH_LOCK_UNLOCK;
431 break;
432 default:
433 doutc(cl, "Have unknown lock type %d\n",
434 lock->c.flc_type);
435 err = -EINVAL;
436 }
437
438 return err;
439 }
440
441 /*
442 * Encode the flock and fcntl locks for the given inode into the ceph_filelock
443 * array. Must be called with inode->i_lock already held.
444 * If we encounter more of a specific lock type than expected, return -ENOSPC.
445 */
ceph_encode_locks_to_buffer(struct inode * inode,struct ceph_filelock * flocks,int num_fcntl_locks,int num_flock_locks)446 int ceph_encode_locks_to_buffer(struct inode *inode,
447 struct ceph_filelock *flocks,
448 int num_fcntl_locks, int num_flock_locks)
449 {
450 struct file_lock *lock;
451 struct file_lock_context *ctx = locks_inode_context(inode);
452 struct ceph_client *cl = ceph_inode_to_client(inode);
453 int err = 0;
454 int seen_fcntl = 0;
455 int seen_flock = 0;
456 int l = 0;
457
458 doutc(cl, "encoding %d flock and %d fcntl locks\n", num_flock_locks,
459 num_fcntl_locks);
460
461 if (!ctx)
462 return 0;
463
464 spin_lock(&ctx->flc_lock);
465 for_each_file_lock(lock, &ctx->flc_posix) {
466 ++seen_fcntl;
467 if (seen_fcntl > num_fcntl_locks) {
468 err = -ENOSPC;
469 goto fail;
470 }
471 err = lock_to_ceph_filelock(inode, lock, &flocks[l]);
472 if (err)
473 goto fail;
474 ++l;
475 }
476 for_each_file_lock(lock, &ctx->flc_flock) {
477 ++seen_flock;
478 if (seen_flock > num_flock_locks) {
479 err = -ENOSPC;
480 goto fail;
481 }
482 err = lock_to_ceph_filelock(inode, lock, &flocks[l]);
483 if (err)
484 goto fail;
485 ++l;
486 }
487 fail:
488 spin_unlock(&ctx->flc_lock);
489 return err;
490 }
491
492 /*
493 * Copy the encoded flock and fcntl locks into the pagelist.
494 * Format is: #fcntl locks, sequential fcntl locks, #flock locks,
495 * sequential flock locks.
496 * Returns zero on success.
497 */
ceph_locks_to_pagelist(struct ceph_filelock * flocks,struct ceph_pagelist * pagelist,int num_fcntl_locks,int num_flock_locks)498 int ceph_locks_to_pagelist(struct ceph_filelock *flocks,
499 struct ceph_pagelist *pagelist,
500 int num_fcntl_locks, int num_flock_locks)
501 {
502 int err = 0;
503 __le32 nlocks;
504
505 nlocks = cpu_to_le32(num_fcntl_locks);
506 err = ceph_pagelist_append(pagelist, &nlocks, sizeof(nlocks));
507 if (err)
508 goto out_fail;
509
510 if (num_fcntl_locks > 0) {
511 err = ceph_pagelist_append(pagelist, flocks,
512 num_fcntl_locks * sizeof(*flocks));
513 if (err)
514 goto out_fail;
515 }
516
517 nlocks = cpu_to_le32(num_flock_locks);
518 err = ceph_pagelist_append(pagelist, &nlocks, sizeof(nlocks));
519 if (err)
520 goto out_fail;
521
522 if (num_flock_locks > 0) {
523 err = ceph_pagelist_append(pagelist, &flocks[num_fcntl_locks],
524 num_flock_locks * sizeof(*flocks));
525 }
526 out_fail:
527 return err;
528 }
529