1 // SPDX-License-Identifier: GPL-2.0-only
2 /*
3 * eventfd support for mshv
4 *
5 * Heavily inspired from KVM implementation of irqfd/ioeventfd. The basic
6 * framework code is taken from the kvm implementation.
7 *
8 * All credits to kvm developers.
9 */
10
11 #include <linux/syscalls.h>
12 #include <linux/wait.h>
13 #include <linux/poll.h>
14 #include <linux/file.h>
15 #include <linux/list.h>
16 #include <linux/workqueue.h>
17 #include <linux/eventfd.h>
18
19 #if IS_ENABLED(CONFIG_X86_64)
20 #include <asm/apic.h>
21 #endif
22 #include <asm/mshyperv.h>
23
24 #include "mshv_eventfd.h"
25 #include "mshv.h"
26 #include "mshv_root.h"
27
28 static struct workqueue_struct *irqfd_cleanup_wq;
29
mshv_register_irq_ack_notifier(struct mshv_partition * partition,struct mshv_irq_ack_notifier * mian)30 void mshv_register_irq_ack_notifier(struct mshv_partition *partition,
31 struct mshv_irq_ack_notifier *mian)
32 {
33 mutex_lock(&partition->pt_irq_lock);
34 hlist_add_head_rcu(&mian->link, &partition->irq_ack_notifier_list);
35 mutex_unlock(&partition->pt_irq_lock);
36 }
37
mshv_unregister_irq_ack_notifier(struct mshv_partition * partition,struct mshv_irq_ack_notifier * mian)38 void mshv_unregister_irq_ack_notifier(struct mshv_partition *partition,
39 struct mshv_irq_ack_notifier *mian)
40 {
41 mutex_lock(&partition->pt_irq_lock);
42 hlist_del_init_rcu(&mian->link);
43 mutex_unlock(&partition->pt_irq_lock);
44 synchronize_rcu();
45 }
46
mshv_notify_acked_gsi(struct mshv_partition * partition,int gsi)47 bool mshv_notify_acked_gsi(struct mshv_partition *partition, int gsi)
48 {
49 struct mshv_irq_ack_notifier *mian;
50 bool acked = false;
51
52 rcu_read_lock();
53 hlist_for_each_entry_rcu(mian, &partition->irq_ack_notifier_list,
54 link) {
55 if (mian->irq_ack_gsi == gsi) {
56 mian->irq_acked(mian);
57 acked = true;
58 }
59 }
60 rcu_read_unlock();
61
62 return acked;
63 }
64
65 #if IS_ENABLED(CONFIG_ARM64)
hv_should_clear_interrupt(enum hv_interrupt_type type)66 static inline bool hv_should_clear_interrupt(enum hv_interrupt_type type)
67 {
68 return false;
69 }
70 #elif IS_ENABLED(CONFIG_X86_64)
hv_should_clear_interrupt(enum hv_interrupt_type type)71 static inline bool hv_should_clear_interrupt(enum hv_interrupt_type type)
72 {
73 return type == HV_X64_INTERRUPT_TYPE_EXTINT;
74 }
75 #endif
76
mshv_irqfd_resampler_ack(struct mshv_irq_ack_notifier * mian)77 static void mshv_irqfd_resampler_ack(struct mshv_irq_ack_notifier *mian)
78 {
79 struct mshv_irqfd_resampler *resampler;
80 struct mshv_partition *partition;
81 struct mshv_irqfd *irqfd;
82 int idx;
83
84 resampler = container_of(mian, struct mshv_irqfd_resampler,
85 rsmplr_notifier);
86 partition = resampler->rsmplr_partn;
87
88 idx = srcu_read_lock(&partition->pt_irq_srcu);
89
90 hlist_for_each_entry_srcu(irqfd, &resampler->rsmplr_irqfd_list,
91 irqfd_resampler_hnode,
92 srcu_read_lock_held(&partition->pt_irq_srcu)) {
93 if (hv_should_clear_interrupt(irqfd->irqfd_lapic_irq.lapic_control.interrupt_type))
94 hv_call_clear_virtual_interrupt(partition->pt_id);
95
96 eventfd_signal(irqfd->irqfd_resamplefd);
97 }
98
99 srcu_read_unlock(&partition->pt_irq_srcu, idx);
100 }
101
102 #if IS_ENABLED(CONFIG_X86_64)
103 static bool
mshv_vp_irq_vector_injected(union hv_vp_register_page_interrupt_vectors iv,u32 vector)104 mshv_vp_irq_vector_injected(union hv_vp_register_page_interrupt_vectors iv,
105 u32 vector)
106 {
107 int i;
108
109 for (i = 0; i < iv.vector_count; i++) {
110 if (iv.vector[i] == vector)
111 return true;
112 }
113
114 return false;
115 }
116
mshv_vp_irq_try_set_vector(struct mshv_vp * vp,u32 vector)117 static int mshv_vp_irq_try_set_vector(struct mshv_vp *vp, u32 vector)
118 {
119 union hv_vp_register_page_interrupt_vectors iv, new_iv;
120
121 iv = vp->vp_register_page->interrupt_vectors;
122 new_iv = iv;
123
124 if (mshv_vp_irq_vector_injected(iv, vector))
125 return 0;
126
127 if (iv.vector_count >= HV_VP_REGISTER_PAGE_MAX_VECTOR_COUNT)
128 return -ENOSPC;
129
130 new_iv.vector[new_iv.vector_count++] = vector;
131
132 if (!try_cmpxchg(&vp->vp_register_page->interrupt_vectors.as_uint64,
133 &iv.as_uint64, new_iv.as_uint64))
134 return -EAGAIN;
135
136 return 0;
137 }
138
mshv_vp_irq_set_vector(struct mshv_vp * vp,u32 vector)139 static int mshv_vp_irq_set_vector(struct mshv_vp *vp, u32 vector)
140 {
141 int ret;
142
143 do {
144 ret = mshv_vp_irq_try_set_vector(vp, vector);
145 } while (ret == -EAGAIN && !need_resched());
146
147 return ret;
148 }
149
150 /*
151 * Try to raise irq for guest via shared vector array. hyp does the actual
152 * inject of the interrupt.
153 */
mshv_try_assert_irq_fast(struct mshv_irqfd * irqfd)154 static int mshv_try_assert_irq_fast(struct mshv_irqfd *irqfd)
155 {
156 struct mshv_partition *partition = irqfd->irqfd_partn;
157 struct mshv_lapic_irq *irq = &irqfd->irqfd_lapic_irq;
158 struct mshv_vp *vp;
159
160 if (!(ms_hyperv.ext_features &
161 HV_VP_DISPATCH_INTERRUPT_INJECTION_AVAILABLE))
162 return -EOPNOTSUPP;
163
164 if (hv_scheduler_type != HV_SCHEDULER_TYPE_ROOT)
165 return -EOPNOTSUPP;
166
167 #if IS_ENABLED(CONFIG_X86)
168 if (irq->lapic_control.logical_dest_mode)
169 return -EOPNOTSUPP;
170 #endif
171
172 /*
173 * Pairs with smp_store_release() in mshv_partition_ioctl_create_vp().
174 * MSHV_IRQFD does not require the target lapic_apic_id to refer to an
175 * existing VP, so this read can race a concurrent VP creation; the
176 * acquire ensures that a non-NULL pointer implies the VP's
177 * initialising stores are visible.
178 */
179 vp = smp_load_acquire(&partition->pt_vp_array[irq->lapic_apic_id]);
180
181 if (!vp->vp_register_page)
182 return -EOPNOTSUPP;
183
184 if (mshv_vp_irq_set_vector(vp, irq->lapic_vector))
185 return -EINVAL;
186
187 if (vp->run.flags.root_sched_dispatched &&
188 vp->vp_register_page->interrupt_vectors.as_uint64)
189 return -EBUSY;
190
191 wake_up(&vp->run.vp_suspend_queue);
192
193 return 0;
194 }
195 #else /* CONFIG_X86_64 */
mshv_try_assert_irq_fast(struct mshv_irqfd * irqfd)196 static int mshv_try_assert_irq_fast(struct mshv_irqfd *irqfd)
197 {
198 return -EOPNOTSUPP;
199 }
200 #endif
201
mshv_assert_irq_slow(struct mshv_irqfd * irqfd)202 static void mshv_assert_irq_slow(struct mshv_irqfd *irqfd)
203 {
204 struct mshv_partition *partition = irqfd->irqfd_partn;
205 struct mshv_lapic_irq *irq = &irqfd->irqfd_lapic_irq;
206 unsigned int seq;
207 int idx;
208
209 #if IS_ENABLED(CONFIG_X86)
210 WARN_ON(irqfd->irqfd_resampler &&
211 !irq->lapic_control.level_triggered);
212 #endif
213
214 idx = srcu_read_lock(&partition->pt_irq_srcu);
215 if (irqfd->irqfd_girq_ent.guest_irq_num) {
216 if (!irqfd->irqfd_girq_ent.girq_entry_valid) {
217 srcu_read_unlock(&partition->pt_irq_srcu, idx);
218 return;
219 }
220
221 do {
222 seq = read_seqcount_begin(&irqfd->irqfd_irqe_sc);
223 } while (read_seqcount_retry(&irqfd->irqfd_irqe_sc, seq));
224 }
225
226 hv_call_assert_virtual_interrupt(irqfd->irqfd_partn->pt_id,
227 irq->lapic_vector, irq->lapic_apic_id,
228 irq->lapic_control);
229 srcu_read_unlock(&partition->pt_irq_srcu, idx);
230 }
231
mshv_irqfd_resampler_shutdown(struct mshv_irqfd * irqfd)232 static void mshv_irqfd_resampler_shutdown(struct mshv_irqfd *irqfd)
233 {
234 struct mshv_irqfd_resampler *rp = irqfd->irqfd_resampler;
235 struct mshv_partition *pt = rp->rsmplr_partn;
236
237 mutex_lock(&pt->irqfds_resampler_lock);
238
239 hlist_del_rcu(&irqfd->irqfd_resampler_hnode);
240 synchronize_srcu(&pt->pt_irq_srcu);
241
242 if (hlist_empty(&rp->rsmplr_irqfd_list)) {
243 hlist_del(&rp->rsmplr_hnode);
244 mshv_unregister_irq_ack_notifier(pt, &rp->rsmplr_notifier);
245 kfree(rp);
246 }
247
248 mutex_unlock(&pt->irqfds_resampler_lock);
249 }
250
251 /*
252 * Race-free decouple logic (ordering is critical)
253 */
mshv_irqfd_shutdown(struct work_struct * work)254 static void mshv_irqfd_shutdown(struct work_struct *work)
255 {
256 struct mshv_irqfd *irqfd =
257 container_of(work, struct mshv_irqfd, irqfd_shutdown);
258 u64 cnt;
259
260 /*
261 * Synchronize with the wait-queue and unhook ourselves to prevent
262 * further events.
263 */
264 eventfd_ctx_remove_wait_queue(irqfd->irqfd_eventfd_ctx, &irqfd->irqfd_wait, &cnt);
265
266 if (irqfd->irqfd_resampler) {
267 mshv_irqfd_resampler_shutdown(irqfd);
268 eventfd_ctx_put(irqfd->irqfd_resamplefd);
269 }
270
271 /*
272 * It is now safe to release the object's resources
273 */
274 eventfd_ctx_put(irqfd->irqfd_eventfd_ctx);
275 kfree(irqfd);
276 }
277
278 /* assumes partition->pt_irqfds_lock is held */
mshv_irqfd_is_active(struct mshv_irqfd * irqfd)279 static bool mshv_irqfd_is_active(struct mshv_irqfd *irqfd)
280 {
281 return !hlist_unhashed(&irqfd->irqfd_hnode);
282 }
283
284 /*
285 * Mark the irqfd as inactive and schedule it for removal
286 *
287 * assumes partition->pt_irqfds_lock is held
288 */
mshv_irqfd_deactivate(struct mshv_irqfd * irqfd)289 static void mshv_irqfd_deactivate(struct mshv_irqfd *irqfd)
290 {
291 if (!mshv_irqfd_is_active(irqfd))
292 return;
293
294 hlist_del_init(&irqfd->irqfd_hnode);
295
296 queue_work(irqfd_cleanup_wq, &irqfd->irqfd_shutdown);
297 }
298
299 /*
300 * Called with wqh->lock held and interrupts disabled
301 */
mshv_irqfd_wakeup(wait_queue_entry_t * wait,unsigned int mode,int sync,void * key)302 static int mshv_irqfd_wakeup(wait_queue_entry_t *wait, unsigned int mode,
303 int sync, void *key)
304 {
305 struct mshv_irqfd *irqfd = container_of(wait, struct mshv_irqfd,
306 irqfd_wait);
307 __poll_t flags = key_to_poll(key);
308 int idx;
309 unsigned int seq;
310 struct mshv_partition *pt = irqfd->irqfd_partn;
311 int ret = 0;
312
313 if (flags & EPOLLIN) {
314 u64 cnt;
315
316 eventfd_ctx_do_read(irqfd->irqfd_eventfd_ctx, &cnt);
317 idx = srcu_read_lock(&pt->pt_irq_srcu);
318 do {
319 seq = read_seqcount_begin(&irqfd->irqfd_irqe_sc);
320 } while (read_seqcount_retry(&irqfd->irqfd_irqe_sc, seq));
321
322 /* An event has been signaled, raise an interrupt */
323 ret = mshv_try_assert_irq_fast(irqfd);
324 if (ret)
325 mshv_assert_irq_slow(irqfd);
326
327 srcu_read_unlock(&pt->pt_irq_srcu, idx);
328
329 ret = 1;
330 }
331
332 if (flags & EPOLLHUP) {
333 /* The eventfd is closing, detach from the partition */
334 unsigned long flags;
335
336 spin_lock_irqsave(&pt->pt_irqfds_lock, flags);
337
338 /*
339 * We must check if someone deactivated the irqfd before
340 * we could acquire the pt_irqfds_lock since the item is
341 * deactivated from the mshv side before it is unhooked from
342 * the wait-queue. If it is already deactivated, we can
343 * simply return knowing the other side will cleanup for us.
344 * We cannot race against the irqfd going away since the
345 * other side is required to acquire wqh->lock, which we hold
346 */
347 if (mshv_irqfd_is_active(irqfd))
348 mshv_irqfd_deactivate(irqfd);
349
350 spin_unlock_irqrestore(&pt->pt_irqfds_lock, flags);
351 }
352
353 return ret;
354 }
355
356 /* Must be called under pt_irqfds_lock */
mshv_irqfd_update(struct mshv_partition * pt,struct mshv_irqfd * irqfd)357 static void mshv_irqfd_update(struct mshv_partition *pt,
358 struct mshv_irqfd *irqfd)
359 {
360 write_seqcount_begin(&irqfd->irqfd_irqe_sc);
361 irqfd->irqfd_girq_ent = mshv_ret_girq_entry(pt,
362 irqfd->irqfd_irqnum);
363 mshv_copy_girq_info(&irqfd->irqfd_girq_ent, &irqfd->irqfd_lapic_irq);
364 write_seqcount_end(&irqfd->irqfd_irqe_sc);
365 }
366
mshv_irqfd_routing_update(struct mshv_partition * pt)367 void mshv_irqfd_routing_update(struct mshv_partition *pt)
368 {
369 struct mshv_irqfd *irqfd;
370
371 spin_lock_irq(&pt->pt_irqfds_lock);
372 hlist_for_each_entry(irqfd, &pt->pt_irqfds_list, irqfd_hnode)
373 mshv_irqfd_update(pt, irqfd);
374 spin_unlock_irq(&pt->pt_irqfds_lock);
375 }
376
mshv_irqfd_queue_proc(struct file * file,wait_queue_head_t * wqh,poll_table * polltbl)377 static void mshv_irqfd_queue_proc(struct file *file, wait_queue_head_t *wqh,
378 poll_table *polltbl)
379 {
380 struct mshv_irqfd *irqfd =
381 container_of(polltbl, struct mshv_irqfd, irqfd_polltbl);
382
383 /*
384 * TODO: Ensure there isn't already an exclusive, priority waiter, e.g.
385 * that the irqfd isn't already bound to another partition. Only the
386 * first exclusive waiter encountered will be notified, and
387 * add_wait_queue_priority() doesn't enforce exclusivity.
388 */
389 irqfd->irqfd_wait.flags |= WQ_FLAG_EXCLUSIVE;
390 add_wait_queue_priority(wqh, &irqfd->irqfd_wait);
391 }
392
mshv_irqfd_assign(struct mshv_partition * pt,struct mshv_user_irqfd * args)393 static int mshv_irqfd_assign(struct mshv_partition *pt,
394 struct mshv_user_irqfd *args)
395 {
396 struct eventfd_ctx *eventfd = NULL, *resamplefd = NULL;
397 struct mshv_irqfd *irqfd, *tmp;
398 __poll_t events;
399 int ret;
400 int idx;
401
402 CLASS(fd, f)(args->fd);
403
404 irqfd = kzalloc_obj(*irqfd);
405 if (!irqfd)
406 return -ENOMEM;
407
408 irqfd->irqfd_partn = pt;
409 irqfd->irqfd_irqnum = args->gsi;
410 INIT_WORK(&irqfd->irqfd_shutdown, mshv_irqfd_shutdown);
411 seqcount_spinlock_init(&irqfd->irqfd_irqe_sc, &pt->pt_irqfds_lock);
412
413 if (fd_empty(f)) {
414 ret = -EBADF;
415 goto out;
416 }
417
418 eventfd = eventfd_ctx_fileget(fd_file(f));
419 if (IS_ERR(eventfd)) {
420 ret = PTR_ERR(eventfd);
421 goto fail;
422 }
423
424 irqfd->irqfd_eventfd_ctx = eventfd;
425
426 if (args->flags & BIT(MSHV_IRQFD_BIT_RESAMPLE)) {
427 struct mshv_irqfd_resampler *rp;
428
429 resamplefd = eventfd_ctx_fdget(args->resamplefd);
430 if (IS_ERR(resamplefd)) {
431 ret = PTR_ERR(resamplefd);
432 goto fail;
433 }
434
435 irqfd->irqfd_resamplefd = resamplefd;
436
437 mutex_lock(&pt->irqfds_resampler_lock);
438
439 hlist_for_each_entry(rp, &pt->irqfds_resampler_list,
440 rsmplr_hnode) {
441 if (rp->rsmplr_notifier.irq_ack_gsi ==
442 irqfd->irqfd_irqnum) {
443 irqfd->irqfd_resampler = rp;
444 break;
445 }
446 }
447
448 if (!irqfd->irqfd_resampler) {
449 rp = kzalloc_obj(*rp, GFP_KERNEL_ACCOUNT);
450 if (!rp) {
451 ret = -ENOMEM;
452 mutex_unlock(&pt->irqfds_resampler_lock);
453 goto fail;
454 }
455
456 rp->rsmplr_partn = pt;
457 INIT_HLIST_HEAD(&rp->rsmplr_irqfd_list);
458 rp->rsmplr_notifier.irq_ack_gsi = irqfd->irqfd_irqnum;
459 rp->rsmplr_notifier.irq_acked =
460 mshv_irqfd_resampler_ack;
461
462 hlist_add_head(&rp->rsmplr_hnode,
463 &pt->irqfds_resampler_list);
464 mshv_register_irq_ack_notifier(pt,
465 &rp->rsmplr_notifier);
466 irqfd->irqfd_resampler = rp;
467 }
468
469 hlist_add_head_rcu(&irqfd->irqfd_resampler_hnode,
470 &irqfd->irqfd_resampler->rsmplr_irqfd_list);
471
472 mutex_unlock(&pt->irqfds_resampler_lock);
473 }
474
475 /*
476 * Install our own custom wake-up handling so we are notified via
477 * a callback whenever someone signals the underlying eventfd
478 */
479 init_waitqueue_func_entry(&irqfd->irqfd_wait, mshv_irqfd_wakeup);
480 init_poll_funcptr(&irqfd->irqfd_polltbl, mshv_irqfd_queue_proc);
481
482 spin_lock_irq(&pt->pt_irqfds_lock);
483 ret = 0;
484 hlist_for_each_entry(tmp, &pt->pt_irqfds_list, irqfd_hnode) {
485 if (irqfd->irqfd_eventfd_ctx != tmp->irqfd_eventfd_ctx)
486 continue;
487 /* This fd is used for another irq already. */
488 ret = -EBUSY;
489 spin_unlock_irq(&pt->pt_irqfds_lock);
490 goto fail;
491 }
492
493 idx = srcu_read_lock(&pt->pt_irq_srcu);
494 mshv_irqfd_update(pt, irqfd);
495
496 #if IS_ENABLED(CONFIG_X86)
497 if (args->flags & BIT(MSHV_IRQFD_BIT_RESAMPLE) &&
498 !irqfd->irqfd_lapic_irq.lapic_control.level_triggered) {
499 /*
500 * Resample Fd must be for level triggered interrupt
501 * Otherwise return with failure
502 */
503 spin_unlock_irq(&pt->pt_irqfds_lock);
504 srcu_read_unlock(&pt->pt_irq_srcu, idx);
505 ret = -EINVAL;
506 goto fail;
507 }
508 #endif
509
510 hlist_add_head(&irqfd->irqfd_hnode, &pt->pt_irqfds_list);
511 spin_unlock_irq(&pt->pt_irqfds_lock);
512
513 /*
514 * Check if there was an event already pending on the eventfd
515 * before we registered, and trigger it as if we didn't miss it.
516 */
517 events = vfs_poll(fd_file(f), &irqfd->irqfd_polltbl);
518
519 if (events & EPOLLIN)
520 mshv_assert_irq_slow(irqfd);
521
522 srcu_read_unlock(&pt->pt_irq_srcu, idx);
523 return 0;
524
525 fail:
526 if (irqfd->irqfd_resampler)
527 mshv_irqfd_resampler_shutdown(irqfd);
528
529 if (resamplefd && !IS_ERR(resamplefd))
530 eventfd_ctx_put(resamplefd);
531
532 if (eventfd && !IS_ERR(eventfd))
533 eventfd_ctx_put(eventfd);
534
535 out:
536 kfree(irqfd);
537 return ret;
538 }
539
540 /*
541 * shutdown any irqfd's that match fd+gsi
542 */
mshv_irqfd_deassign(struct mshv_partition * pt,struct mshv_user_irqfd * args)543 static int mshv_irqfd_deassign(struct mshv_partition *pt,
544 struct mshv_user_irqfd *args)
545 {
546 struct mshv_irqfd *irqfd;
547 struct hlist_node *n;
548 struct eventfd_ctx *eventfd;
549
550 eventfd = eventfd_ctx_fdget(args->fd);
551 if (IS_ERR(eventfd))
552 return PTR_ERR(eventfd);
553
554 spin_lock_irq(&pt->pt_irqfds_lock);
555 hlist_for_each_entry_safe(irqfd, n, &pt->pt_irqfds_list,
556 irqfd_hnode) {
557 if (irqfd->irqfd_eventfd_ctx == eventfd &&
558 irqfd->irqfd_irqnum == args->gsi)
559 mshv_irqfd_deactivate(irqfd);
560 }
561 spin_unlock_irq(&pt->pt_irqfds_lock);
562
563 eventfd_ctx_put(eventfd);
564
565 /*
566 * Block until we know all outstanding shutdown jobs have completed
567 * so that we guarantee there will not be any more interrupts on this
568 * gsi once this deassign function returns.
569 */
570 flush_workqueue(irqfd_cleanup_wq);
571
572 return 0;
573 }
574
mshv_set_unset_irqfd(struct mshv_partition * pt,struct mshv_user_irqfd * args)575 int mshv_set_unset_irqfd(struct mshv_partition *pt,
576 struct mshv_user_irqfd *args)
577 {
578 if (args->flags & ~MSHV_IRQFD_FLAGS_MASK)
579 return -EINVAL;
580
581 if (args->flags & BIT(MSHV_IRQFD_BIT_DEASSIGN))
582 return mshv_irqfd_deassign(pt, args);
583
584 return mshv_irqfd_assign(pt, args);
585 }
586
587 /*
588 * This function is called as the mshv VM fd is being released.
589 * Shutdown all irqfds that still remain open
590 */
mshv_irqfd_release(struct mshv_partition * pt)591 static void mshv_irqfd_release(struct mshv_partition *pt)
592 {
593 struct mshv_irqfd *irqfd;
594 struct hlist_node *n;
595
596 spin_lock_irq(&pt->pt_irqfds_lock);
597
598 hlist_for_each_entry_safe(irqfd, n, &pt->pt_irqfds_list, irqfd_hnode)
599 mshv_irqfd_deactivate(irqfd);
600
601 spin_unlock_irq(&pt->pt_irqfds_lock);
602
603 /*
604 * Block until we know all outstanding shutdown jobs have completed
605 * since we do not take a mshv_partition* reference.
606 */
607 flush_workqueue(irqfd_cleanup_wq);
608 }
609
mshv_irqfd_wq_init(void)610 int mshv_irqfd_wq_init(void)
611 {
612 irqfd_cleanup_wq = alloc_workqueue("mshv-irqfd-cleanup", WQ_PERCPU, 0);
613 if (!irqfd_cleanup_wq)
614 return -ENOMEM;
615
616 return 0;
617 }
618
mshv_irqfd_wq_cleanup(void)619 void mshv_irqfd_wq_cleanup(void)
620 {
621 destroy_workqueue(irqfd_cleanup_wq);
622 }
623
624 /*
625 * --------------------------------------------------------------------
626 * ioeventfd: translate a MMIO memory write to an eventfd signal.
627 *
628 * userspace can register a MMIO address with an eventfd for receiving
629 * notification when the memory has been touched.
630 * --------------------------------------------------------------------
631 */
632
ioeventfd_release(struct mshv_ioeventfd * p,u64 partition_id)633 static void ioeventfd_release(struct mshv_ioeventfd *p, u64 partition_id)
634 {
635 if (p->iovntfd_doorbell_id > 0)
636 mshv_unregister_doorbell(partition_id, p->iovntfd_doorbell_id);
637 eventfd_ctx_put(p->iovntfd_eventfd);
638 kfree(p);
639 }
640
641 /* MMIO writes trigger an event if the addr/val match */
ioeventfd_mmio_write(int doorbell_id,void * data)642 static void ioeventfd_mmio_write(int doorbell_id, void *data)
643 {
644 struct mshv_partition *partition = (struct mshv_partition *)data;
645 struct mshv_ioeventfd *p;
646
647 rcu_read_lock();
648 hlist_for_each_entry_rcu(p, &partition->ioeventfds_list, iovntfd_hnode)
649 if (p->iovntfd_doorbell_id == doorbell_id) {
650 eventfd_signal(p->iovntfd_eventfd);
651 break;
652 }
653
654 rcu_read_unlock();
655 }
656
ioeventfd_check_collision(struct mshv_partition * pt,struct mshv_ioeventfd * p)657 static bool ioeventfd_check_collision(struct mshv_partition *pt,
658 struct mshv_ioeventfd *p)
659 __must_hold(&pt->mutex)
660 {
661 struct mshv_ioeventfd *_p;
662
663 hlist_for_each_entry(_p, &pt->ioeventfds_list, iovntfd_hnode)
664 if (_p->iovntfd_addr == p->iovntfd_addr &&
665 _p->iovntfd_length == p->iovntfd_length &&
666 (_p->iovntfd_wildcard || p->iovntfd_wildcard ||
667 _p->iovntfd_datamatch == p->iovntfd_datamatch))
668 return true;
669
670 return false;
671 }
672
mshv_assign_ioeventfd(struct mshv_partition * pt,struct mshv_user_ioeventfd * args)673 static int mshv_assign_ioeventfd(struct mshv_partition *pt,
674 struct mshv_user_ioeventfd *args)
675 __must_hold(&pt->mutex)
676 {
677 struct mshv_ioeventfd *p;
678 struct eventfd_ctx *eventfd;
679 u64 doorbell_flags = 0;
680 int ret;
681
682 /* This mutex is currently protecting ioeventfd.items list */
683 WARN_ON_ONCE(!mutex_is_locked(&pt->pt_mutex));
684
685 if (args->flags & BIT(MSHV_IOEVENTFD_BIT_PIO))
686 return -EOPNOTSUPP;
687
688 /* must be natural-word sized */
689 switch (args->len) {
690 case 0:
691 doorbell_flags = HV_DOORBELL_FLAG_TRIGGER_SIZE_ANY;
692 break;
693 case 1:
694 doorbell_flags = HV_DOORBELL_FLAG_TRIGGER_SIZE_BYTE;
695 break;
696 case 2:
697 doorbell_flags = HV_DOORBELL_FLAG_TRIGGER_SIZE_WORD;
698 break;
699 case 4:
700 doorbell_flags = HV_DOORBELL_FLAG_TRIGGER_SIZE_DWORD;
701 break;
702 case 8:
703 doorbell_flags = HV_DOORBELL_FLAG_TRIGGER_SIZE_QWORD;
704 break;
705 default:
706 return -EINVAL;
707 }
708
709 /* check for range overflow */
710 if (args->addr + args->len < args->addr)
711 return -EINVAL;
712
713 /* check for extra flags that we don't understand */
714 if (args->flags & ~MSHV_IOEVENTFD_FLAGS_MASK)
715 return -EINVAL;
716
717 eventfd = eventfd_ctx_fdget(args->fd);
718 if (IS_ERR(eventfd))
719 return PTR_ERR(eventfd);
720
721 p = kzalloc_obj(*p);
722 if (!p) {
723 ret = -ENOMEM;
724 goto fail;
725 }
726
727 p->iovntfd_addr = args->addr;
728 p->iovntfd_length = args->len;
729 p->iovntfd_eventfd = eventfd;
730
731 /* The datamatch feature is optional, otherwise this is a wildcard */
732 if (args->flags & BIT(MSHV_IOEVENTFD_BIT_DATAMATCH)) {
733 p->iovntfd_datamatch = args->datamatch;
734 } else {
735 p->iovntfd_wildcard = true;
736 doorbell_flags |= HV_DOORBELL_FLAG_TRIGGER_ANY_VALUE;
737 }
738
739 if (ioeventfd_check_collision(pt, p)) {
740 ret = -EEXIST;
741 goto unlock_fail;
742 }
743
744 ret = mshv_register_doorbell(pt->pt_id, ioeventfd_mmio_write,
745 (void *)pt, p->iovntfd_addr,
746 p->iovntfd_datamatch, doorbell_flags);
747
748 trace_mshv_assign_ioeventfd(pt->pt_id, p->iovntfd_addr,
749 p->iovntfd_length,
750 p->iovntfd_datamatch,
751 p->iovntfd_wildcard,
752 p->iovntfd_eventfd,
753 ret);
754
755 if (ret < 0)
756 goto unlock_fail;
757
758 p->iovntfd_doorbell_id = ret;
759
760 hlist_add_head_rcu(&p->iovntfd_hnode, &pt->ioeventfds_list);
761
762 return 0;
763
764 unlock_fail:
765 kfree(p);
766
767 fail:
768 eventfd_ctx_put(eventfd);
769
770 return ret;
771 }
772
mshv_deassign_ioeventfd(struct mshv_partition * pt,struct mshv_user_ioeventfd * args)773 static int mshv_deassign_ioeventfd(struct mshv_partition *pt,
774 struct mshv_user_ioeventfd *args)
775 __must_hold(&pt->mutex)
776 {
777 struct mshv_ioeventfd *p;
778 struct eventfd_ctx *eventfd;
779 struct hlist_node *n;
780 int ret = -ENOENT;
781
782 /* This mutex is currently protecting ioeventfd.items list */
783 WARN_ON_ONCE(!mutex_is_locked(&pt->pt_mutex));
784
785 eventfd = eventfd_ctx_fdget(args->fd);
786 if (IS_ERR(eventfd))
787 return PTR_ERR(eventfd);
788
789 hlist_for_each_entry_safe(p, n, &pt->ioeventfds_list, iovntfd_hnode) {
790 bool wildcard = !(args->flags & BIT(MSHV_IOEVENTFD_BIT_DATAMATCH));
791
792 if (p->iovntfd_eventfd != eventfd ||
793 p->iovntfd_addr != args->addr ||
794 p->iovntfd_length != args->len ||
795 p->iovntfd_wildcard != wildcard)
796 continue;
797
798 if (!p->iovntfd_wildcard &&
799 p->iovntfd_datamatch != args->datamatch)
800 continue;
801
802 trace_mshv_deassign_ioeventfd(pt->pt_id, p->iovntfd_addr,
803 p->iovntfd_length,
804 p->iovntfd_datamatch,
805 p->iovntfd_wildcard,
806 p->iovntfd_eventfd);
807
808 hlist_del_rcu(&p->iovntfd_hnode);
809 synchronize_rcu();
810 ioeventfd_release(p, pt->pt_id);
811 ret = 0;
812 break;
813 }
814
815 eventfd_ctx_put(eventfd);
816
817 return ret;
818 }
819
mshv_set_unset_ioeventfd(struct mshv_partition * pt,struct mshv_user_ioeventfd * args)820 int mshv_set_unset_ioeventfd(struct mshv_partition *pt,
821 struct mshv_user_ioeventfd *args)
822 __must_hold(&pt->mutex)
823 {
824 if ((args->flags & ~MSHV_IOEVENTFD_FLAGS_MASK) ||
825 mshv_field_nonzero(*args, rsvd))
826 return -EINVAL;
827
828 /* PIO not yet implemented */
829 if (args->flags & BIT(MSHV_IOEVENTFD_BIT_PIO))
830 return -EOPNOTSUPP;
831
832 if (args->flags & BIT(MSHV_IOEVENTFD_BIT_DEASSIGN))
833 return mshv_deassign_ioeventfd(pt, args);
834
835 return mshv_assign_ioeventfd(pt, args);
836 }
837
mshv_eventfd_init(struct mshv_partition * pt)838 void mshv_eventfd_init(struct mshv_partition *pt)
839 {
840 spin_lock_init(&pt->pt_irqfds_lock);
841 INIT_HLIST_HEAD(&pt->pt_irqfds_list);
842
843 INIT_HLIST_HEAD(&pt->irqfds_resampler_list);
844 mutex_init(&pt->irqfds_resampler_lock);
845
846 INIT_HLIST_HEAD(&pt->ioeventfds_list);
847 }
848
mshv_eventfd_release(struct mshv_partition * pt)849 void mshv_eventfd_release(struct mshv_partition *pt)
850 {
851 struct hlist_head items;
852 struct hlist_node *n;
853 struct mshv_ioeventfd *p;
854
855 hlist_move_list(&pt->ioeventfds_list, &items);
856 synchronize_rcu();
857
858 hlist_for_each_entry_safe(p, n, &items, iovntfd_hnode) {
859 hlist_del(&p->iovntfd_hnode);
860 ioeventfd_release(p, pt->pt_id);
861 }
862
863 mshv_irqfd_release(pt);
864 }
865