1 /*
2 * CDDL HEADER START
3 *
4 * The contents of this file are subject to the terms of the
5 * Common Development and Distribution License (the "License").
6 * You may not use this file except in compliance with the License.
7 *
8 * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
9 * or http://www.opensolaris.org/os/licensing.
10 * See the License for the specific language governing permissions
11 * and limitations under the License.
12 *
13 * When distributing Covered Code, include this CDDL HEADER in each
14 * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
15 * If applicable, add the following below this CDDL HEADER, with the
16 * fields enclosed by brackets "[]" replaced with your own identifying
17 * information: Portions Copyright [yyyy] [name of copyright owner]
18 *
19 * CDDL HEADER END
20 */
21
22 /*
23 * Copyright 2009 Sun Microsystems, Inc. All rights reserved.
24 * Use is subject to license terms.
25 * Copyright 2026 Oxide Computer Company
26 */
27
28 /*
29 * vm_usage
30 *
31 * This file implements the getvmusage() private system call.
32 * getvmusage() counts the amount of resident memory pages and swap
33 * reserved by the specified process collective. A "process collective" is
34 * the set of processes owned by a particular, zone, project, task, or user.
35 *
36 * rss and swap are counted so that for a given process collective, a page is
37 * only counted once. For example, this means that if multiple processes in
38 * the same project map the same page, then the project will only be charged
39 * once for that page. On the other hand, if two processes in different
40 * projects map the same page, then both projects will be charged
41 * for the page.
42 *
43 * The vm_getusage() calculation is implemented so that the first thread
44 * performs the rss/swap counting. Other callers will wait for that thread to
45 * finish, copying the results. This enables multiple rcapds and prstats to
46 * consume data from the same calculation. The results are also cached so that
47 * a caller interested in recent results can just copy them instead of starting
48 * a new calculation. The caller passes the maximium age (in seconds) of the
49 * data. If the cached data is young enough, the cache is copied, otherwise,
50 * a new calculation is executed and the cache is replaced with the new
51 * data.
52 *
53 * The rss calculation for each process collective is as follows:
54 *
55 * - Inspect flags, determine if counting rss for zones, projects, tasks,
56 * and/or users.
57 * - For each proc:
58 * - Figure out proc's collectives (zone, project, task, and/or user).
59 * - For each seg in proc's address space:
60 * - If seg is private:
61 * - Lookup anons in the amp.
62 * - For incore pages not previously visited each of the
63 * proc's collectives, add incore pagesize to each.
64 * collective.
65 * Anon's with a refcnt of 1 can be assummed to be not
66 * previously visited.
67 * - For address ranges without anons in the amp:
68 * - Lookup pages in underlying vnode.
69 * - For incore pages not previously visiting for
70 * each of the proc's collectives, add incore
71 * pagesize to each collective.
72 * - If seg is shared:
73 * - Lookup pages in the shared amp or vnode.
74 * - For incore pages not previously visited for each of
75 * the proc's collectives, add incore pagesize to each
76 * collective.
77 *
78 * Swap is reserved by private segments, and shared anonymous segments.
79 * The only shared anon segments which do not reserve swap are ISM segments
80 * and schedctl segments, both of which can be identified by having
81 * amp->swresv == 0.
82 *
83 * The swap calculation for each collective is as follows:
84 *
85 * - Inspect flags, determine if counting rss for zones, projects, tasks,
86 * and/or users.
87 * - For each proc:
88 * - Figure out proc's collectives (zone, project, task, and/or user).
89 * - For each seg in proc's address space:
90 * - If seg is private:
91 * - Add svd->swresv pages to swap count for each of the
92 * proc's collectives.
93 * - If seg is anon, shared, and amp->swresv != 0
94 * - For address ranges in amp not previously visited for
95 * each of the proc's collectives, add size of address
96 * range to the swap count for each collective.
97 *
98 * These two calculations are done simultaneously, with most of the work
99 * being done in vmu_calculate_seg(). The results of the calculation are
100 * copied into "vmu_data.vmu_cache_results".
101 *
102 * To perform the calculation, various things are tracked and cached:
103 *
104 * - incore/not-incore page ranges for all vnodes.
105 * (vmu_data.vmu_all_vnodes_hash)
106 * This eliminates looking up the same page more than once.
107 *
108 * - incore/not-incore page ranges for all shared amps.
109 * (vmu_data.vmu_all_amps_hash)
110 * This eliminates looking up the same page more than once.
111 *
112 * - visited page ranges for each collective.
113 * - per vnode (entity->vme_vnode_hash)
114 * - per shared amp (entity->vme_amp_hash)
115 * For accurate counting of map-shared and COW-shared pages.
116 *
117 * - visited private anons (refcnt > 1) for each collective.
118 * (entity->vme_anon_hash)
119 * For accurate counting of COW-shared pages.
120 *
121 * The common accounting structure is the vmu_entity_t, which represents
122 * collectives:
123 *
124 * - A zone.
125 * - A project, task, or user within a zone.
126 * - The entire system (vmu_data.vmu_system).
127 * - Each collapsed (col) project and user. This means a given projid or
128 * uid, regardless of which zone the process is in. For instance,
129 * project 0 in the global zone and project 0 in a non global zone are
130 * the same collapsed project.
131 *
132 * Each entity structure tracks which pages have been already visited for
133 * that entity (via previously inspected processes) so that these pages are
134 * not double counted.
135 */
136
137 #include <sys/errno.h>
138 #include <sys/types.h>
139 #include <sys/zone.h>
140 #include <sys/proc.h>
141 #include <sys/project.h>
142 #include <sys/task.h>
143 #include <sys/thread.h>
144 #include <sys/time.h>
145 #include <sys/mman.h>
146 #include <sys/modhash.h>
147 #include <sys/modhash_impl.h>
148 #include <sys/shm.h>
149 #include <sys/swap.h>
150 #include <sys/synch.h>
151 #include <sys/systm.h>
152 #include <sys/uio.h>
153 #include <sys/var.h>
154 #include <sys/vm_usage.h>
155 #include <sys/zone.h>
156 #include <sys/sunddi.h>
157 #include <sys/avl.h>
158 #include <vm/anon.h>
159 #include <vm/as.h>
160 #include <vm/seg_vn.h>
161 #include <vm/seg_spt.h>
162
163 #define VMUSAGE_HASH_SIZE 512
164
165 #define VMUSAGE_TYPE_VNODE 1
166 #define VMUSAGE_TYPE_AMP 2
167 #define VMUSAGE_TYPE_ANON 3
168
169 #define VMUSAGE_BOUND_UNKNOWN 0
170 #define VMUSAGE_BOUND_INCORE 1
171 #define VMUSAGE_BOUND_NOT_INCORE 2
172
173 #define ISWITHIN(node, addr) ((node)->vmb_start <= addr && \
174 (node)->vmb_end >= addr ? 1 : 0)
175
176 /*
177 * bounds for vnodes and shared amps
178 * Each bound is either entirely incore, entirely not in core, or
179 * entirely unknown. bounds are stored in an avl tree sorted by start member
180 * when in use, otherwise (free or temporary lists) they're strung
181 * together off of vmb_next.
182 */
183 typedef struct vmu_bound {
184 avl_node_t vmb_node;
185 struct vmu_bound *vmb_next; /* NULL in tree else on free or temp list */
186 pgcnt_t vmb_start; /* page offset in vnode/amp on which bound starts */
187 pgcnt_t vmb_end; /* page offset in vnode/amp on which bound ends */
188 char vmb_type; /* One of VMUSAGE_BOUND_* */
189 } vmu_bound_t;
190
191 /*
192 * hash of visited objects (vnodes or shared amps)
193 * key is address of vnode or amp. Bounds lists known incore/non-incore
194 * bounds for vnode/amp.
195 */
196 typedef struct vmu_object {
197 struct vmu_object *vmo_next; /* free list */
198 caddr_t vmo_key;
199 short vmo_type;
200 avl_tree_t vmo_bounds;
201 } vmu_object_t;
202
203 /*
204 * Entity by which to count results.
205 *
206 * The entity structure keeps the current rss/swap counts for each entity
207 * (zone, project, etc), and hashes of vm structures that have already
208 * been visited for the entity.
209 *
210 * vme_next: links the list of all entities currently being counted by
211 * vmu_calculate().
212 *
213 * vme_next_calc: links the list of entities related to the current process
214 * being counted by vmu_calculate_proc().
215 *
216 * vmu_calculate_proc() walks all processes. For each process, it makes a
217 * list of the entities related to that process using vme_next_calc. This
218 * list changes each time vmu_calculate_proc() is called.
219 *
220 */
221 typedef struct vmu_entity {
222 struct vmu_entity *vme_next;
223 struct vmu_entity *vme_next_calc;
224 mod_hash_t *vme_vnode_hash; /* vnodes visited for entity */
225 mod_hash_t *vme_amp_hash; /* shared amps visited for entity */
226 mod_hash_t *vme_anon_hash; /* COW anons visited for entity */
227 vmusage_t vme_result; /* identifies entity and results */
228 } vmu_entity_t;
229
230 /*
231 * Hash of entities visited within a zone, and an entity for the zone
232 * itself.
233 */
234 typedef struct vmu_zone {
235 struct vmu_zone *vmz_next; /* free list */
236 id_t vmz_id;
237 vmu_entity_t *vmz_zone;
238 mod_hash_t *vmz_projects_hash;
239 mod_hash_t *vmz_tasks_hash;
240 mod_hash_t *vmz_rusers_hash;
241 mod_hash_t *vmz_eusers_hash;
242 } vmu_zone_t;
243
244 /*
245 * Cache of results from last calculation
246 */
247 typedef struct vmu_cache {
248 vmusage_t *vmc_results; /* Results from last call to */
249 /* vm_getusage(). */
250 uint64_t vmc_nresults; /* Count of cached results */
251 uint64_t vmc_refcnt; /* refcnt for free */
252 uint_t vmc_flags; /* Flags for vm_getusage() */
253 hrtime_t vmc_timestamp; /* when cache was created */
254 } vmu_cache_t;
255
256 /*
257 * top level rss info for the system
258 */
259 typedef struct vmu_data {
260 kmutex_t vmu_lock; /* Protects vmu_data */
261 kcondvar_t vmu_cv; /* Used to signal threads */
262 /* Waiting for */
263 /* Rss_calc_thread to finish */
264 vmu_entity_t *vmu_system; /* Entity for tracking */
265 /* rss/swap for all processes */
266 /* in all zones */
267 mod_hash_t *vmu_zones_hash; /* Zones visited */
268 mod_hash_t *vmu_projects_col_hash; /* These *_col_hash hashes */
269 mod_hash_t *vmu_rusers_col_hash; /* keep track of entities, */
270 mod_hash_t *vmu_eusers_col_hash; /* ignoring zoneid, in order */
271 /* to implement VMUSAGE_COL_* */
272 /* flags, which aggregate by */
273 /* project or user regardless */
274 /* of zoneid. */
275 mod_hash_t *vmu_all_vnodes_hash; /* System wide visited vnodes */
276 /* to track incore/not-incore */
277 mod_hash_t *vmu_all_amps_hash; /* System wide visited shared */
278 /* amps to track incore/not- */
279 /* incore */
280 vmu_entity_t *vmu_entities; /* Linked list of entities */
281 size_t vmu_nentities; /* Count of entities in list */
282 vmu_cache_t *vmu_cache; /* Cached results */
283 kthread_t *vmu_calc_thread; /* NULL, or thread running */
284 /* vmu_calculate() */
285 uint_t vmu_calc_flags; /* Flags being using by */
286 /* currently running calc */
287 /* thread */
288 uint_t vmu_pending_flags; /* Flags of vm_getusage() */
289 /* threads waiting for */
290 /* calc thread to finish */
291 uint_t vmu_pending_waiters; /* Number of threads waiting */
292 /* for calc thread */
293 vmu_bound_t *vmu_free_bounds;
294 vmu_object_t *vmu_free_objects;
295 vmu_entity_t *vmu_free_entities;
296 vmu_zone_t *vmu_free_zones;
297 } vmu_data_t;
298
299 extern struct as kas;
300 extern proc_t *practive;
301 extern zone_t *global_zone;
302 extern struct seg_ops segvn_ops;
303 extern struct seg_ops segspt_shmops;
304
305 static vmu_data_t vmu_data;
306 static kmem_cache_t *vmu_bound_cache;
307 static kmem_cache_t *vmu_object_cache;
308
309 /*
310 * Comparison routine for AVL tree. We base our comparison on vmb_start.
311 */
312 static int
bounds_cmp(const void * bnd1,const void * bnd2)313 bounds_cmp(const void *bnd1, const void *bnd2)
314 {
315 const vmu_bound_t *bound1 = bnd1;
316 const vmu_bound_t *bound2 = bnd2;
317
318 if (bound1->vmb_start == bound2->vmb_start) {
319 return (0);
320 }
321 if (bound1->vmb_start < bound2->vmb_start) {
322 return (-1);
323 }
324
325 return (1);
326 }
327
328 /*
329 * Save a bound on the free list.
330 */
331 static void
vmu_free_bound(vmu_bound_t * bound)332 vmu_free_bound(vmu_bound_t *bound)
333 {
334 bound->vmb_next = vmu_data.vmu_free_bounds;
335 bound->vmb_start = 0;
336 bound->vmb_end = 0;
337 bound->vmb_type = 0;
338 vmu_data.vmu_free_bounds = bound;
339 }
340
341 /*
342 * Free an object, and all visited bound info.
343 */
344 static void
vmu_free_object(mod_hash_val_t val)345 vmu_free_object(mod_hash_val_t val)
346 {
347 vmu_object_t *obj = (vmu_object_t *)val;
348 avl_tree_t *tree = &(obj->vmo_bounds);
349 vmu_bound_t *bound;
350 void *cookie = NULL;
351
352 while ((bound = avl_destroy_nodes(tree, &cookie)) != NULL)
353 vmu_free_bound(bound);
354 avl_destroy(tree);
355
356 obj->vmo_type = 0;
357 obj->vmo_next = vmu_data.vmu_free_objects;
358 vmu_data.vmu_free_objects = obj;
359 }
360
361 /*
362 * Free an entity, and hashes of visited objects for that entity.
363 */
364 static void
vmu_free_entity(mod_hash_val_t val)365 vmu_free_entity(mod_hash_val_t val)
366 {
367 vmu_entity_t *entity = (vmu_entity_t *)val;
368
369 if (entity->vme_vnode_hash != NULL)
370 i_mod_hash_clear_nosync(entity->vme_vnode_hash);
371 if (entity->vme_amp_hash != NULL)
372 i_mod_hash_clear_nosync(entity->vme_amp_hash);
373 if (entity->vme_anon_hash != NULL)
374 i_mod_hash_clear_nosync(entity->vme_anon_hash);
375
376 entity->vme_next = vmu_data.vmu_free_entities;
377 vmu_data.vmu_free_entities = entity;
378 }
379
380 /*
381 * Free zone entity, and all hashes of entities inside that zone,
382 * which are projects, tasks, and users.
383 */
384 static void
vmu_free_zone(mod_hash_val_t val)385 vmu_free_zone(mod_hash_val_t val)
386 {
387 vmu_zone_t *zone = (vmu_zone_t *)val;
388
389 if (zone->vmz_zone != NULL) {
390 vmu_free_entity((mod_hash_val_t)zone->vmz_zone);
391 zone->vmz_zone = NULL;
392 }
393 if (zone->vmz_projects_hash != NULL)
394 i_mod_hash_clear_nosync(zone->vmz_projects_hash);
395 if (zone->vmz_tasks_hash != NULL)
396 i_mod_hash_clear_nosync(zone->vmz_tasks_hash);
397 if (zone->vmz_rusers_hash != NULL)
398 i_mod_hash_clear_nosync(zone->vmz_rusers_hash);
399 if (zone->vmz_eusers_hash != NULL)
400 i_mod_hash_clear_nosync(zone->vmz_eusers_hash);
401 zone->vmz_next = vmu_data.vmu_free_zones;
402 vmu_data.vmu_free_zones = zone;
403 }
404
405 /*
406 * Initialize synchronization primitives and hashes for system-wide tracking
407 * of visited vnodes and shared amps. Initialize results cache.
408 */
409 void
vm_usage_init()410 vm_usage_init()
411 {
412 mutex_init(&vmu_data.vmu_lock, NULL, MUTEX_DEFAULT, NULL);
413 cv_init(&vmu_data.vmu_cv, NULL, CV_DEFAULT, NULL);
414
415 vmu_data.vmu_system = NULL;
416 vmu_data.vmu_zones_hash = NULL;
417 vmu_data.vmu_projects_col_hash = NULL;
418 vmu_data.vmu_rusers_col_hash = NULL;
419 vmu_data.vmu_eusers_col_hash = NULL;
420
421 vmu_data.vmu_free_bounds = NULL;
422 vmu_data.vmu_free_objects = NULL;
423 vmu_data.vmu_free_entities = NULL;
424 vmu_data.vmu_free_zones = NULL;
425
426 vmu_data.vmu_all_vnodes_hash = mod_hash_create_ptrhash(
427 "vmusage vnode hash", VMUSAGE_HASH_SIZE, vmu_free_object,
428 sizeof (vnode_t));
429 vmu_data.vmu_all_amps_hash = mod_hash_create_ptrhash(
430 "vmusage amp hash", VMUSAGE_HASH_SIZE, vmu_free_object,
431 sizeof (struct anon_map));
432 vmu_data.vmu_projects_col_hash = mod_hash_create_idhash(
433 "vmusage collapsed project hash", VMUSAGE_HASH_SIZE,
434 vmu_free_entity);
435 vmu_data.vmu_rusers_col_hash = mod_hash_create_idhash(
436 "vmusage collapsed ruser hash", VMUSAGE_HASH_SIZE,
437 vmu_free_entity);
438 vmu_data.vmu_eusers_col_hash = mod_hash_create_idhash(
439 "vmusage collpased euser hash", VMUSAGE_HASH_SIZE,
440 vmu_free_entity);
441 vmu_data.vmu_zones_hash = mod_hash_create_idhash(
442 "vmusage zone hash", VMUSAGE_HASH_SIZE, vmu_free_zone);
443
444 vmu_bound_cache = kmem_cache_create("vmu_bound_cache",
445 sizeof (vmu_bound_t), 0, NULL, NULL, NULL, NULL, NULL, 0);
446 vmu_object_cache = kmem_cache_create("vmu_object_cache",
447 sizeof (vmu_object_t), 0, NULL, NULL, NULL, NULL, NULL, 0);
448
449 vmu_data.vmu_entities = NULL;
450 vmu_data.vmu_nentities = 0;
451
452 vmu_data.vmu_cache = NULL;
453 vmu_data.vmu_calc_thread = NULL;
454 vmu_data.vmu_calc_flags = 0;
455 vmu_data.vmu_pending_flags = 0;
456 vmu_data.vmu_pending_waiters = 0;
457 }
458
459 /*
460 * Allocate hashes for tracking vm objects visited for an entity.
461 * Update list of entities.
462 */
463 static vmu_entity_t *
vmu_alloc_entity(id_t id,int type,id_t zoneid)464 vmu_alloc_entity(id_t id, int type, id_t zoneid)
465 {
466 vmu_entity_t *entity;
467
468 if (vmu_data.vmu_free_entities != NULL) {
469 entity = vmu_data.vmu_free_entities;
470 vmu_data.vmu_free_entities =
471 vmu_data.vmu_free_entities->vme_next;
472 bzero(&entity->vme_result, sizeof (vmusage_t));
473 } else {
474 entity = kmem_zalloc(sizeof (vmu_entity_t), KM_SLEEP);
475 }
476 entity->vme_result.vmu_id = id;
477 entity->vme_result.vmu_zoneid = zoneid;
478 entity->vme_result.vmu_type = type;
479
480 if (entity->vme_vnode_hash == NULL)
481 entity->vme_vnode_hash = mod_hash_create_ptrhash(
482 "vmusage vnode hash", VMUSAGE_HASH_SIZE, vmu_free_object,
483 sizeof (vnode_t));
484
485 if (entity->vme_amp_hash == NULL)
486 entity->vme_amp_hash = mod_hash_create_ptrhash(
487 "vmusage amp hash", VMUSAGE_HASH_SIZE, vmu_free_object,
488 sizeof (struct anon_map));
489
490 if (entity->vme_anon_hash == NULL)
491 entity->vme_anon_hash = mod_hash_create_ptrhash(
492 "vmusage anon hash", VMUSAGE_HASH_SIZE,
493 mod_hash_null_valdtor, sizeof (struct anon));
494
495 entity->vme_next = vmu_data.vmu_entities;
496 vmu_data.vmu_entities = entity;
497 vmu_data.vmu_nentities++;
498
499 return (entity);
500 }
501
502 /*
503 * Allocate a zone entity, and hashes for tracking visited vm objects
504 * for projects, tasks, and users within that zone.
505 */
506 static vmu_zone_t *
vmu_alloc_zone(id_t id)507 vmu_alloc_zone(id_t id)
508 {
509 vmu_zone_t *zone;
510
511 if (vmu_data.vmu_free_zones != NULL) {
512 zone = vmu_data.vmu_free_zones;
513 vmu_data.vmu_free_zones =
514 vmu_data.vmu_free_zones->vmz_next;
515 zone->vmz_next = NULL;
516 zone->vmz_zone = NULL;
517 } else {
518 zone = kmem_zalloc(sizeof (vmu_zone_t), KM_SLEEP);
519 }
520
521 zone->vmz_id = id;
522
523 if ((vmu_data.vmu_calc_flags & (VMUSAGE_ZONE | VMUSAGE_ALL_ZONES)) != 0)
524 zone->vmz_zone = vmu_alloc_entity(id, VMUSAGE_ZONE, id);
525
526 if ((vmu_data.vmu_calc_flags & (VMUSAGE_PROJECTS |
527 VMUSAGE_ALL_PROJECTS)) != 0 && zone->vmz_projects_hash == NULL)
528 zone->vmz_projects_hash = mod_hash_create_idhash(
529 "vmusage project hash", VMUSAGE_HASH_SIZE, vmu_free_entity);
530
531 if ((vmu_data.vmu_calc_flags & (VMUSAGE_TASKS | VMUSAGE_ALL_TASKS))
532 != 0 && zone->vmz_tasks_hash == NULL)
533 zone->vmz_tasks_hash = mod_hash_create_idhash(
534 "vmusage task hash", VMUSAGE_HASH_SIZE, vmu_free_entity);
535
536 if ((vmu_data.vmu_calc_flags & (VMUSAGE_RUSERS | VMUSAGE_ALL_RUSERS))
537 != 0 && zone->vmz_rusers_hash == NULL)
538 zone->vmz_rusers_hash = mod_hash_create_idhash(
539 "vmusage ruser hash", VMUSAGE_HASH_SIZE, vmu_free_entity);
540
541 if ((vmu_data.vmu_calc_flags & (VMUSAGE_EUSERS | VMUSAGE_ALL_EUSERS))
542 != 0 && zone->vmz_eusers_hash == NULL)
543 zone->vmz_eusers_hash = mod_hash_create_idhash(
544 "vmusage euser hash", VMUSAGE_HASH_SIZE, vmu_free_entity);
545
546 return (zone);
547 }
548
549 /*
550 * Allocate a structure for tracking visited bounds for a vm object.
551 */
552 static vmu_object_t *
vmu_alloc_object(caddr_t key,int type)553 vmu_alloc_object(caddr_t key, int type)
554 {
555 vmu_object_t *object;
556
557 if (vmu_data.vmu_free_objects != NULL) {
558 object = vmu_data.vmu_free_objects;
559 vmu_data.vmu_free_objects =
560 vmu_data.vmu_free_objects->vmo_next;
561 } else {
562 object = kmem_cache_alloc(vmu_object_cache, KM_SLEEP);
563 }
564
565 object->vmo_next = NULL;
566 object->vmo_key = key;
567 object->vmo_type = type;
568 avl_create(&(object->vmo_bounds), bounds_cmp, sizeof (vmu_bound_t), 0);
569
570 return (object);
571 }
572
573 /*
574 * Allocate and return a bound structure.
575 */
576 static vmu_bound_t *
vmu_alloc_bound()577 vmu_alloc_bound()
578 {
579 vmu_bound_t *bound;
580
581 if (vmu_data.vmu_free_bounds != NULL) {
582 bound = vmu_data.vmu_free_bounds;
583 vmu_data.vmu_free_bounds =
584 vmu_data.vmu_free_bounds->vmb_next;
585 } else {
586 bound = kmem_cache_alloc(vmu_bound_cache, KM_SLEEP);
587 }
588
589 bound->vmb_next = NULL;
590 bound->vmb_start = 0;
591 bound->vmb_end = 0;
592 bound->vmb_type = 0;
593 return (bound);
594 }
595
596 /*
597 * vmu_find_insert_* functions implement hash lookup or allocate and
598 * insert operations.
599 */
600 static vmu_object_t *
vmu_find_insert_object(mod_hash_t * hash,caddr_t key,uint_t type)601 vmu_find_insert_object(mod_hash_t *hash, caddr_t key, uint_t type)
602 {
603 int ret;
604 vmu_object_t *object;
605
606 ret = i_mod_hash_find_nosync(hash, (mod_hash_key_t)key,
607 (mod_hash_val_t *)&object);
608 if (ret != 0) {
609 object = vmu_alloc_object(key, type);
610 ret = i_mod_hash_insert_nosync(hash, (mod_hash_key_t)key,
611 (mod_hash_val_t)object, (mod_hash_hndl_t)0);
612 ASSERT(ret == 0);
613 }
614 return (object);
615 }
616
617 static int
vmu_find_insert_anon(mod_hash_t * hash,caddr_t key)618 vmu_find_insert_anon(mod_hash_t *hash, caddr_t key)
619 {
620 int ret;
621 caddr_t val;
622
623 ret = i_mod_hash_find_nosync(hash, (mod_hash_key_t)key,
624 (mod_hash_val_t *)&val);
625
626 if (ret == 0)
627 return (0);
628
629 ret = i_mod_hash_insert_nosync(hash, (mod_hash_key_t)key,
630 (mod_hash_val_t)key, (mod_hash_hndl_t)0);
631
632 ASSERT(ret == 0);
633
634 return (1);
635 }
636
637 static vmu_entity_t *
vmu_find_insert_entity(mod_hash_t * hash,id_t id,uint_t type,id_t zoneid)638 vmu_find_insert_entity(mod_hash_t *hash, id_t id, uint_t type, id_t zoneid)
639 {
640 int ret;
641 vmu_entity_t *entity;
642
643 ret = i_mod_hash_find_nosync(hash, (mod_hash_key_t)(uintptr_t)id,
644 (mod_hash_val_t *)&entity);
645 if (ret != 0) {
646 entity = vmu_alloc_entity(id, type, zoneid);
647 ret = i_mod_hash_insert_nosync(hash,
648 (mod_hash_key_t)(uintptr_t)id, (mod_hash_val_t)entity,
649 (mod_hash_hndl_t)0);
650 ASSERT(ret == 0);
651 }
652 return (entity);
653 }
654
655
656
657
658 /*
659 * Returns list of object bounds between start and end. New bounds inserted
660 * by this call are given type.
661 *
662 * Returns the number of pages covered if new bounds are created. Returns 0
663 * if region between start/end consists of all existing bounds.
664 */
665 static pgcnt_t
vmu_insert_lookup_object_bounds(vmu_object_t * ro,pgcnt_t start,pgcnt_t end,char type,vmu_bound_t ** first,vmu_bound_t ** last)666 vmu_insert_lookup_object_bounds(vmu_object_t *ro, pgcnt_t start, pgcnt_t
667 end, char type, vmu_bound_t **first, vmu_bound_t **last)
668 {
669 avl_tree_t *tree = &(ro->vmo_bounds);
670 avl_index_t where;
671 vmu_bound_t *walker, *tmp;
672 pgcnt_t ret = 0;
673
674 ASSERT(start <= end);
675
676 *first = *last = NULL;
677
678 tmp = vmu_alloc_bound();
679 tmp->vmb_start = start;
680 tmp->vmb_type = type;
681
682 /* Hopelessly optimistic case. */
683 if (walker = avl_find(tree, tmp, &where)) {
684 /* We got lucky. */
685 vmu_free_bound(tmp);
686 *first = walker;
687 }
688
689 if (walker == NULL) {
690 /* Is start in the previous node? */
691 walker = avl_nearest(tree, where, AVL_BEFORE);
692 if (walker != NULL) {
693 if (ISWITHIN(walker, start)) {
694 /* We found start. */
695 vmu_free_bound(tmp);
696 *first = walker;
697 }
698 }
699 }
700
701 /*
702 * At this point, if *first is still NULL, then we
703 * didn't get a direct hit and start isn't covered
704 * by the previous node. We know that the next node
705 * must have a greater start value than we require
706 * because avl_find tells us where the AVL routines would
707 * insert our new node. We have some gap between the
708 * start we want and the next node.
709 */
710 if (*first == NULL) {
711 walker = avl_nearest(tree, where, AVL_AFTER);
712 if (walker != NULL && walker->vmb_start <= end) {
713 /* Fill the gap. */
714 tmp->vmb_end = walker->vmb_start - 1;
715 *first = tmp;
716 } else {
717 /* We have a gap over [start, end]. */
718 tmp->vmb_end = end;
719 *first = *last = tmp;
720 }
721 ret += tmp->vmb_end - tmp->vmb_start + 1;
722 avl_insert(tree, tmp, where);
723 }
724
725 ASSERT(*first != NULL);
726
727 if (*last != NULL) {
728 /* We're done. */
729 return (ret);
730 }
731
732 /*
733 * If we are here we still need to set *last and
734 * that may involve filling in some gaps.
735 */
736 *last = *first;
737 for (;;) {
738 if (ISWITHIN(*last, end)) {
739 /* We're done. */
740 break;
741 }
742 walker = AVL_NEXT(tree, *last);
743 if (walker == NULL || walker->vmb_start > end) {
744 /* Bottom or mid tree with gap. */
745 tmp = vmu_alloc_bound();
746 tmp->vmb_start = (*last)->vmb_end + 1;
747 tmp->vmb_end = end;
748 tmp->vmb_type = type;
749 ret += tmp->vmb_end - tmp->vmb_start + 1;
750 avl_insert_here(tree, tmp, *last, AVL_AFTER);
751 *last = tmp;
752 break;
753 } else {
754 if ((*last)->vmb_end + 1 != walker->vmb_start) {
755 /* Non-contiguous. */
756 tmp = vmu_alloc_bound();
757 tmp->vmb_start = (*last)->vmb_end + 1;
758 tmp->vmb_end = walker->vmb_start - 1;
759 tmp->vmb_type = type;
760 ret += tmp->vmb_end - tmp->vmb_start + 1;
761 avl_insert_here(tree, tmp, *last, AVL_AFTER);
762 *last = tmp;
763 } else {
764 *last = walker;
765 }
766 }
767 }
768
769 return (ret);
770 }
771
772 /*
773 * vmu_update_bounds()
774 *
775 * tree: avl_tree in which first and last hang.
776 *
777 * first, last: list of continuous bounds, of which zero or more are of
778 * type VMUSAGE_BOUND_UNKNOWN.
779 *
780 * new_tree: avl_tree in which new_first and new_last hang.
781 *
782 * new_first, new_last: list of continuous bounds, of which none are of
783 * type VMUSAGE_BOUND_UNKNOWN. These bounds are used to
784 * update the types of bounds in (first,last) with
785 * type VMUSAGE_BOUND_UNKNOWN.
786 *
787 * For the list of bounds (first,last), this function updates any bounds
788 * with type VMUSAGE_BOUND_UNKNOWN using the type of the corresponding bound in
789 * the list (new_first, new_last).
790 *
791 * If a bound of type VMUSAGE_BOUND_UNKNOWN spans multiple bounds in the list
792 * (new_first, new_last), it will be split into multiple bounds.
793 *
794 * Return value:
795 * The number of pages in the list of bounds (first,last) that were of
796 * type VMUSAGE_BOUND_UNKNOWN, which have been updated to be of type
797 * VMUSAGE_BOUND_INCORE.
798 *
799 */
800 static pgcnt_t
vmu_update_bounds(avl_tree_t * tree,vmu_bound_t ** first,vmu_bound_t ** last,avl_tree_t * new_tree,vmu_bound_t * new_first,vmu_bound_t * new_last)801 vmu_update_bounds(avl_tree_t *tree, vmu_bound_t **first, vmu_bound_t **last,
802 avl_tree_t *new_tree, vmu_bound_t *new_first, vmu_bound_t *new_last)
803 {
804 vmu_bound_t *next, *new_next, *tmp;
805 pgcnt_t rss = 0;
806
807 next = *first;
808 new_next = new_first;
809
810 /*
811 * Verify first and last bound are covered by new bounds if they
812 * have unknown type.
813 */
814 ASSERT((*first)->vmb_type != VMUSAGE_BOUND_UNKNOWN ||
815 (*first)->vmb_start >= new_first->vmb_start);
816 ASSERT((*last)->vmb_type != VMUSAGE_BOUND_UNKNOWN ||
817 (*last)->vmb_end <= new_last->vmb_end);
818 for (;;) {
819 /* If bound already has type, proceed to next bound. */
820 if (next->vmb_type != VMUSAGE_BOUND_UNKNOWN) {
821 if (next == *last)
822 break;
823 next = AVL_NEXT(tree, next);
824 continue;
825 }
826 while (new_next->vmb_end < next->vmb_start)
827 new_next = AVL_NEXT(new_tree, new_next);
828 ASSERT(new_next->vmb_type != VMUSAGE_BOUND_UNKNOWN);
829 next->vmb_type = new_next->vmb_type;
830 if (new_next->vmb_end < next->vmb_end) {
831 /* need to split bound */
832 tmp = vmu_alloc_bound();
833 tmp->vmb_type = VMUSAGE_BOUND_UNKNOWN;
834 tmp->vmb_start = new_next->vmb_end + 1;
835 tmp->vmb_end = next->vmb_end;
836 avl_insert_here(tree, tmp, next, AVL_AFTER);
837 next->vmb_end = new_next->vmb_end;
838 if (*last == next)
839 *last = tmp;
840 if (next->vmb_type == VMUSAGE_BOUND_INCORE)
841 rss += next->vmb_end - next->vmb_start + 1;
842 next = tmp;
843 } else {
844 if (next->vmb_type == VMUSAGE_BOUND_INCORE)
845 rss += next->vmb_end - next->vmb_start + 1;
846 if (next == *last)
847 break;
848 next = AVL_NEXT(tree, next);
849 }
850 }
851 return (rss);
852 }
853
854 /*
855 * Merges adjacent bounds with same type between first and last bound.
856 * After merge, last pointer may point to a different bound, as (incoming)
857 * last bound may have been merged away.
858 */
859 static void
vmu_merge_bounds(avl_tree_t * tree,vmu_bound_t ** first,vmu_bound_t ** last)860 vmu_merge_bounds(avl_tree_t *tree, vmu_bound_t **first, vmu_bound_t **last)
861 {
862 vmu_bound_t *current;
863 vmu_bound_t *next;
864
865 ASSERT(tree != NULL);
866 ASSERT(*first != NULL);
867 ASSERT(*last != NULL);
868
869 current = *first;
870 while (current != *last) {
871 next = AVL_NEXT(tree, current);
872 if ((current->vmb_end + 1) == next->vmb_start &&
873 current->vmb_type == next->vmb_type) {
874 current->vmb_end = next->vmb_end;
875 avl_remove(tree, next);
876 vmu_free_bound(next);
877 if (next == *last) {
878 *last = current;
879 }
880 } else {
881 current = AVL_NEXT(tree, current);
882 }
883 }
884 }
885
886 /*
887 * Given an amp and a list of bounds, updates each bound's type with
888 * VMUSAGE_BOUND_INCORE or VMUSAGE_BOUND_NOT_INCORE.
889 *
890 * If a bound is partially incore, it will be split into two bounds.
891 * first and last may be modified, as bounds may be split into multiple
892 * bounds if they are partially incore/not-incore.
893 *
894 * Set incore to non-zero if bounds are already known to be incore.
895 *
896 */
897 static void
vmu_amp_update_incore_bounds(avl_tree_t * tree,struct anon_map * amp,vmu_bound_t ** first,vmu_bound_t ** last,boolean_t incore)898 vmu_amp_update_incore_bounds(avl_tree_t *tree, struct anon_map *amp,
899 vmu_bound_t **first, vmu_bound_t **last, boolean_t incore)
900 {
901 vmu_bound_t *next;
902 vmu_bound_t *tmp;
903 pgcnt_t index;
904 short bound_type;
905 short page_type;
906 vnode_t *vn;
907 anoff_t off;
908 struct anon *ap;
909
910 next = *first;
911 /* Shared anon slots don't change once set. */
912 ANON_LOCK_ENTER(&->a_rwlock, RW_READER);
913 for (;;) {
914 if (incore == B_TRUE)
915 next->vmb_type = VMUSAGE_BOUND_INCORE;
916
917 if (next->vmb_type != VMUSAGE_BOUND_UNKNOWN) {
918 if (next == *last)
919 break;
920 next = AVL_NEXT(tree, next);
921 continue;
922 }
923 bound_type = next->vmb_type;
924 index = next->vmb_start;
925 while (index <= next->vmb_end) {
926
927 /*
928 * These are used to determine how much to increment
929 * index when a large page is found.
930 */
931 page_t *page;
932 pgcnt_t pgcnt = 1;
933 uint_t pgshft;
934 pgcnt_t pgmsk;
935
936 ap = anon_get_ptr(amp->ahp, index);
937 if (ap != NULL)
938 swap_xlate(ap, &vn, &off);
939
940 if (ap != NULL && vn != NULL && vn->v_pages != NULL &&
941 (page = page_exists(vn, off)) != NULL) {
942 page_type = VMUSAGE_BOUND_INCORE;
943 if (page->p_szc > 0) {
944 pgcnt = page_get_pagecnt(page->p_szc);
945 pgshft = page_get_shift(page->p_szc);
946 pgmsk = (0x1 << (pgshft - PAGESHIFT))
947 - 1;
948 }
949 } else {
950 page_type = VMUSAGE_BOUND_NOT_INCORE;
951 }
952 if (bound_type == VMUSAGE_BOUND_UNKNOWN) {
953 next->vmb_type = page_type;
954 } else if (next->vmb_type != page_type) {
955 /*
956 * If current bound type does not match page
957 * type, need to split off new bound.
958 */
959 tmp = vmu_alloc_bound();
960 tmp->vmb_type = page_type;
961 tmp->vmb_start = index;
962 tmp->vmb_end = next->vmb_end;
963 avl_insert_here(tree, tmp, next, AVL_AFTER);
964 next->vmb_end = index - 1;
965 if (*last == next)
966 *last = tmp;
967 next = tmp;
968 }
969 if (pgcnt > 1) {
970 /*
971 * If inside large page, jump to next large
972 * page
973 */
974 index = (index & ~pgmsk) + pgcnt;
975 } else {
976 index++;
977 }
978 }
979 if (next == *last) {
980 ASSERT(next->vmb_type != VMUSAGE_BOUND_UNKNOWN);
981 break;
982 } else
983 next = AVL_NEXT(tree, next);
984 }
985 ANON_LOCK_EXIT(&->a_rwlock);
986 }
987
988 /*
989 * Same as vmu_amp_update_incore_bounds(), except for tracking
990 * incore-/not-incore for vnodes.
991 */
992 static void
vmu_vnode_update_incore_bounds(avl_tree_t * tree,vnode_t * vnode,vmu_bound_t ** first,vmu_bound_t ** last)993 vmu_vnode_update_incore_bounds(avl_tree_t *tree, vnode_t *vnode,
994 vmu_bound_t **first, vmu_bound_t **last)
995 {
996 vmu_bound_t *next;
997 vmu_bound_t *tmp;
998 pgcnt_t index;
999 short bound_type;
1000 short page_type;
1001
1002 next = *first;
1003 for (;;) {
1004 if (vnode->v_pages == NULL)
1005 next->vmb_type = VMUSAGE_BOUND_NOT_INCORE;
1006
1007 if (next->vmb_type != VMUSAGE_BOUND_UNKNOWN) {
1008 if (next == *last)
1009 break;
1010 next = AVL_NEXT(tree, next);
1011 continue;
1012 }
1013
1014 bound_type = next->vmb_type;
1015 index = next->vmb_start;
1016 while (index <= next->vmb_end) {
1017
1018 /*
1019 * These are used to determine how much to increment
1020 * index when a large page is found.
1021 */
1022 page_t *page;
1023 pgcnt_t pgcnt = 1;
1024 uint_t pgshft;
1025 pgcnt_t pgmsk;
1026
1027 if (vnode->v_pages != NULL &&
1028 (page = page_exists(vnode, ptob(index))) != NULL) {
1029 page_type = VMUSAGE_BOUND_INCORE;
1030 if (page->p_szc > 0) {
1031 pgcnt = page_get_pagecnt(page->p_szc);
1032 pgshft = page_get_shift(page->p_szc);
1033 pgmsk = (0x1 << (pgshft - PAGESHIFT))
1034 - 1;
1035 }
1036 } else {
1037 page_type = VMUSAGE_BOUND_NOT_INCORE;
1038 }
1039 if (bound_type == VMUSAGE_BOUND_UNKNOWN) {
1040 next->vmb_type = page_type;
1041 } else if (next->vmb_type != page_type) {
1042 /*
1043 * If current bound type does not match page
1044 * type, need to split off new bound.
1045 */
1046 tmp = vmu_alloc_bound();
1047 tmp->vmb_type = page_type;
1048 tmp->vmb_start = index;
1049 tmp->vmb_end = next->vmb_end;
1050 avl_insert_here(tree, tmp, next, AVL_AFTER);
1051 next->vmb_end = index - 1;
1052 if (*last == next)
1053 *last = tmp;
1054 next = tmp;
1055 }
1056 if (pgcnt > 1) {
1057 /*
1058 * If inside large page, jump to next large
1059 * page
1060 */
1061 index = (index & ~pgmsk) + pgcnt;
1062 } else {
1063 index++;
1064 }
1065 }
1066 if (next == *last) {
1067 ASSERT(next->vmb_type != VMUSAGE_BOUND_UNKNOWN);
1068 break;
1069 } else
1070 next = AVL_NEXT(tree, next);
1071 }
1072 }
1073
1074 /*
1075 * Calculate the rss and swap consumed by a segment. vmu_entities is the
1076 * list of entities to visit. For shared segments, the vnode or amp
1077 * is looked up in each entity to see if it has been already counted. Private
1078 * anon pages are checked per entity to ensure that COW pages are not
1079 * double counted.
1080 *
1081 * For private mapped files, first the amp is checked for private pages.
1082 * Bounds not backed by the amp are looked up in the vnode for each entity
1083 * to avoid double counting of private COW vnode pages.
1084 */
1085 static void
vmu_calculate_seg(vmu_entity_t * vmu_entities,struct seg * seg)1086 vmu_calculate_seg(vmu_entity_t *vmu_entities, struct seg *seg)
1087 {
1088 struct segvn_data *svd;
1089 struct shm_data *shmd;
1090 struct spt_data *sptd;
1091 vmu_object_t *shared_object = NULL;
1092 vmu_object_t *entity_object = NULL;
1093 vmu_entity_t *entity;
1094 vmusage_t *result;
1095 vmu_bound_t *first = NULL;
1096 vmu_bound_t *last = NULL;
1097 vmu_bound_t *cur = NULL;
1098 vmu_bound_t *e_first = NULL;
1099 vmu_bound_t *e_last = NULL;
1100 vmu_bound_t *tmp;
1101 pgcnt_t p_index, s_index, p_start, p_end, s_start, s_end, rss, virt;
1102 struct anon_map *private_amp = NULL;
1103 boolean_t incore = B_FALSE;
1104 boolean_t shared = B_FALSE;
1105 int file = 0;
1106 pgcnt_t swresv = 0;
1107 pgcnt_t panon = 0;
1108
1109 s_start = 0;
1110 p_end = 0;
1111 /* Can zero-length segments exist? Not sure, so paranoia. */
1112 if (seg->s_size <= 0)
1113 return;
1114
1115 /*
1116 * Figure out if there is a shared object (such as a named vnode or
1117 * a shared amp, then figure out if there is a private amp, which
1118 * identifies private pages.
1119 */
1120 if (seg->s_ops == &segvn_ops) {
1121 svd = (struct segvn_data *)seg->s_data;
1122 if (svd->type == MAP_SHARED) {
1123 shared = B_TRUE;
1124 } else {
1125 swresv = svd->swresv;
1126
1127 if (SEGVN_LOCK_TRYENTER(seg->s_as, &svd->lock,
1128 RW_READER) != 0) {
1129 /*
1130 * Text replication anon maps can be shared
1131 * across all zones. Space used for text
1132 * replication is typically capped as a small %
1133 * of memory. To keep it simple for now we
1134 * don't account for swap and memory space used
1135 * for text replication.
1136 */
1137 if (svd->tr_state == SEGVN_TR_OFF &&
1138 svd->amp != NULL) {
1139 private_amp = svd->amp;
1140 p_start = svd->anon_index;
1141 p_end = svd->anon_index +
1142 btop(seg->s_size) - 1;
1143 }
1144 SEGVN_LOCK_EXIT(seg->s_as, &svd->lock);
1145 }
1146 }
1147 if (svd->vp != NULL) {
1148 file = 1;
1149 shared_object = vmu_find_insert_object(
1150 vmu_data.vmu_all_vnodes_hash, (caddr_t)svd->vp,
1151 VMUSAGE_TYPE_VNODE);
1152 s_start = btop(svd->offset);
1153 s_end = btop(svd->offset + seg->s_size) - 1;
1154 }
1155 if (svd->amp != NULL && svd->type == MAP_SHARED) {
1156 ASSERT(shared_object == NULL);
1157 shared_object = vmu_find_insert_object(
1158 vmu_data.vmu_all_amps_hash, (caddr_t)svd->amp,
1159 VMUSAGE_TYPE_AMP);
1160 s_start = svd->anon_index;
1161 s_end = svd->anon_index + btop(seg->s_size) - 1;
1162 /* schedctl mappings are always in core */
1163 if (svd->amp->swresv == 0)
1164 incore = B_TRUE;
1165 }
1166 } else if (seg->s_ops == &segspt_shmops) {
1167 shared = B_TRUE;
1168 shmd = (struct shm_data *)seg->s_data;
1169 shared_object = vmu_find_insert_object(
1170 vmu_data.vmu_all_amps_hash, (caddr_t)shmd->shm_amp,
1171 VMUSAGE_TYPE_AMP);
1172 s_start = 0;
1173 s_end = btop(seg->s_size) - 1;
1174 sptd = shmd->shm_sptseg->s_data;
1175
1176 /* ism segments are always incore and do not reserve swap */
1177 if (sptd->spt_flags & SHM_SHARE_MMU)
1178 incore = B_TRUE;
1179
1180 } else {
1181 return;
1182 }
1183
1184 /*
1185 * If there is a private amp, count anon pages that exist. If an
1186 * anon has a refcnt > 1 (COW sharing), then save the anon in a
1187 * hash so that it is not double counted.
1188 *
1189 * If there is also a shared object, then figure out the bounds
1190 * which are not mapped by the private amp.
1191 */
1192 if (private_amp != NULL) {
1193
1194 /* Enter as writer to prevent COW anons from being freed */
1195 ANON_LOCK_ENTER(&private_amp->a_rwlock, RW_WRITER);
1196
1197 p_index = p_start;
1198 s_index = s_start;
1199
1200 while (p_index <= p_end) {
1201
1202 pgcnt_t p_index_next;
1203 pgcnt_t p_bound_size;
1204 int cnt;
1205 anoff_t off;
1206 struct vnode *vn;
1207 struct anon *ap;
1208 page_t *page; /* For handling of large */
1209 pgcnt_t pgcnt = 1; /* pages */
1210 pgcnt_t pgstart;
1211 pgcnt_t pgend;
1212 uint_t pgshft;
1213 pgcnt_t pgmsk;
1214
1215 p_index_next = p_index;
1216 ap = anon_get_next_ptr(private_amp->ahp,
1217 &p_index_next);
1218
1219 /*
1220 * If next anon is past end of mapping, simulate
1221 * end of anon so loop terminates.
1222 */
1223 if (p_index_next > p_end) {
1224 p_index_next = p_end + 1;
1225 ap = NULL;
1226 }
1227 /*
1228 * For COW segments, keep track of bounds not
1229 * backed by private amp so they can be looked
1230 * up in the backing vnode
1231 */
1232 if (p_index_next != p_index) {
1233
1234 /*
1235 * Compute index difference between anon and
1236 * previous anon.
1237 */
1238 p_bound_size = p_index_next - p_index - 1;
1239
1240 if (shared_object != NULL) {
1241 cur = vmu_alloc_bound();
1242 cur->vmb_start = s_index;
1243 cur->vmb_end = s_index + p_bound_size;
1244 cur->vmb_type = VMUSAGE_BOUND_UNKNOWN;
1245 if (first == NULL) {
1246 first = cur;
1247 last = cur;
1248 } else {
1249 last->vmb_next = cur;
1250 last = cur;
1251 }
1252 }
1253 p_index = p_index + p_bound_size + 1;
1254 s_index = s_index + p_bound_size + 1;
1255 }
1256
1257 /* Detect end of anons in amp */
1258 if (ap == NULL)
1259 break;
1260
1261 cnt = ap->an_refcnt;
1262 swap_xlate(ap, &vn, &off);
1263
1264 if (vn == NULL || vn->v_pages == NULL ||
1265 (page = page_exists(vn, off)) == NULL) {
1266 p_index++;
1267 s_index++;
1268 continue;
1269 }
1270
1271 /*
1272 * If large page is found, compute portion of large
1273 * page in mapping, and increment indicies to the next
1274 * large page.
1275 */
1276 if (page->p_szc > 0) {
1277
1278 pgcnt = page_get_pagecnt(page->p_szc);
1279 pgshft = page_get_shift(page->p_szc);
1280 pgmsk = (0x1 << (pgshft - PAGESHIFT)) - 1;
1281
1282 /* First page in large page */
1283 pgstart = p_index & ~pgmsk;
1284 /* Last page in large page */
1285 pgend = pgstart + pgcnt - 1;
1286 /*
1287 * Artifically end page if page extends past
1288 * end of mapping.
1289 */
1290 if (pgend > p_end)
1291 pgend = p_end;
1292
1293 /*
1294 * Compute number of pages from large page
1295 * which are mapped.
1296 */
1297 pgcnt = pgend - p_index + 1;
1298
1299 /*
1300 * Point indicies at page after large page,
1301 * or at page after end of mapping.
1302 */
1303 p_index += pgcnt;
1304 s_index += pgcnt;
1305 } else {
1306 p_index++;
1307 s_index++;
1308 }
1309
1310 /*
1311 * Assume anon structs with a refcnt
1312 * of 1 are not COW shared, so there
1313 * is no reason to track them per entity.
1314 */
1315 if (cnt == 1) {
1316 panon += pgcnt;
1317 continue;
1318 }
1319 for (entity = vmu_entities; entity != NULL;
1320 entity = entity->vme_next_calc) {
1321
1322 result = &entity->vme_result;
1323 /*
1324 * Track COW anons per entity so
1325 * they are not double counted.
1326 */
1327 if (vmu_find_insert_anon(entity->vme_anon_hash,
1328 (caddr_t)ap) == 0)
1329 continue;
1330
1331 result->vmu_rss_all += (pgcnt << PAGESHIFT);
1332 result->vmu_rss_private +=
1333 (pgcnt << PAGESHIFT);
1334 }
1335 }
1336 ANON_LOCK_EXIT(&private_amp->a_rwlock);
1337 }
1338
1339 /* Add up resident anon and swap reserved for private mappings */
1340 if (swresv > 0 || panon > 0) {
1341 for (entity = vmu_entities; entity != NULL;
1342 entity = entity->vme_next_calc) {
1343 result = &entity->vme_result;
1344 result->vmu_swap_all += swresv;
1345 result->vmu_swap_private += swresv;
1346 result->vmu_rss_all += (panon << PAGESHIFT);
1347 result->vmu_rss_private += (panon << PAGESHIFT);
1348 }
1349 }
1350
1351 /* Compute resident pages backing shared amp or named vnode */
1352 if (shared_object != NULL) {
1353 avl_tree_t *tree = &(shared_object->vmo_bounds);
1354
1355 if (first == NULL) {
1356 /*
1357 * No private amp, or private amp has no anon
1358 * structs. This means entire segment is backed by
1359 * the shared object.
1360 */
1361 first = vmu_alloc_bound();
1362 first->vmb_start = s_start;
1363 first->vmb_end = s_end;
1364 first->vmb_type = VMUSAGE_BOUND_UNKNOWN;
1365 }
1366 /*
1367 * Iterate bounds not backed by private amp, and compute
1368 * resident pages.
1369 */
1370 cur = first;
1371 while (cur != NULL) {
1372
1373 if (vmu_insert_lookup_object_bounds(shared_object,
1374 cur->vmb_start, cur->vmb_end, VMUSAGE_BOUND_UNKNOWN,
1375 &first, &last) > 0) {
1376 /* new bounds, find incore/not-incore */
1377 if (shared_object->vmo_type ==
1378 VMUSAGE_TYPE_VNODE) {
1379 vmu_vnode_update_incore_bounds(
1380 tree,
1381 (vnode_t *)
1382 shared_object->vmo_key, &first,
1383 &last);
1384 } else {
1385 vmu_amp_update_incore_bounds(
1386 tree,
1387 (struct anon_map *)
1388 shared_object->vmo_key, &first,
1389 &last, incore);
1390 }
1391 vmu_merge_bounds(tree, &first, &last);
1392 }
1393 for (entity = vmu_entities; entity != NULL;
1394 entity = entity->vme_next_calc) {
1395 avl_tree_t *e_tree;
1396
1397 result = &entity->vme_result;
1398
1399 entity_object = vmu_find_insert_object(
1400 shared_object->vmo_type ==
1401 VMUSAGE_TYPE_VNODE ? entity->vme_vnode_hash:
1402 entity->vme_amp_hash,
1403 shared_object->vmo_key,
1404 shared_object->vmo_type);
1405
1406 virt = vmu_insert_lookup_object_bounds(
1407 entity_object, cur->vmb_start, cur->vmb_end,
1408 VMUSAGE_BOUND_UNKNOWN, &e_first, &e_last);
1409
1410 if (virt == 0)
1411 continue;
1412 /*
1413 * Range visited for this entity
1414 */
1415 e_tree = &(entity_object->vmo_bounds);
1416 rss = vmu_update_bounds(e_tree, &e_first,
1417 &e_last, tree, first, last);
1418 result->vmu_rss_all += (rss << PAGESHIFT);
1419 if (shared == B_TRUE && file == B_FALSE) {
1420 /* shared anon mapping */
1421 result->vmu_swap_all +=
1422 (virt << PAGESHIFT);
1423 result->vmu_swap_shared +=
1424 (virt << PAGESHIFT);
1425 result->vmu_rss_shared +=
1426 (rss << PAGESHIFT);
1427 } else if (shared == B_TRUE && file == B_TRUE) {
1428 /* shared file mapping */
1429 result->vmu_rss_shared +=
1430 (rss << PAGESHIFT);
1431 } else if (shared == B_FALSE &&
1432 file == B_TRUE) {
1433 /* private file mapping */
1434 result->vmu_rss_private +=
1435 (rss << PAGESHIFT);
1436 }
1437 vmu_merge_bounds(e_tree, &e_first, &e_last);
1438 }
1439 tmp = cur;
1440 cur = cur->vmb_next;
1441 vmu_free_bound(tmp);
1442 }
1443 }
1444 }
1445
1446 /*
1447 * Based on the current calculation flags, find the relevant entities
1448 * which are relative to the process. Then calculate each segment
1449 * in the process'es address space for each relevant entity.
1450 */
1451 static void
vmu_calculate_proc(proc_t * p)1452 vmu_calculate_proc(proc_t *p)
1453 {
1454 vmu_entity_t *entities = NULL;
1455 vmu_zone_t *zone;
1456 vmu_entity_t *tmp;
1457 struct as *as;
1458 struct seg *seg;
1459 int ret;
1460
1461 /* Figure out which entities are being computed */
1462 if ((vmu_data.vmu_system) != NULL) {
1463 tmp = vmu_data.vmu_system;
1464 tmp->vme_next_calc = entities;
1465 entities = tmp;
1466 }
1467 if (vmu_data.vmu_calc_flags &
1468 (VMUSAGE_ZONE | VMUSAGE_ALL_ZONES | VMUSAGE_PROJECTS |
1469 VMUSAGE_ALL_PROJECTS | VMUSAGE_TASKS | VMUSAGE_ALL_TASKS |
1470 VMUSAGE_RUSERS | VMUSAGE_ALL_RUSERS | VMUSAGE_EUSERS |
1471 VMUSAGE_ALL_EUSERS)) {
1472 ret = i_mod_hash_find_nosync(vmu_data.vmu_zones_hash,
1473 (mod_hash_key_t)(uintptr_t)p->p_zone->zone_id,
1474 (mod_hash_val_t *)&zone);
1475 if (ret != 0) {
1476 zone = vmu_alloc_zone(p->p_zone->zone_id);
1477 ret = i_mod_hash_insert_nosync(vmu_data.vmu_zones_hash,
1478 (mod_hash_key_t)(uintptr_t)p->p_zone->zone_id,
1479 (mod_hash_val_t)zone, (mod_hash_hndl_t)0);
1480 ASSERT(ret == 0);
1481 }
1482 if (zone->vmz_zone != NULL) {
1483 tmp = zone->vmz_zone;
1484 tmp->vme_next_calc = entities;
1485 entities = tmp;
1486 }
1487 if (vmu_data.vmu_calc_flags &
1488 (VMUSAGE_PROJECTS | VMUSAGE_ALL_PROJECTS)) {
1489 tmp = vmu_find_insert_entity(zone->vmz_projects_hash,
1490 p->p_task->tk_proj->kpj_id, VMUSAGE_PROJECTS,
1491 zone->vmz_id);
1492 tmp->vme_next_calc = entities;
1493 entities = tmp;
1494 }
1495 if (vmu_data.vmu_calc_flags &
1496 (VMUSAGE_TASKS | VMUSAGE_ALL_TASKS)) {
1497 tmp = vmu_find_insert_entity(zone->vmz_tasks_hash,
1498 p->p_task->tk_tkid, VMUSAGE_TASKS, zone->vmz_id);
1499 tmp->vme_next_calc = entities;
1500 entities = tmp;
1501 }
1502 if (vmu_data.vmu_calc_flags &
1503 (VMUSAGE_RUSERS | VMUSAGE_ALL_RUSERS)) {
1504 tmp = vmu_find_insert_entity(zone->vmz_rusers_hash,
1505 crgetruid(p->p_cred), VMUSAGE_RUSERS, zone->vmz_id);
1506 tmp->vme_next_calc = entities;
1507 entities = tmp;
1508 }
1509 if (vmu_data.vmu_calc_flags &
1510 (VMUSAGE_EUSERS | VMUSAGE_ALL_EUSERS)) {
1511 tmp = vmu_find_insert_entity(zone->vmz_eusers_hash,
1512 crgetuid(p->p_cred), VMUSAGE_EUSERS, zone->vmz_id);
1513 tmp->vme_next_calc = entities;
1514 entities = tmp;
1515 }
1516 }
1517 /* Entities which collapse projects and users for all zones */
1518 if (vmu_data.vmu_calc_flags & VMUSAGE_COL_PROJECTS) {
1519 tmp = vmu_find_insert_entity(vmu_data.vmu_projects_col_hash,
1520 p->p_task->tk_proj->kpj_id, VMUSAGE_PROJECTS, ALL_ZONES);
1521 tmp->vme_next_calc = entities;
1522 entities = tmp;
1523 }
1524 if (vmu_data.vmu_calc_flags & VMUSAGE_COL_RUSERS) {
1525 tmp = vmu_find_insert_entity(vmu_data.vmu_rusers_col_hash,
1526 crgetruid(p->p_cred), VMUSAGE_RUSERS, ALL_ZONES);
1527 tmp->vme_next_calc = entities;
1528 entities = tmp;
1529 }
1530 if (vmu_data.vmu_calc_flags & VMUSAGE_COL_EUSERS) {
1531 tmp = vmu_find_insert_entity(vmu_data.vmu_eusers_col_hash,
1532 crgetuid(p->p_cred), VMUSAGE_EUSERS, ALL_ZONES);
1533 tmp->vme_next_calc = entities;
1534 entities = tmp;
1535 }
1536
1537 ASSERT(entities != NULL);
1538 /* process all segs in process's address space */
1539 as = p->p_as;
1540 AS_LOCK_ENTER(as, RW_READER);
1541 for (seg = AS_SEGFIRST(as); seg != NULL;
1542 seg = AS_SEGNEXT(as, seg)) {
1543 vmu_calculate_seg(entities, seg);
1544 }
1545 AS_LOCK_EXIT(as);
1546 }
1547
1548 /*
1549 * Free data created by previous call to vmu_calculate().
1550 */
1551 static void
vmu_clear_calc()1552 vmu_clear_calc()
1553 {
1554 if (vmu_data.vmu_system != NULL) {
1555 vmu_free_entity(vmu_data.vmu_system);
1556 vmu_data.vmu_system = NULL;
1557 }
1558 if (vmu_data.vmu_zones_hash != NULL)
1559 i_mod_hash_clear_nosync(vmu_data.vmu_zones_hash);
1560 if (vmu_data.vmu_projects_col_hash != NULL)
1561 i_mod_hash_clear_nosync(vmu_data.vmu_projects_col_hash);
1562 if (vmu_data.vmu_rusers_col_hash != NULL)
1563 i_mod_hash_clear_nosync(vmu_data.vmu_rusers_col_hash);
1564 if (vmu_data.vmu_eusers_col_hash != NULL)
1565 i_mod_hash_clear_nosync(vmu_data.vmu_eusers_col_hash);
1566
1567 i_mod_hash_clear_nosync(vmu_data.vmu_all_vnodes_hash);
1568 i_mod_hash_clear_nosync(vmu_data.vmu_all_amps_hash);
1569 }
1570
1571 /*
1572 * Free unused data structures. These can result if the system workload
1573 * decreases between calculations.
1574 */
1575 static void
vmu_free_extra()1576 vmu_free_extra()
1577 {
1578 vmu_bound_t *tb;
1579 vmu_object_t *to;
1580 vmu_entity_t *te;
1581 vmu_zone_t *tz;
1582
1583 while (vmu_data.vmu_free_bounds != NULL) {
1584 tb = vmu_data.vmu_free_bounds;
1585 vmu_data.vmu_free_bounds = vmu_data.vmu_free_bounds->vmb_next;
1586 kmem_cache_free(vmu_bound_cache, tb);
1587 }
1588 while (vmu_data.vmu_free_objects != NULL) {
1589 to = vmu_data.vmu_free_objects;
1590 vmu_data.vmu_free_objects =
1591 vmu_data.vmu_free_objects->vmo_next;
1592 kmem_cache_free(vmu_object_cache, to);
1593 }
1594 while (vmu_data.vmu_free_entities != NULL) {
1595 te = vmu_data.vmu_free_entities;
1596 vmu_data.vmu_free_entities =
1597 vmu_data.vmu_free_entities->vme_next;
1598 if (te->vme_vnode_hash != NULL)
1599 mod_hash_destroy_hash(te->vme_vnode_hash);
1600 if (te->vme_amp_hash != NULL)
1601 mod_hash_destroy_hash(te->vme_amp_hash);
1602 if (te->vme_anon_hash != NULL)
1603 mod_hash_destroy_hash(te->vme_anon_hash);
1604 kmem_free(te, sizeof (vmu_entity_t));
1605 }
1606 while (vmu_data.vmu_free_zones != NULL) {
1607 tz = vmu_data.vmu_free_zones;
1608 vmu_data.vmu_free_zones =
1609 vmu_data.vmu_free_zones->vmz_next;
1610 if (tz->vmz_projects_hash != NULL)
1611 mod_hash_destroy_hash(tz->vmz_projects_hash);
1612 if (tz->vmz_tasks_hash != NULL)
1613 mod_hash_destroy_hash(tz->vmz_tasks_hash);
1614 if (tz->vmz_rusers_hash != NULL)
1615 mod_hash_destroy_hash(tz->vmz_rusers_hash);
1616 if (tz->vmz_eusers_hash != NULL)
1617 mod_hash_destroy_hash(tz->vmz_eusers_hash);
1618 kmem_free(tz, sizeof (vmu_zone_t));
1619 }
1620 }
1621
1622 extern kcondvar_t *pr_pid_cv;
1623
1624 /*
1625 * Determine which entity types are relevant and allocate the hashes to
1626 * track them. Then walk the process table and count rss and swap
1627 * for each process'es address space. Address space object such as
1628 * vnodes, amps and anons are tracked per entity, so that they are
1629 * not double counted in the results.
1630 *
1631 */
1632 static void
vmu_calculate()1633 vmu_calculate()
1634 {
1635 int i = 0;
1636 int ret;
1637 proc_t *p;
1638
1639 vmu_clear_calc();
1640
1641 if (vmu_data.vmu_calc_flags & VMUSAGE_SYSTEM)
1642 vmu_data.vmu_system = vmu_alloc_entity(0, VMUSAGE_SYSTEM,
1643 ALL_ZONES);
1644
1645 /*
1646 * Walk process table and calculate rss of each proc.
1647 *
1648 * Pidlock and p_lock cannot be held while doing the rss calculation.
1649 * This is because:
1650 * 1. The calculation allocates using KM_SLEEP.
1651 * 2. The calculation grabs a_lock, which cannot be grabbed
1652 * after p_lock.
1653 *
1654 * Since pidlock must be dropped, we cannot simply just walk the
1655 * practive list. Instead, we walk the process table, and sprlock
1656 * each process to ensure that it does not exit during the
1657 * calculation.
1658 */
1659
1660 mutex_enter(&pidlock);
1661 for (i = 0; i < v.v_proc; i++) {
1662 again:
1663 p = pid_entry(i);
1664 if (p == NULL)
1665 continue;
1666
1667 mutex_enter(&p->p_lock);
1668 mutex_exit(&pidlock);
1669
1670 if (panicstr) {
1671 mutex_exit(&p->p_lock);
1672 return;
1673 }
1674
1675 /* Try to set P_PR_LOCK */
1676 ret = sprtrylock_proc(p);
1677 if (ret == -1) {
1678 /* Process in invalid state */
1679 mutex_exit(&p->p_lock);
1680 mutex_enter(&pidlock);
1681 continue;
1682 } else if (ret == 1) {
1683 /*
1684 * P_PR_LOCK is already set. Wait and try again.
1685 * This also drops p_lock.
1686 */
1687 sprwaitlock_proc(p);
1688 mutex_enter(&pidlock);
1689 goto again;
1690 }
1691 mutex_exit(&p->p_lock);
1692
1693 vmu_calculate_proc(p);
1694
1695 mutex_enter(&p->p_lock);
1696 sprunlock(p);
1697 mutex_enter(&pidlock);
1698 }
1699 mutex_exit(&pidlock);
1700
1701 vmu_free_extra();
1702 }
1703
1704 /*
1705 * allocate a new cache for N results satisfying flags
1706 */
1707 vmu_cache_t *
vmu_cache_alloc(size_t nres,uint_t flags)1708 vmu_cache_alloc(size_t nres, uint_t flags)
1709 {
1710 vmu_cache_t *cache;
1711
1712 cache = kmem_zalloc(sizeof (vmu_cache_t), KM_SLEEP);
1713 cache->vmc_results = kmem_zalloc(sizeof (vmusage_t) * nres, KM_SLEEP);
1714 cache->vmc_nresults = nres;
1715 cache->vmc_flags = flags;
1716 cache->vmc_refcnt = 1;
1717 return (cache);
1718 }
1719
1720 /*
1721 * Make sure cached results are not freed
1722 */
1723 static void
vmu_cache_hold(vmu_cache_t * cache)1724 vmu_cache_hold(vmu_cache_t *cache)
1725 {
1726 ASSERT(MUTEX_HELD(&vmu_data.vmu_lock));
1727 cache->vmc_refcnt++;
1728 }
1729
1730 /*
1731 * free cache data
1732 */
1733 static void
vmu_cache_rele(vmu_cache_t * cache)1734 vmu_cache_rele(vmu_cache_t *cache)
1735 {
1736 ASSERT(MUTEX_HELD(&vmu_data.vmu_lock));
1737 ASSERT(cache->vmc_refcnt > 0);
1738 cache->vmc_refcnt--;
1739 if (cache->vmc_refcnt == 0) {
1740 kmem_free(cache->vmc_results, sizeof (vmusage_t) *
1741 cache->vmc_nresults);
1742 kmem_free(cache, sizeof (vmu_cache_t));
1743 }
1744 }
1745
1746 /*
1747 * Copy out the cached results to a caller. Inspect the callers flags
1748 * and zone to determine which cached results should be copied.
1749 */
1750 static int
vmu_copyout_results(vmu_cache_t * cache,vmusage_t * buf,size_t * nres,uint_t flags,uio_seg_t seg)1751 vmu_copyout_results(vmu_cache_t *cache, vmusage_t *buf, size_t *nres,
1752 uint_t flags, uio_seg_t seg)
1753 {
1754 vmusage_t *result, *out_result;
1755 vmusage_t dummy;
1756 size_t i, count = 0;
1757 size_t bufsize;
1758 int ret = 0;
1759 uint_t types = 0;
1760
1761 if (nres != NULL) {
1762 if (uio_copyin(nres, &bufsize, sizeof (size_t), seg) != 0)
1763 return (set_errno(EFAULT));
1764 } else {
1765 bufsize = 0;
1766 }
1767
1768 /* figure out what results the caller is interested in. */
1769 if ((flags & VMUSAGE_SYSTEM) && curproc->p_zone == global_zone)
1770 types |= VMUSAGE_SYSTEM;
1771 if (flags & (VMUSAGE_ZONE | VMUSAGE_ALL_ZONES))
1772 types |= VMUSAGE_ZONE;
1773 if (flags & (VMUSAGE_PROJECTS | VMUSAGE_ALL_PROJECTS |
1774 VMUSAGE_COL_PROJECTS))
1775 types |= VMUSAGE_PROJECTS;
1776 if (flags & (VMUSAGE_TASKS | VMUSAGE_ALL_TASKS))
1777 types |= VMUSAGE_TASKS;
1778 if (flags & (VMUSAGE_RUSERS | VMUSAGE_ALL_RUSERS | VMUSAGE_COL_RUSERS))
1779 types |= VMUSAGE_RUSERS;
1780 if (flags & (VMUSAGE_EUSERS | VMUSAGE_ALL_EUSERS | VMUSAGE_COL_EUSERS))
1781 types |= VMUSAGE_EUSERS;
1782
1783 /* count results for current zone */
1784 out_result = buf;
1785 for (result = cache->vmc_results, i = 0;
1786 i < cache->vmc_nresults; result++, i++) {
1787
1788 /* Do not return "other-zone" results to non-global zones */
1789 if (curproc->p_zone != global_zone &&
1790 curproc->p_zone->zone_id != result->vmu_zoneid)
1791 continue;
1792
1793 /*
1794 * If non-global zone requests VMUSAGE_SYSTEM, fake
1795 * up VMUSAGE_ZONE result as VMUSAGE_SYSTEM result.
1796 */
1797 if (curproc->p_zone != global_zone &&
1798 (flags & VMUSAGE_SYSTEM) != 0 &&
1799 result->vmu_type == VMUSAGE_ZONE) {
1800 count++;
1801 if (out_result != NULL) {
1802 if (bufsize < count) {
1803 ret = set_errno(EOVERFLOW);
1804 } else {
1805 dummy = *result;
1806 dummy.vmu_zoneid = ALL_ZONES;
1807 dummy.vmu_id = 0;
1808 dummy.vmu_type = VMUSAGE_SYSTEM;
1809 if (uio_copyout(&dummy, out_result,
1810 sizeof (vmusage_t), seg) != 0)
1811 return (set_errno(EFAULT));
1812 out_result++;
1813 }
1814 }
1815 }
1816
1817 /* Skip results that do not match requested type */
1818 if ((result->vmu_type & types) == 0)
1819 continue;
1820
1821 /* Skip collated results if not requested */
1822 if (result->vmu_zoneid == ALL_ZONES) {
1823 if (result->vmu_type == VMUSAGE_PROJECTS &&
1824 (flags & VMUSAGE_COL_PROJECTS) == 0)
1825 continue;
1826 if (result->vmu_type == VMUSAGE_EUSERS &&
1827 (flags & VMUSAGE_COL_EUSERS) == 0)
1828 continue;
1829 if (result->vmu_type == VMUSAGE_RUSERS &&
1830 (flags & VMUSAGE_COL_RUSERS) == 0)
1831 continue;
1832 }
1833
1834 /* Skip "other zone" results if not requested */
1835 if (result->vmu_zoneid != curproc->p_zone->zone_id) {
1836 if (result->vmu_type == VMUSAGE_ZONE &&
1837 (flags & VMUSAGE_ALL_ZONES) == 0)
1838 continue;
1839 if (result->vmu_type == VMUSAGE_PROJECTS &&
1840 (flags & (VMUSAGE_ALL_PROJECTS |
1841 VMUSAGE_COL_PROJECTS)) == 0)
1842 continue;
1843 if (result->vmu_type == VMUSAGE_TASKS &&
1844 (flags & VMUSAGE_ALL_TASKS) == 0)
1845 continue;
1846 if (result->vmu_type == VMUSAGE_RUSERS &&
1847 (flags & (VMUSAGE_ALL_RUSERS |
1848 VMUSAGE_COL_RUSERS)) == 0)
1849 continue;
1850 if (result->vmu_type == VMUSAGE_EUSERS &&
1851 (flags & (VMUSAGE_ALL_EUSERS |
1852 VMUSAGE_COL_EUSERS)) == 0)
1853 continue;
1854 }
1855 count++;
1856 if (out_result != NULL) {
1857 if (bufsize < count) {
1858 ret = set_errno(EOVERFLOW);
1859 } else {
1860 if (uio_copyout(result, out_result,
1861 sizeof (vmusage_t), seg) != 0)
1862 return (set_errno(EFAULT));
1863 out_result++;
1864 }
1865 }
1866 }
1867 if (nres != NULL)
1868 if (uio_copyout(&count, nres, sizeof (size_t), seg) != 0)
1869 return (set_errno(EFAULT));
1870
1871 return (ret);
1872 }
1873
1874 /*
1875 * vm_getusage()
1876 *
1877 * Counts rss and swap by zone, project, task, and/or user. The flags argument
1878 * determines the type of results structures returned. Flags requesting
1879 * results from more than one zone are "flattened" to the local zone if the
1880 * caller is not the global zone.
1881 *
1882 * args:
1883 * flags: bitmap consisting of one or more of VMUSAGE_*.
1884 * age: maximum allowable age (time since counting was done) in
1885 * seconds of the results. Results from previous callers are
1886 * cached in kernel.
1887 * buf: pointer to buffer array of vmusage_t. If NULL, then only nres
1888 * set on success.
1889 * nres: Set to number of vmusage_t structures pointed to by buf
1890 * before calling vm_getusage().
1891 * On return 0 (success) or ENOSPC, is set to the number of result
1892 * structures returned or attempted to return.
1893 * seg: address space in which buf and nres reside, either
1894 * UIO_USERSPACE or UIO_SYSSPACE.
1895 *
1896 * returns 0 on success, -1 on failure:
1897 * EINTR (interrupted)
1898 * ENOSPC (nres to small for results, nres set to needed value for success)
1899 * EINVAL (flags invalid)
1900 * EFAULT (bad address for buf or nres)
1901 */
1902 int
vm_getusage(uint_t flags,time_t age,vmusage_t * buf,size_t * nres,uio_seg_t seg)1903 vm_getusage(uint_t flags, time_t age, vmusage_t *buf, size_t *nres,
1904 uio_seg_t seg)
1905 {
1906 vmu_entity_t *entity;
1907 vmusage_t *result;
1908 int ret = 0;
1909 int cacherecent = 0;
1910 hrtime_t now;
1911 uint_t flags_orig;
1912
1913 /*
1914 * Non-global zones cannot request system wide and/or collated
1915 * results, or the system result, so munge the flags accordingly.
1916 */
1917 flags_orig = flags;
1918 if (curproc->p_zone != global_zone) {
1919 if (flags & (VMUSAGE_ALL_PROJECTS | VMUSAGE_COL_PROJECTS)) {
1920 flags &= ~(VMUSAGE_ALL_PROJECTS | VMUSAGE_COL_PROJECTS);
1921 flags |= VMUSAGE_PROJECTS;
1922 }
1923 if (flags & (VMUSAGE_ALL_RUSERS | VMUSAGE_COL_RUSERS)) {
1924 flags &= ~(VMUSAGE_ALL_RUSERS | VMUSAGE_COL_RUSERS);
1925 flags |= VMUSAGE_RUSERS;
1926 }
1927 if (flags & (VMUSAGE_ALL_EUSERS | VMUSAGE_COL_EUSERS)) {
1928 flags &= ~(VMUSAGE_ALL_EUSERS | VMUSAGE_COL_EUSERS);
1929 flags |= VMUSAGE_EUSERS;
1930 }
1931 if (flags & VMUSAGE_SYSTEM) {
1932 flags &= ~VMUSAGE_SYSTEM;
1933 flags |= VMUSAGE_ZONE;
1934 }
1935 }
1936
1937 /* Check for unknown flags */
1938 if ((flags & (~VMUSAGE_MASK)) != 0)
1939 return (set_errno(EINVAL));
1940
1941 /* Check for no flags */
1942 if ((flags & VMUSAGE_MASK) == 0)
1943 return (set_errno(EINVAL));
1944
1945 mutex_enter(&vmu_data.vmu_lock);
1946 now = gethrtime();
1947
1948 start:
1949 if (vmu_data.vmu_cache != NULL) {
1950
1951 vmu_cache_t *cache;
1952
1953 if ((vmu_data.vmu_cache->vmc_timestamp +
1954 ((hrtime_t)age * NANOSEC)) > now)
1955 cacherecent = 1;
1956
1957 if ((vmu_data.vmu_cache->vmc_flags & flags) == flags &&
1958 cacherecent == 1) {
1959 cache = vmu_data.vmu_cache;
1960 vmu_cache_hold(cache);
1961 mutex_exit(&vmu_data.vmu_lock);
1962
1963 ret = vmu_copyout_results(cache, buf, nres, flags_orig,
1964 seg);
1965 mutex_enter(&vmu_data.vmu_lock);
1966 vmu_cache_rele(cache);
1967 if (vmu_data.vmu_pending_waiters > 0)
1968 cv_broadcast(&vmu_data.vmu_cv);
1969 mutex_exit(&vmu_data.vmu_lock);
1970 return (ret);
1971 }
1972 /*
1973 * If the cache is recent, it is likely that there are other
1974 * consumers of vm_getusage running, so add their flags to the
1975 * desired flags for the calculation.
1976 */
1977 if (cacherecent == 1)
1978 flags = vmu_data.vmu_cache->vmc_flags | flags;
1979 }
1980 if (vmu_data.vmu_calc_thread == NULL) {
1981
1982 vmu_cache_t *cache;
1983
1984 vmu_data.vmu_calc_thread = curthread;
1985 vmu_data.vmu_calc_flags = flags;
1986 vmu_data.vmu_entities = NULL;
1987 vmu_data.vmu_nentities = 0;
1988 if (vmu_data.vmu_pending_waiters > 0)
1989 vmu_data.vmu_calc_flags |=
1990 vmu_data.vmu_pending_flags;
1991
1992 vmu_data.vmu_pending_flags = 0;
1993 mutex_exit(&vmu_data.vmu_lock);
1994 vmu_calculate();
1995 mutex_enter(&vmu_data.vmu_lock);
1996 /* copy results to cache */
1997 if (vmu_data.vmu_cache != NULL)
1998 vmu_cache_rele(vmu_data.vmu_cache);
1999 cache = vmu_data.vmu_cache =
2000 vmu_cache_alloc(vmu_data.vmu_nentities,
2001 vmu_data.vmu_calc_flags);
2002
2003 result = cache->vmc_results;
2004 for (entity = vmu_data.vmu_entities; entity != NULL;
2005 entity = entity->vme_next) {
2006 *result = entity->vme_result;
2007 result++;
2008 }
2009 cache->vmc_timestamp = gethrtime();
2010 vmu_cache_hold(cache);
2011
2012 vmu_data.vmu_calc_flags = 0;
2013 vmu_data.vmu_calc_thread = NULL;
2014
2015 if (vmu_data.vmu_pending_waiters > 0)
2016 cv_broadcast(&vmu_data.vmu_cv);
2017
2018 mutex_exit(&vmu_data.vmu_lock);
2019
2020 /* copy cache */
2021 ret = vmu_copyout_results(cache, buf, nres, flags_orig, seg);
2022 mutex_enter(&vmu_data.vmu_lock);
2023 vmu_cache_rele(cache);
2024 mutex_exit(&vmu_data.vmu_lock);
2025
2026 return (ret);
2027 }
2028 vmu_data.vmu_pending_flags |= flags;
2029 vmu_data.vmu_pending_waiters++;
2030 while (vmu_data.vmu_calc_thread != NULL) {
2031 if (cv_wait_sig(&vmu_data.vmu_cv,
2032 &vmu_data.vmu_lock) == 0) {
2033 vmu_data.vmu_pending_waiters--;
2034 mutex_exit(&vmu_data.vmu_lock);
2035 return (set_errno(EINTR));
2036 }
2037 }
2038 vmu_data.vmu_pending_waiters--;
2039 goto start;
2040 }
2041