1 /*- 2 * SPDX-License-Identifier: BSD-2-Clause 3 * 4 * Copyright (c) 2011 NetApp, Inc. 5 * All rights reserved. 6 * 7 * Redistribution and use in source and binary forms, with or without 8 * modification, are permitted provided that the following conditions 9 * are met: 10 * 1. Redistributions of source code must retain the above copyright 11 * notice, this list of conditions and the following disclaimer. 12 * 2. Redistributions in binary form must reproduce the above copyright 13 * notice, this list of conditions and the following disclaimer in the 14 * documentation and/or other materials provided with the distribution. 15 * 16 * THIS SOFTWARE IS PROVIDED BY NETAPP, INC ``AS IS'' AND 17 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE 18 * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE 19 * ARE DISCLAIMED. IN NO EVENT SHALL NETAPP, INC OR CONTRIBUTORS BE LIABLE 20 * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL 21 * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS 22 * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) 23 * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT 24 * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY 25 * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF 26 * SUCH DAMAGE. 27 */ 28 29 /*- 30 * Copyright (c) 2011 Google, Inc. 31 * All rights reserved. 32 * 33 * Redistribution and use in source and binary forms, with or without 34 * modification, are permitted provided that the following conditions 35 * are met: 36 * 1. Redistributions of source code must retain the above copyright 37 * notice, this list of conditions and the following disclaimer. 38 * 2. Redistributions in binary form must reproduce the above copyright 39 * notice, this list of conditions and the following disclaimer in the 40 * documentation and/or other materials provided with the distribution. 41 * 42 * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND 43 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE 44 * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE 45 * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE 46 * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL 47 * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS 48 * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) 49 * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT 50 * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY 51 * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF 52 * SUCH DAMAGE. 53 */ 54 55 #include <sys/cdefs.h> 56 #include <sys/ioctl.h> 57 #include <sys/stat.h> 58 #include <sys/disk.h> 59 #include <sys/queue.h> 60 61 #include <machine/specialreg.h> 62 #include <machine/vmm.h> 63 64 #include <assert.h> 65 #include <dirent.h> 66 #include <dlfcn.h> 67 #include <errno.h> 68 #include <err.h> 69 #include <fcntl.h> 70 #include <getopt.h> 71 #include <libgen.h> 72 #include <limits.h> 73 #include <setjmp.h> 74 #include <stdio.h> 75 #include <stdlib.h> 76 #include <string.h> 77 #include <sysexits.h> 78 #include <termios.h> 79 #include <unistd.h> 80 81 #include <capsicum_helpers.h> 82 #include <vmmapi.h> 83 84 #include "userboot.h" 85 86 #define MB (1024 * 1024UL) 87 #define GB (1024 * 1024 * 1024UL) 88 #define BSP 0 89 90 #define NDISKS 32 91 92 /* 93 * Reason for our loader reload and reentry, though these aren't really used 94 * at the moment. 95 */ 96 enum { 97 /* 0 cannot be allocated; setjmp(3) return. */ 98 JMP_SWAPLOADER = 0x01, 99 JMP_REBOOT, 100 }; 101 102 static struct termios term, oldterm; 103 static int disk_fd[NDISKS]; 104 static int ndisks; 105 static int bootfd = -1; 106 static int consin_fd, consout_fd; 107 static int hostbase_fd = -1; 108 109 static void *loader_hdl; 110 static char *loader; 111 static int explicit_loader_fd = -1; 112 static jmp_buf jb; 113 114 static char *vmname, *progname; 115 static struct vmctx *ctx; 116 static struct vcpu *vcpu; 117 118 static uint64_t gdtbase, cr3, rsp; 119 120 static void cb_exit(void *arg, int v); 121 122 /* 123 * Console i/o callbacks 124 */ 125 126 static void 127 cb_putc(void *arg __unused, int ch) 128 { 129 char c = ch; 130 131 (void) write(consout_fd, &c, 1); 132 } 133 134 static int 135 cb_getc(void *arg __unused) 136 { 137 char c; 138 ssize_t nread; 139 140 nread = read(consin_fd, &c, 1); 141 if (nread == 0) 142 cb_exit(NULL, USERBOOT_EXIT_QUIT); 143 if (nread == 1) 144 return (c); 145 return (-1); 146 } 147 148 static int 149 cb_poll(void *arg __unused) 150 { 151 int n; 152 153 if (ioctl(consin_fd, FIONREAD, &n) >= 0) 154 return (n > 0); 155 return (0); 156 } 157 158 /* 159 * Host filesystem i/o callbacks 160 */ 161 162 struct cb_file { 163 int cf_isdir; 164 size_t cf_size; 165 struct stat cf_stat; 166 union { 167 int fd; 168 DIR *dir; 169 } cf_u; 170 }; 171 172 static int 173 cb_open(void *arg __unused, const char *filename, void **hp) 174 { 175 struct cb_file *cf; 176 struct stat sb; 177 int fd; 178 179 cf = NULL; 180 if (hostbase_fd == -1) 181 return (ENOENT); 182 183 /* Absolute paths are relative to our hostbase, chop off leading /. */ 184 if (filename[0] == '/') 185 filename++; 186 187 /* Lookup of /, use . instead. */ 188 if (filename[0] == '\0') 189 filename = "."; 190 191 /* May be opening the root dir */ 192 fd = openat(hostbase_fd, filename, O_RDONLY | O_RESOLVE_BENEATH); 193 if (fd < 0) 194 return (errno); 195 196 if (fstat(fd, &sb) < 0) { 197 int serrno = errno; 198 199 close(fd); 200 return (serrno); 201 } else if (!S_ISDIR(sb.st_mode) && !S_ISREG(sb.st_mode)) { 202 close(fd); 203 return (EINVAL); 204 } 205 206 cf = malloc(sizeof(struct cb_file)); 207 if (cf == NULL) { 208 close(fd); 209 return (ENOMEM); 210 } 211 212 cf->cf_stat = sb; 213 cf->cf_size = cf->cf_stat.st_size; 214 215 if (S_ISDIR(cf->cf_stat.st_mode)) { 216 cf->cf_isdir = 1; 217 cf->cf_u.dir = fdopendir(fd); 218 if (cf->cf_u.dir == NULL) { 219 close(fd); 220 free(cf); 221 return (ENOMEM); 222 } 223 } else { 224 cf->cf_isdir = 0; 225 cf->cf_u.fd = fd; 226 } 227 *hp = cf; 228 return (0); 229 } 230 231 static int 232 cb_close(void *arg __unused, void *h) 233 { 234 struct cb_file *cf = h; 235 236 if (cf->cf_isdir) 237 closedir(cf->cf_u.dir); 238 else 239 close(cf->cf_u.fd); 240 free(cf); 241 242 return (0); 243 } 244 245 static int 246 cb_isdir(void *arg __unused, void *h) 247 { 248 struct cb_file *cf = h; 249 250 return (cf->cf_isdir); 251 } 252 253 static int 254 cb_read(void *arg __unused, void *h, void *buf, size_t size, size_t *resid) 255 { 256 struct cb_file *cf = h; 257 ssize_t sz; 258 259 if (cf->cf_isdir) 260 return (EINVAL); 261 sz = read(cf->cf_u.fd, buf, size); 262 if (sz < 0) 263 return (EINVAL); 264 *resid = size - sz; 265 return (0); 266 } 267 268 static int 269 cb_readdir(void *arg __unused, void *h, uint32_t *fileno_return, 270 uint8_t *type_return, size_t *namelen_return, char *name) 271 { 272 struct cb_file *cf = h; 273 struct dirent *dp; 274 275 if (!cf->cf_isdir) 276 return (EINVAL); 277 278 dp = readdir(cf->cf_u.dir); 279 if (!dp) 280 return (ENOENT); 281 282 /* 283 * Note: d_namlen is in the range 0..255 and therefore less 284 * than PATH_MAX so we don't need to test before copying. 285 */ 286 *fileno_return = dp->d_fileno; 287 *type_return = dp->d_type; 288 *namelen_return = dp->d_namlen; 289 memcpy(name, dp->d_name, dp->d_namlen); 290 name[dp->d_namlen] = 0; 291 292 return (0); 293 } 294 295 static int 296 cb_seek(void *arg __unused, void *h, uint64_t offset, int whence) 297 { 298 struct cb_file *cf = h; 299 300 if (cf->cf_isdir) 301 return (EINVAL); 302 if (lseek(cf->cf_u.fd, offset, whence) < 0) 303 return (errno); 304 return (0); 305 } 306 307 static int 308 cb_stat(void *arg __unused, void *h, struct stat *sbp) 309 { 310 struct cb_file *cf = h; 311 312 memset(sbp, 0, sizeof(struct stat)); 313 sbp->st_mode = cf->cf_stat.st_mode; 314 sbp->st_uid = cf->cf_stat.st_uid; 315 sbp->st_gid = cf->cf_stat.st_gid; 316 sbp->st_size = cf->cf_stat.st_size; 317 sbp->st_mtime = cf->cf_stat.st_mtime; 318 sbp->st_dev = cf->cf_stat.st_dev; 319 sbp->st_ino = cf->cf_stat.st_ino; 320 321 return (0); 322 } 323 324 /* 325 * Disk image i/o callbacks 326 */ 327 328 static int 329 cb_diskread(void *arg __unused, int unit, uint64_t from, void *to, size_t size, 330 size_t *resid) 331 { 332 ssize_t n; 333 334 if (unit < 0 || unit >= ndisks) 335 return (EIO); 336 n = pread(disk_fd[unit], to, size, from); 337 if (n < 0) 338 return (errno); 339 *resid = size - n; 340 return (0); 341 } 342 343 static int 344 cb_diskwrite(void *arg __unused, int unit, uint64_t offset, void *src, 345 size_t size, size_t *resid) 346 { 347 ssize_t n; 348 349 if (unit < 0 || unit >= ndisks) 350 return (EIO); 351 n = pwrite(disk_fd[unit], src, size, offset); 352 if (n < 0) 353 return (errno); 354 *resid = size - n; 355 return (0); 356 } 357 358 static int 359 cb_diskioctl(void *arg __unused, int unit, u_long cmd, void *data) 360 { 361 struct stat sb; 362 363 if (unit < 0 || unit >= ndisks) 364 return (EBADF); 365 366 switch (cmd) { 367 case DIOCGSECTORSIZE: 368 if (fstat(disk_fd[unit], &sb) != 0) 369 return (ENOTTY); 370 if (S_ISCHR(sb.st_mode)) { 371 if (ioctl(disk_fd[unit], DIOCGSECTORSIZE, data) != 0) 372 return (ENOTTY); 373 } else { 374 *(u_int *)data = 512; 375 } 376 break; 377 case DIOCGMEDIASIZE: 378 if (fstat(disk_fd[unit], &sb) != 0) 379 return (ENOTTY); 380 if (S_ISCHR(sb.st_mode) && 381 ioctl(disk_fd[unit], DIOCGMEDIASIZE, &sb.st_size) != 0) 382 return (ENOTTY); 383 *(off_t *)data = sb.st_size; 384 break; 385 default: 386 return (ENOTTY); 387 } 388 389 return (0); 390 } 391 392 /* 393 * Guest virtual machine i/o callbacks 394 */ 395 static int 396 cb_copyin(void *arg __unused, const void *from, uint64_t to, size_t size) 397 { 398 char *ptr; 399 400 to &= 0x7fffffff; 401 402 ptr = vm_map_gpa(ctx, to, size); 403 if (ptr == NULL) 404 return (EFAULT); 405 406 memcpy(ptr, from, size); 407 return (0); 408 } 409 410 static int 411 cb_copyout(void *arg __unused, uint64_t from, void *to, size_t size) 412 { 413 char *ptr; 414 415 from &= 0x7fffffff; 416 417 ptr = vm_map_gpa(ctx, from, size); 418 if (ptr == NULL) 419 return (EFAULT); 420 421 memcpy(to, ptr, size); 422 return (0); 423 } 424 425 static void 426 cb_setreg(void *arg __unused, int r, uint64_t v) 427 { 428 int error; 429 enum vm_reg_name vmreg; 430 431 vmreg = VM_REG_LAST; 432 433 switch (r) { 434 case 4: 435 vmreg = VM_REG_GUEST_RSP; 436 rsp = v; 437 break; 438 default: 439 break; 440 } 441 442 if (vmreg == VM_REG_LAST) { 443 printf("test_setreg(%d): not implemented\n", r); 444 cb_exit(NULL, USERBOOT_EXIT_QUIT); 445 } 446 447 error = vm_set_register(vcpu, vmreg, v); 448 if (error) { 449 perror("vm_set_register"); 450 cb_exit(NULL, USERBOOT_EXIT_QUIT); 451 } 452 } 453 454 static void 455 cb_setmsr(void *arg __unused, int r, uint64_t v) 456 { 457 int error; 458 enum vm_reg_name vmreg; 459 460 vmreg = VM_REG_LAST; 461 462 switch (r) { 463 case MSR_EFER: 464 vmreg = VM_REG_GUEST_EFER; 465 break; 466 default: 467 break; 468 } 469 470 if (vmreg == VM_REG_LAST) { 471 printf("test_setmsr(%d): not implemented\n", r); 472 cb_exit(NULL, USERBOOT_EXIT_QUIT); 473 } 474 475 error = vm_set_register(vcpu, vmreg, v); 476 if (error) { 477 perror("vm_set_msr"); 478 cb_exit(NULL, USERBOOT_EXIT_QUIT); 479 } 480 } 481 482 static void 483 cb_setcr(void *arg __unused, int r, uint64_t v) 484 { 485 int error; 486 enum vm_reg_name vmreg; 487 488 vmreg = VM_REG_LAST; 489 490 switch (r) { 491 case 0: 492 vmreg = VM_REG_GUEST_CR0; 493 break; 494 case 3: 495 vmreg = VM_REG_GUEST_CR3; 496 cr3 = v; 497 break; 498 case 4: 499 vmreg = VM_REG_GUEST_CR4; 500 break; 501 default: 502 break; 503 } 504 505 if (vmreg == VM_REG_LAST) { 506 printf("test_setcr(%d): not implemented\n", r); 507 cb_exit(NULL, USERBOOT_EXIT_QUIT); 508 } 509 510 error = vm_set_register(vcpu, vmreg, v); 511 if (error) { 512 perror("vm_set_cr"); 513 cb_exit(NULL, USERBOOT_EXIT_QUIT); 514 } 515 } 516 517 static void 518 cb_setgdt(void *arg __unused, uint64_t base, size_t size) 519 { 520 int error; 521 522 error = vm_set_desc(vcpu, VM_REG_GUEST_GDTR, base, size - 1, 0); 523 if (error != 0) { 524 perror("vm_set_desc(gdt)"); 525 cb_exit(NULL, USERBOOT_EXIT_QUIT); 526 } 527 528 gdtbase = base; 529 } 530 531 static void 532 cb_exec(void *arg __unused, uint64_t rip) 533 { 534 int error; 535 536 if (cr3 == 0) 537 error = vm_setup_freebsd_registers_i386(vcpu, rip, gdtbase, 538 rsp); 539 else 540 error = vm_setup_freebsd_registers(vcpu, rip, cr3, gdtbase, 541 rsp); 542 if (error) { 543 perror("vm_setup_freebsd_registers"); 544 cb_exit(NULL, USERBOOT_EXIT_QUIT); 545 } 546 547 cb_exit(NULL, 0); 548 } 549 550 /* 551 * Misc 552 */ 553 554 static void 555 cb_delay(void *arg __unused, int usec) 556 { 557 558 usleep(usec); 559 } 560 561 static void 562 cb_exit(void *arg __unused, int v) 563 { 564 565 tcsetattr(consout_fd, TCSAFLUSH, &oldterm); 566 if (v == USERBOOT_EXIT_REBOOT) 567 longjmp(jb, JMP_REBOOT); 568 exit(v); 569 } 570 571 static void 572 cb_getmem(void *arg __unused, uint64_t *ret_lowmem, uint64_t *ret_highmem) 573 { 574 575 *ret_lowmem = vm_get_lowmem_size(ctx); 576 *ret_highmem = vm_get_highmem_size(ctx); 577 } 578 579 struct env { 580 char *str; /* name=value */ 581 SLIST_ENTRY(env) next; 582 }; 583 584 static SLIST_HEAD(envhead, env) envhead; 585 586 static void 587 addenv(const char *str) 588 { 589 struct env *env; 590 591 env = malloc(sizeof(struct env)); 592 if (env == NULL) 593 err(EX_OSERR, "malloc"); 594 env->str = strdup(str); 595 if (env->str == NULL) 596 err(EX_OSERR, "strdup"); 597 SLIST_INSERT_HEAD(&envhead, env, next); 598 } 599 600 static char * 601 cb_getenv(void *arg __unused, int num) 602 { 603 int i; 604 struct env *env; 605 606 i = 0; 607 SLIST_FOREACH(env, &envhead, next) { 608 if (i == num) 609 return (env->str); 610 i++; 611 } 612 613 return (NULL); 614 } 615 616 static int 617 cb_vm_set_register(void *arg __unused, int vcpuid, int reg, uint64_t val) 618 { 619 620 assert(vcpuid == BSP); 621 return (vm_set_register(vcpu, reg, val)); 622 } 623 624 static int 625 cb_vm_set_desc(void *arg __unused, int vcpuid, int reg, uint64_t base, 626 u_int limit, u_int access) 627 { 628 629 assert(vcpuid == BSP); 630 return (vm_set_desc(vcpu, reg, base, limit, access)); 631 } 632 633 static void 634 cb_swap_interpreter(void *arg __unused, const char *interp_req) 635 { 636 637 /* 638 * If the user specified a loader but we detected a mismatch, we should 639 * not try to pivot to a different loader on them. 640 */ 641 free(loader); 642 if (explicit_loader_fd != -1) { 643 perror("requested loader interpreter does not match guest userboot"); 644 cb_exit(NULL, 1); 645 } 646 if (interp_req == NULL || *interp_req == '\0') { 647 perror("guest failed to request an interpreter"); 648 cb_exit(NULL, 1); 649 } 650 651 if (asprintf(&loader, "userboot_%s.so", interp_req) == -1) 652 err(EX_OSERR, "malloc"); 653 longjmp(jb, JMP_SWAPLOADER); 654 } 655 656 static void 657 cb_accept_interpreter(void *arg __unused) 658 { 659 if (bootfd == -1) 660 return; 661 662 close(bootfd); 663 bootfd = -1; 664 } 665 666 static struct loader_callbacks cb = { 667 .getc = cb_getc, 668 .putc = cb_putc, 669 .poll = cb_poll, 670 671 .open = cb_open, 672 .close = cb_close, 673 .isdir = cb_isdir, 674 .read = cb_read, 675 .readdir = cb_readdir, 676 .seek = cb_seek, 677 .stat = cb_stat, 678 679 .diskread = cb_diskread, 680 .diskwrite = cb_diskwrite, 681 .diskioctl = cb_diskioctl, 682 683 .copyin = cb_copyin, 684 .copyout = cb_copyout, 685 .setreg = cb_setreg, 686 .setmsr = cb_setmsr, 687 .setcr = cb_setcr, 688 .setgdt = cb_setgdt, 689 .exec = cb_exec, 690 691 .delay = cb_delay, 692 .exit = cb_exit, 693 .getmem = cb_getmem, 694 695 .getenv = cb_getenv, 696 697 /* Version 4 additions */ 698 .vm_set_register = cb_vm_set_register, 699 .vm_set_desc = cb_vm_set_desc, 700 701 /* Version 5 additions */ 702 .swap_interpreter = cb_swap_interpreter, 703 704 /* Version 6 additions */ 705 .accept_interpreter = cb_accept_interpreter, 706 }; 707 708 static int 709 altcons_open(char *path) 710 { 711 struct stat sb; 712 int err; 713 int fd; 714 715 /* 716 * Allow stdio to be passed in so that the same string 717 * can be used for the bhyveload console and bhyve com-port 718 * parameters 719 */ 720 if (!strcmp(path, "stdio")) 721 return (0); 722 723 err = stat(path, &sb); 724 if (err == 0) { 725 if (!S_ISCHR(sb.st_mode)) 726 err = ENOTSUP; 727 else { 728 fd = open(path, O_RDWR | O_NONBLOCK); 729 if (fd < 0) 730 err = errno; 731 else 732 consin_fd = consout_fd = fd; 733 } 734 } 735 736 return (err); 737 } 738 739 static int 740 disk_open(char *path) 741 { 742 struct stat sbuf; 743 off_t size; 744 int fd, ret, sectsz; 745 746 if (ndisks >= NDISKS) 747 return (ERANGE); 748 749 fd = open(path, O_RDWR); 750 if (fd < 0) 751 fd = open(path, O_RDONLY); 752 if (fd < 0) 753 return (errno); 754 if (fstat(fd, &sbuf) < 0) { 755 ret = errno; 756 goto err; 757 } 758 759 size = sbuf.st_size; 760 sectsz = DEV_BSIZE; 761 if (S_ISCHR(sbuf.st_mode)) { 762 if (ioctl(fd, DIOCGMEDIASIZE, &size) < 0 || 763 ioctl(fd, DIOCGSECTORSIZE, §sz) < 0) { 764 ret = errno; 765 goto err; 766 } 767 assert(size != 0); 768 assert(sectsz != 0); 769 } 770 771 disk_fd[ndisks] = fd; 772 ndisks++; 773 774 return (0); 775 err: 776 close(fd); 777 return (ret); 778 } 779 780 static void 781 usage(void) 782 { 783 784 fprintf(stderr, 785 "usage: %s [-S][-c <console-device>] [-d <disk-path>] [-e <name=value>]\n" 786 " %*s [-h <host-path>] [-m memsize[K|k|M|m|G|g|T|t]] <vmname>\n", 787 progname, 788 (int)strlen(progname), ""); 789 exit(1); 790 } 791 792 static void 793 hostbase_open(const char *base) 794 { 795 cap_rights_t rights; 796 797 if (hostbase_fd != -1) 798 close(hostbase_fd); 799 hostbase_fd = open(base, O_DIRECTORY | O_PATH); 800 if (hostbase_fd == -1) 801 err(EX_OSERR, "open"); 802 803 if (caph_rights_limit(hostbase_fd, cap_rights_init(&rights, CAP_FSTATAT, 804 CAP_LOOKUP, CAP_PREAD)) < 0) 805 err(EX_OSERR, "caph_rights_limit"); 806 } 807 808 static void 809 loader_open(void) 810 { 811 int fd; 812 813 if (loader == NULL) { 814 loader = strdup("userboot.so"); 815 if (loader == NULL) 816 err(EX_OSERR, "malloc"); 817 } 818 819 assert(bootfd >= 0 || explicit_loader_fd >= 0); 820 if (explicit_loader_fd >= 0) 821 fd = explicit_loader_fd; 822 else 823 fd = openat(bootfd, loader, O_RDONLY | O_RESOLVE_BENEATH); 824 if (fd == -1) 825 err(EX_OSERR, "openat"); 826 827 loader_hdl = fdlopen(fd, RTLD_LOCAL); 828 if (!loader_hdl) 829 errx(EX_OSERR, "dlopen: %s", dlerror()); 830 if (fd != explicit_loader_fd) 831 close(fd); 832 } 833 834 int 835 main(int argc, char** argv) 836 { 837 void (*func)(struct loader_callbacks *, void *, int, int); 838 uint64_t mem_size; 839 int opt, error, memflags, need_reinit; 840 841 progname = basename(argv[0]); 842 843 memflags = 0; 844 mem_size = 256 * MB; 845 846 consin_fd = STDIN_FILENO; 847 consout_fd = STDOUT_FILENO; 848 849 while ((opt = getopt(argc, argv, "CSc:d:e:h:l:m:")) != -1) { 850 switch (opt) { 851 case 'c': 852 error = altcons_open(optarg); 853 if (error != 0) 854 errx(EX_USAGE, "Could not open '%s'", optarg); 855 break; 856 857 case 'd': 858 error = disk_open(optarg); 859 if (error != 0) 860 errc(EX_USAGE, error, "Could not open '%s'", optarg); 861 break; 862 863 case 'e': 864 addenv(optarg); 865 break; 866 867 case 'h': 868 hostbase_open(optarg); 869 break; 870 871 case 'l': 872 if (loader != NULL) 873 errx(EX_USAGE, "-l can only be given once"); 874 loader = strdup(optarg); 875 if (loader == NULL) 876 err(EX_OSERR, "malloc"); 877 explicit_loader_fd = open(loader, O_RDONLY); 878 if (explicit_loader_fd == -1) 879 err(EX_OSERR, "%s", loader); 880 break; 881 882 case 'm': 883 error = vm_parse_memsize(optarg, &mem_size); 884 if (error != 0) 885 errx(EX_USAGE, "Invalid memsize '%s'", optarg); 886 break; 887 case 'C': 888 memflags |= VM_MEM_F_INCORE; 889 break; 890 case 'S': 891 memflags |= VM_MEM_F_WIRED; 892 break; 893 case '?': 894 usage(); 895 } 896 } 897 898 argc -= optind; 899 argv += optind; 900 901 if (argc != 1) 902 usage(); 903 904 vmname = argv[0]; 905 906 need_reinit = 0; 907 error = vm_create(vmname); 908 if (error) { 909 if (errno != EEXIST) 910 err(1, "vm_create"); 911 need_reinit = 1; 912 } 913 914 ctx = vm_open(vmname); 915 if (ctx == NULL) 916 err(1, "vm_open"); 917 918 /* 919 * If we weren't given an explicit loader to use, we need to support the 920 * guest requesting a different one. 921 */ 922 if (explicit_loader_fd == -1) { 923 cap_rights_t rights; 924 925 bootfd = open("/boot", O_DIRECTORY | O_PATH); 926 if (bootfd == -1) 927 err(1, "open"); 928 929 /* 930 * bootfd will be used to do a lookup of our loader and do an 931 * fdlopen(3) on the loader; thus, we need mmap(2) in addition 932 * to the more usual lookup rights. 933 */ 934 if (caph_rights_limit(bootfd, cap_rights_init(&rights, 935 CAP_FSTATAT, CAP_LOOKUP, CAP_MMAP_RX, CAP_PREAD)) < 0) 936 err(1, "caph_rights_limit"); 937 } 938 939 vcpu = vm_vcpu_open(ctx, BSP); 940 if (vcpu == NULL) 941 err(1, "vm_vcpu_open"); 942 943 caph_cache_catpages(); 944 if (caph_enter() < 0) 945 err(1, "caph_enter"); 946 947 /* 948 * setjmp in the case the guest wants to swap out interpreter, 949 * cb_swap_interpreter will swap out loader as appropriate and set 950 * need_reinit so that we end up in a clean state once again. 951 */ 952 if (setjmp(jb) != 0) { 953 dlclose(loader_hdl); 954 loader_hdl = NULL; 955 956 need_reinit = 1; 957 } 958 959 if (need_reinit) { 960 error = vm_reinit(ctx); 961 if (error) 962 err(1, "vm_reinit"); 963 } 964 965 vm_set_memflags(ctx, memflags); 966 error = vm_setup_memory(ctx, mem_size, VM_MMAP_ALL); 967 if (error) 968 err(1, "vm_setup_memory"); 969 970 loader_open(); 971 func = dlsym(loader_hdl, "loader_main"); 972 if (!func) 973 errx(1, "dlsym: %s", dlerror()); 974 975 tcgetattr(consout_fd, &term); 976 oldterm = term; 977 cfmakeraw(&term); 978 term.c_cflag |= CLOCAL; 979 980 tcsetattr(consout_fd, TCSAFLUSH, &term); 981 982 addenv("smbios.bios.vendor=BHYVE"); 983 addenv("boot_serial=1"); 984 985 func(&cb, NULL, USERBOOT_VERSION_6, ndisks); 986 987 free(loader); 988 return (0); 989 } 990