xref: /freebsd/sys/compat/linux/linux_emul.c (revision 3a1bf59d195ced99c0f69774969d7090d21f6097)
1 /*-
2  * SPDX-License-Identifier: BSD-2-Clause
3  *
4  * Copyright (c) 1994-1996 Søren Schmidt
5  * Copyright (c) 2006 Roman Divacky
6  * All rights reserved.
7  * Copyright (c) 2013 Dmitry Chagin <dchagin@FreeBSD.org>
8  *
9  * Redistribution and use in source and binary forms, with or without
10  * modification, are permitted provided that the following conditions
11  * are met:
12  * 1. Redistributions of source code must retain the above copyright
13  *    notice, this list of conditions and the following disclaimer.
14  * 2. Redistributions in binary form must reproduce the above copyright
15  *    notice, this list of conditions and the following disclaimer in the
16  *    documentation and/or other materials provided with the distribution.
17  *
18  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
19  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
20  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
21  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
22  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
23  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
24  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
25  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
26  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
27  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
28  * SUCH DAMAGE.
29  */
30 
31 #include <sys/param.h>
32 #include <sys/fcntl.h>
33 #include <sys/imgact.h>
34 #include <sys/ktr.h>
35 #include <sys/lock.h>
36 #include <sys/malloc.h>
37 #include <sys/mutex.h>
38 #include <sys/proc.h>
39 #include <sys/resourcevar.h>
40 #include <sys/sx.h>
41 #include <sys/syscallsubr.h>
42 #include <sys/sysent.h>
43 
44 #include <compat/linux/linux_emul.h>
45 #include <compat/linux/linux_mib.h>
46 #include <compat/linux/linux_misc.h>
47 #include <compat/linux/linux_persona.h>
48 #include <compat/linux/linux_util.h>
49 
50 #if BYTE_ORDER == LITTLE_ENDIAN
51 #define SHELLMAGIC	0x2123 /* #! */
52 #else
53 #define SHELLMAGIC	0x2321
54 #endif
55 
56 /*
57  * This returns reference to the thread emuldata entry (if found)
58  *
59  * Hold PROC_LOCK when referencing emuldata from other threads.
60  */
61 struct linux_emuldata *
62 em_find(struct thread *td)
63 {
64 	struct linux_emuldata *em;
65 
66 	em = td->td_emuldata;
67 
68 	return (em);
69 }
70 
71 /*
72  * This returns reference to the proc pemuldata entry (if found)
73  *
74  * Hold PROC_LOCK when referencing proc pemuldata from other threads.
75  * Hold LINUX_PEM_LOCK wher referencing pemuldata members.
76  */
77 struct linux_pemuldata *
78 pem_find(struct proc *p)
79 {
80 	struct linux_pemuldata *pem;
81 
82 	pem = p->p_emuldata;
83 
84 	return (pem);
85 }
86 
87 /*
88  * Linux apps generally expect the soft open file limit to be set
89  * to 1024, often iterating over all the file descriptors up to that
90  * limit instead of using closefrom(2).  Give them what they want,
91  * unless there already is a resource limit in place.
92  */
93 static void
94 linux_set_default_openfiles(struct thread *td, struct proc *p)
95 {
96 	struct rlimit rlim;
97 	int error __diagused;
98 
99 	if (linux_default_openfiles < 0)
100 		return;
101 
102 	PROC_LOCK(p);
103 	lim_rlimit_proc(p, RLIMIT_NOFILE, &rlim);
104 	PROC_UNLOCK(p);
105 	if (rlim.rlim_cur != rlim.rlim_max ||
106 	    rlim.rlim_cur <= linux_default_openfiles)
107 		return;
108 	rlim.rlim_cur = linux_default_openfiles;
109 	error = kern_proc_setrlimit(td, p, RLIMIT_NOFILE, &rlim);
110 	KASSERT(error == 0, ("kern_proc_setrlimit failed"));
111 }
112 
113 /*
114  * The default stack size limit in Linux is 8MB.
115  */
116 static void
117 linux_set_default_stacksize(struct thread *td, struct proc *p)
118 {
119 	struct rlimit rlim;
120 	int error __diagused;
121 
122 	if (linux_default_stacksize < 0)
123 		return;
124 
125 	PROC_LOCK(p);
126 	lim_rlimit_proc(p, RLIMIT_STACK, &rlim);
127 	PROC_UNLOCK(p);
128 	if (rlim.rlim_cur != rlim.rlim_max ||
129 	    rlim.rlim_cur <= linux_default_stacksize)
130 		return;
131 	rlim.rlim_cur = linux_default_stacksize;
132 	error = kern_proc_setrlimit(td, p, RLIMIT_STACK, &rlim);
133 	KASSERT(error == 0, ("kern_proc_setrlimit failed"));
134 }
135 
136 void
137 linux_proc_init(struct thread *td, struct thread *newtd, bool init_thread)
138 {
139 	struct linux_emuldata *em;
140 	struct linux_pemuldata *pem;
141 	struct proc *p;
142 
143 	if (newtd != NULL) {
144 		p = newtd->td_proc;
145 
146 		/* non-exec call */
147 		em = malloc(sizeof(*em), M_LINUX, M_WAITOK | M_ZERO);
148 		if (init_thread) {
149 			LINUX_CTR1(proc_init, "thread newtd(%d)",
150 			    newtd->td_tid);
151 
152 			em->em_tid = newtd->td_tid;
153 		} else {
154 			LINUX_CTR1(proc_init, "fork newtd(%d)", p->p_pid);
155 
156 			em->em_tid = p->p_pid;
157 
158 			pem = malloc(sizeof(*pem), M_LINUX, M_WAITOK | M_ZERO);
159 			sx_init(&pem->pem_sx, "lpemlk");
160 			linux_pemuldata_init_md(td, pem);
161 			p->p_emuldata = pem;
162 		}
163 		newtd->td_emuldata = em;
164 
165 		linux_set_default_openfiles(td, p);
166 		linux_set_default_stacksize(td, p);
167 	} else {
168 		p = td->td_proc;
169 
170 		/* exec */
171 		LINUX_CTR1(proc_init, "exec newtd(%d)", p->p_pid);
172 
173 		/* lookup the old one */
174 		em = em_find(td);
175 		KASSERT(em != NULL, ("proc_init: thread emuldata not found.\n"));
176 
177 		em->em_tid = p->p_pid;
178 		em->flags = 0;
179 		em->robust_futexes = NULL;
180 		em->child_clear_tid = NULL;
181 		em->child_set_tid = NULL;
182 
183 		pem = pem_find(p);
184 		KASSERT(pem != NULL, ("proc_init: proc emuldata not found.\n"));
185 		pem->persona = 0;
186 		pem->oom_score_adj = 0;
187 		linux_pemuldata_exec_md(pem);
188 	}
189 }
190 
191 void
192 linux_on_exit(struct proc *p)
193 {
194 	struct linux_pemuldata *pem;
195 	struct thread *td = curthread;
196 
197 	MPASS(SV_CURPROC_ABI() == SV_ABI_LINUX);
198 
199 	LINUX_CTR3(proc_exit, "thread(%d) proc(%d) p %p",
200 	    td->td_tid, p->p_pid, p);
201 
202 	pem = pem_find(p);
203 	if (pem == NULL)
204 		return;
205 	(p->p_sysent->sv_thread_detach)(td);
206 
207 	p->p_emuldata = NULL;
208 
209 	sx_destroy(&pem->pem_sx);
210 	free(pem, M_LINUX);
211 }
212 
213 int
214 linux_common_execve(struct thread *td, struct image_args *eargs)
215 {
216 	struct linux_pemuldata *pem;
217 	struct vmspace *oldvmspace;
218 	struct linux_emuldata *em;
219 	struct proc *p;
220 	int error;
221 
222 	p = td->td_proc;
223 
224 	error = pre_execve(td, &oldvmspace);
225 	if (error != 0)
226 		return (error);
227 
228 	error = kern_execve(td, eargs, NULL, oldvmspace);
229 	post_execve(td, error, oldvmspace);
230 	if (error != EJUSTRETURN)
231 		return (error);
232 
233 	/*
234 	 * In a case of transition from Linux binary execing to
235 	 * FreeBSD binary we destroy Linux emuldata thread & proc entries.
236 	 */
237 	if (SV_CURPROC_ABI() != SV_ABI_LINUX) {
238 
239 		/* Clear ABI root directory if set. */
240 		linux_pwd_onexec_native(td);
241 
242 		PROC_LOCK(p);
243 		em = em_find(td);
244 		KASSERT(em != NULL, ("proc_exec: thread emuldata not found.\n"));
245 		td->td_emuldata = NULL;
246 
247 		pem = pem_find(p);
248 		KASSERT(pem != NULL, ("proc_exec: proc pemuldata not found.\n"));
249 		p->p_emuldata = NULL;
250 		PROC_UNLOCK(p);
251 
252 		free(em, M_LINUX);
253 		free(pem, M_LINUX);
254 	}
255 	return (EJUSTRETURN);
256 }
257 
258 int
259 linux_on_exec(struct proc *p, struct image_params *imgp)
260 {
261 	struct thread *td;
262 	struct thread *othertd;
263 #if defined(__amd64__)
264 	struct linux_pemuldata *pem;
265 #endif
266 	int error;
267 
268 	td = curthread;
269 	MPASS((imgp->sysent->sv_flags & SV_ABI_MASK) == SV_ABI_LINUX);
270 
271 	/*
272 	 * When execing to Linux binary, we create Linux emuldata
273 	 * thread entry.
274 	 */
275 	if (SV_PROC_ABI(p) == SV_ABI_LINUX) {
276 		/*
277 		 * Process already was under Linuxolator
278 		 * before exec.  Update emuldata to reflect
279 		 * single-threaded cleaned state after exec.
280 		 */
281 		linux_proc_init(td, NULL, false);
282 	} else {
283 		/*
284 		 * We are switching the process to Linux emulator.
285 		 */
286 		linux_proc_init(td, td, false);
287 
288 		/*
289 		 * Create a transient td_emuldata for all suspended
290 		 * threads, so that p->p_sysent->sv_thread_detach() ==
291 		 * linux_thread_detach() can find expected but unused
292 		 * emuldata.
293 		 */
294 		FOREACH_THREAD_IN_PROC(td->td_proc, othertd) {
295 			if (othertd == td)
296 				continue;
297 			linux_proc_init(td, othertd, true);
298 		}
299 
300 		/* Set ABI root directory. */
301 		if ((error = linux_pwd_onexec(td)) != 0)
302 			return (error);
303 	}
304 #if defined(__amd64__)
305 	/*
306 	 * An IA32 executable which has executable stack will have the
307 	 * READ_IMPLIES_EXEC personality flag set automatically.
308 	 */
309 	if (SV_PROC_FLAG(td->td_proc, SV_ILP32) &&
310 	    imgp->stack_prot & VM_PROT_EXECUTE) {
311 		pem = pem_find(p);
312 		pem->persona |= LINUX_READ_IMPLIES_EXEC;
313 	}
314 #endif
315 	return (0);
316 }
317 
318 void
319 linux_thread_dtor(struct thread *td)
320 {
321 	struct linux_emuldata *em;
322 
323 	em = em_find(td);
324 	if (em == NULL)
325 		return;
326 	td->td_emuldata = NULL;
327 
328 	LINUX_CTR1(thread_dtor, "thread(%d)", em->em_tid);
329 
330 	free(em, M_LINUX);
331 }
332 
333 void
334 linux_schedtail(struct thread *td)
335 {
336 	struct linux_emuldata *em;
337 #ifdef KTR
338 	int error;
339 #else
340 	int error __unused;
341 #endif
342 	int *child_set_tid;
343 
344 	em = em_find(td);
345 	KASSERT(em != NULL, ("linux_schedtail: thread emuldata not found.\n"));
346 	child_set_tid = em->child_set_tid;
347 
348 	if (child_set_tid != NULL) {
349 		error = copyout(&em->em_tid, child_set_tid,
350 		    sizeof(em->em_tid));
351 		LINUX_CTR4(schedtail, "thread(%d) %p stored %d error %d",
352 		    td->td_tid, child_set_tid, em->em_tid, error);
353 	} else
354 		LINUX_CTR1(schedtail, "thread(%d)", em->em_tid);
355 }
356