xref: /freebsd-12.1/sys/compat/linux/linux_misc.c (revision cc6219c5)
1 /*-
2  * Copyright (c) 1994-1995 S�ren Schmidt
3  * All rights reserved.
4  *
5  * Redistribution and use in source and binary forms, with or without
6  * modification, are permitted provided that the following conditions
7  * are met:
8  * 1. Redistributions of source code must retain the above copyright
9  *    notice, this list of conditions and the following disclaimer
10  *    in this position and unchanged.
11  * 2. Redistributions in binary form must reproduce the above copyright
12  *    notice, this list of conditions and the following disclaimer in the
13  *    documentation and/or other materials provided with the distribution.
14  * 3. The name of the author may not be used to endorse or promote products
15  *    derived from this software withough specific prior written permission
16  *
17  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
18  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
19  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
20  * IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, INDIRECT,
21  * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
22  * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
23  * DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
24  * THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
25  * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
26  * THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
27  *
28  *  $Id: linux_misc.c,v 1.60 1999/08/08 11:26:46 marcel Exp $
29  */
30 
31 #include <sys/param.h>
32 #include <sys/systm.h>
33 #include <sys/sysproto.h>
34 #include <sys/kernel.h>
35 #include <sys/mman.h>
36 #include <sys/proc.h>
37 #include <sys/fcntl.h>
38 #include <sys/imgact_aout.h>
39 #include <sys/mount.h>
40 #include <sys/namei.h>
41 #include <sys/resourcevar.h>
42 #include <sys/stat.h>
43 #include <sys/sysctl.h>
44 #include <sys/unistd.h>
45 #include <sys/vnode.h>
46 #include <sys/wait.h>
47 #include <sys/time.h>
48 
49 #include <vm/vm.h>
50 #include <vm/pmap.h>
51 #include <vm/vm_kern.h>
52 #include <vm/vm_prot.h>
53 #include <vm/vm_map.h>
54 #include <vm/vm_extern.h>
55 
56 #include <machine/frame.h>
57 #include <machine/psl.h>
58 
59 #include <i386/linux/linux.h>
60 #include <i386/linux/linux_proto.h>
61 #include <i386/linux/linux_util.h>
62 
63 int osetrlimit __P((struct proc*, struct linux_setrlimit_args*));
64 int ogetrlimit __P((struct proc*, struct linux_getrlimit_args*));
65 
66 static unsigned int linux_to_bsd_resource[LINUX_RLIM_NLIMITS] =
67 { RLIMIT_CPU, RLIMIT_FSIZE, RLIMIT_DATA, RLIMIT_STACK,
68   RLIMIT_CORE, RLIMIT_RSS, RLIMIT_NPROC, RLIMIT_NOFILE,
69   RLIMIT_MEMLOCK, -1
70 };
71 
72 int
73 linux_alarm(struct proc *p, struct linux_alarm_args *args)
74 {
75     struct itimerval it, old_it;
76     struct timeval tv;
77     int s;
78 
79 #ifdef DEBUG
80     printf("Linux-emul(%ld): alarm(%u)\n", (long)p->p_pid, args->secs);
81 #endif
82     if (args->secs > 100000000)
83 	return EINVAL;
84     it.it_value.tv_sec = (long)args->secs;
85     it.it_value.tv_usec = 0;
86     it.it_interval.tv_sec = 0;
87     it.it_interval.tv_usec = 0;
88     s = splsoftclock();
89     old_it = p->p_realtimer;
90     getmicrouptime(&tv);
91     if (timevalisset(&old_it.it_value))
92 	untimeout(realitexpire, (caddr_t)p, p->p_ithandle);
93     if (it.it_value.tv_sec != 0) {
94 	p->p_ithandle = timeout(realitexpire, (caddr_t)p, tvtohz(&it.it_value));
95 	timevaladd(&it.it_value, &tv);
96     }
97     p->p_realtimer = it;
98     splx(s);
99     if (timevalcmp(&old_it.it_value, &tv, >)) {
100 	timevalsub(&old_it.it_value, &tv);
101 	if (old_it.it_value.tv_usec != 0)
102 	    old_it.it_value.tv_sec++;
103 	p->p_retval[0] = old_it.it_value.tv_sec;
104     }
105     return 0;
106 }
107 
108 int
109 linux_brk(struct proc *p, struct linux_brk_args *args)
110 {
111 #if 0
112     struct vmspace *vm = p->p_vmspace;
113     vm_offset_t new, old;
114     int error;
115 
116     if ((vm_offset_t)args->dsend < (vm_offset_t)vm->vm_daddr)
117 	return EINVAL;
118     if (((caddr_t)args->dsend - (caddr_t)vm->vm_daddr)
119 	> p->p_rlimit[RLIMIT_DATA].rlim_cur)
120 	return ENOMEM;
121 
122     old = round_page((vm_offset_t)vm->vm_daddr) + ctob(vm->vm_dsize);
123     new = round_page((vm_offset_t)args->dsend);
124     p->p_retval[0] = old;
125     if ((new-old) > 0) {
126 	if (swap_pager_full)
127 	    return ENOMEM;
128 	error = vm_map_find(&vm->vm_map, NULL, 0, &old, (new-old), FALSE,
129 			VM_PROT_ALL, VM_PROT_ALL, 0);
130 	if (error)
131 	    return error;
132 	vm->vm_dsize += btoc((new-old));
133 	p->p_retval[0] = (int)(vm->vm_daddr + ctob(vm->vm_dsize));
134     }
135     return 0;
136 #else
137     struct vmspace *vm = p->p_vmspace;
138     vm_offset_t new, old;
139     struct obreak_args /* {
140 	char * nsize;
141     } */ tmp;
142 
143 #ifdef DEBUG
144     printf("Linux-emul(%ld): brk(%p)\n", (long)p->p_pid, (void *)args->dsend);
145 #endif
146     old = (vm_offset_t)vm->vm_daddr + ctob(vm->vm_dsize);
147     new = (vm_offset_t)args->dsend;
148     tmp.nsize = (char *) new;
149     if (((caddr_t)new > vm->vm_daddr) && !obreak(p, &tmp))
150 	p->p_retval[0] = (int)new;
151     else
152 	p->p_retval[0] = (int)old;
153 
154     return 0;
155 #endif
156 }
157 
158 int
159 linux_uselib(struct proc *p, struct linux_uselib_args *args)
160 {
161     struct nameidata ni;
162     struct vnode *vp;
163     struct exec *a_out;
164     struct vattr attr;
165     vm_offset_t vmaddr;
166     unsigned long file_offset;
167     vm_offset_t buffer;
168     unsigned long bss_size;
169     int error;
170     caddr_t sg;
171     int locked;
172 
173     sg = stackgap_init();
174     CHECKALTEXIST(p, &sg, args->library);
175 
176 #ifdef DEBUG
177     printf("Linux-emul(%d): uselib(%s)\n", p->p_pid, args->library);
178 #endif
179 
180     a_out = NULL;
181     locked = 0;
182     vp = NULL;
183 
184     NDINIT(&ni, LOOKUP, FOLLOW | LOCKLEAF, UIO_USERSPACE, args->library, p);
185     error = namei(&ni);
186     if (error)
187 	goto cleanup;
188 
189     vp = ni.ni_vp;
190     if (vp == NULL) {
191 	error = ENOEXEC;	/* ?? */
192 	goto cleanup;
193     }
194 
195     /*
196      * From here on down, we have a locked vnode that must be unlocked.
197      */
198     locked++;
199 
200     /*
201      * Writable?
202      */
203     if (vp->v_writecount) {
204 	error = ETXTBSY;
205 	goto cleanup;
206     }
207 
208     /*
209      * Executable?
210      */
211     error = VOP_GETATTR(vp, &attr, p->p_ucred, p);
212     if (error)
213 	goto cleanup;
214 
215     if ((vp->v_mount->mnt_flag & MNT_NOEXEC) ||
216 	((attr.va_mode & 0111) == 0) ||
217 	(attr.va_type != VREG)) {
218 	    error = ENOEXEC;
219 	    goto cleanup;
220     }
221 
222     /*
223      * Sensible size?
224      */
225     if (attr.va_size == 0) {
226 	error = ENOEXEC;
227 	goto cleanup;
228     }
229 
230     /*
231      * Can we access it?
232      */
233     error = VOP_ACCESS(vp, VEXEC, p->p_ucred, p);
234     if (error)
235 	goto cleanup;
236 
237     error = VOP_OPEN(vp, FREAD, p->p_ucred, p);
238     if (error)
239 	goto cleanup;
240 
241     /*
242      * Lock no longer needed
243      */
244     VOP_UNLOCK(vp, 0, p);
245     locked = 0;
246 
247     /*
248      * Pull in executable header into kernel_map
249      */
250     error = vm_mmap(kernel_map, (vm_offset_t *)&a_out, PAGE_SIZE,
251 	    	    VM_PROT_READ, VM_PROT_READ, 0, (caddr_t)vp, 0);
252     if (error)
253 	goto cleanup;
254 
255     /*
256      * Is it a Linux binary ?
257      */
258     if (((a_out->a_magic >> 16) & 0xff) != 0x64) {
259 	error = ENOEXEC;
260 	goto cleanup;
261     }
262 
263     /* While we are here, we should REALLY do some more checks */
264 
265     /*
266      * Set file/virtual offset based on a.out variant.
267      */
268     switch ((int)(a_out->a_magic & 0xffff)) {
269     case 0413:	/* ZMAGIC */
270 	file_offset = 1024;
271 	break;
272     case 0314:	/* QMAGIC */
273 	file_offset = 0;
274 	break;
275     default:
276 	error = ENOEXEC;
277 	goto cleanup;
278     }
279 
280     bss_size = round_page(a_out->a_bss);
281 
282     /*
283      * Check various fields in header for validity/bounds.
284      */
285     if (a_out->a_text & PAGE_MASK || a_out->a_data & PAGE_MASK) {
286 	error = ENOEXEC;
287 	goto cleanup;
288     }
289 
290     /* text + data can't exceed file size */
291     if (a_out->a_data + a_out->a_text > attr.va_size) {
292 	error = EFAULT;
293 	goto cleanup;
294     }
295 
296     /*
297      * text/data/bss must not exceed limits
298      * XXX: this is not complete. it should check current usage PLUS
299      * the resources needed by this library.
300      */
301     if (a_out->a_text > MAXTSIZ ||
302 	a_out->a_data + bss_size > p->p_rlimit[RLIMIT_DATA].rlim_cur) {
303 	error = ENOMEM;
304 	goto cleanup;
305     }
306 
307     /*
308      * prevent more writers
309      */
310     vp->v_flag |= VTEXT;
311 
312     /*
313      * Check if file_offset page aligned,.
314      * Currently we cannot handle misalinged file offsets,
315      * and so we read in the entire image (what a waste).
316      */
317     if (file_offset & PAGE_MASK) {
318 #ifdef DEBUG
319 printf("uselib: Non page aligned binary %lu\n", file_offset);
320 #endif
321 	/*
322 	 * Map text+data read/write/execute
323 	 */
324 
325 	/* a_entry is the load address and is page aligned */
326 	vmaddr = trunc_page(a_out->a_entry);
327 
328 	/* get anon user mapping, read+write+execute */
329 	error = vm_map_find(&p->p_vmspace->vm_map, NULL, 0, &vmaddr,
330 		    	    a_out->a_text + a_out->a_data, FALSE,
331 			    VM_PROT_ALL, VM_PROT_ALL, 0);
332 	if (error)
333 	    goto cleanup;
334 
335 	/* map file into kernel_map */
336 	error = vm_mmap(kernel_map, &buffer,
337 			round_page(a_out->a_text + a_out->a_data + file_offset),
338 		   	VM_PROT_READ, VM_PROT_READ, 0,
339 			(caddr_t)vp, trunc_page(file_offset));
340 	if (error)
341 	    goto cleanup;
342 
343 	/* copy from kernel VM space to user space */
344 	error = copyout((caddr_t)(void *)(uintptr_t)(buffer + file_offset),
345 			(caddr_t)vmaddr, a_out->a_text + a_out->a_data);
346 
347 	/* release temporary kernel space */
348 	vm_map_remove(kernel_map, buffer,
349 		      buffer + round_page(a_out->a_text + a_out->a_data + file_offset));
350 
351 	if (error)
352 	    goto cleanup;
353     }
354     else {
355 #ifdef DEBUG
356 printf("uselib: Page aligned binary %lu\n", file_offset);
357 #endif
358 	/*
359 	 * for QMAGIC, a_entry is 20 bytes beyond the load address
360 	 * to skip the executable header
361 	 */
362 	vmaddr = trunc_page(a_out->a_entry);
363 
364 	/*
365 	 * Map it all into the process's space as a single copy-on-write
366 	 * "data" segment.
367 	 */
368 	error = vm_mmap(&p->p_vmspace->vm_map, &vmaddr,
369 		   	a_out->a_text + a_out->a_data,
370 			VM_PROT_ALL, VM_PROT_ALL, MAP_PRIVATE | MAP_FIXED,
371 			(caddr_t)vp, file_offset);
372 	if (error)
373 	    goto cleanup;
374     }
375 #ifdef DEBUG
376 printf("mem=%08x = %08x %08x\n", vmaddr, ((int*)vmaddr)[0], ((int*)vmaddr)[1]);
377 #endif
378     if (bss_size != 0) {
379         /*
380 	 * Calculate BSS start address
381 	 */
382 	vmaddr = trunc_page(a_out->a_entry) + a_out->a_text + a_out->a_data;
383 
384 	/*
385 	 * allocate some 'anon' space
386 	 */
387 	error = vm_map_find(&p->p_vmspace->vm_map, NULL, 0, &vmaddr,
388 			    bss_size, FALSE,
389 			    VM_PROT_ALL, VM_PROT_ALL, 0);
390 	if (error)
391 	    goto cleanup;
392     }
393 
394 cleanup:
395     /*
396      * Unlock vnode if needed
397      */
398     if (locked)
399 	VOP_UNLOCK(vp, 0, p);
400 
401     /*
402      * Release the kernel mapping.
403      */
404     if (a_out)
405 	vm_map_remove(kernel_map, (vm_offset_t)a_out, (vm_offset_t)a_out + PAGE_SIZE);
406 
407     return error;
408 }
409 
410 /* XXX move */
411 struct linux_select_argv {
412 	int nfds;
413 	fd_set *readfds;
414 	fd_set *writefds;
415 	fd_set *exceptfds;
416 	struct timeval *timeout;
417 };
418 
419 int
420 linux_select(struct proc *p, struct linux_select_args *args)
421 {
422     struct linux_select_argv linux_args;
423     struct linux_newselect_args newsel;
424     int error;
425 
426 #ifdef SELECT_DEBUG
427     printf("Linux-emul(%d): select(%x)\n",
428 	   p->p_pid, args->ptr);
429 #endif
430     if ((error = copyin((caddr_t)args->ptr, (caddr_t)&linux_args,
431 			sizeof(linux_args))))
432 	return error;
433 
434     newsel.nfds = linux_args.nfds;
435     newsel.readfds = linux_args.readfds;
436     newsel.writefds = linux_args.writefds;
437     newsel.exceptfds = linux_args.exceptfds;
438     newsel.timeout = linux_args.timeout;
439 
440     return linux_newselect(p, &newsel);
441 }
442 
443 int
444 linux_newselect(struct proc *p, struct linux_newselect_args *args)
445 {
446     struct select_args bsa;
447     struct timeval tv0, tv1, utv, *tvp;
448     caddr_t sg;
449     int error;
450 
451 #ifdef DEBUG
452     printf("Linux-emul(%ld): newselect(%d, %p, %p, %p, %p)\n",
453   	(long)p->p_pid, args->nfds, (void *)args->readfds,
454 	(void *)args->writefds, (void *)args->exceptfds,
455 	(void *)args->timeout);
456 #endif
457     error = 0;
458     bsa.nd = args->nfds;
459     bsa.in = args->readfds;
460     bsa.ou = args->writefds;
461     bsa.ex = args->exceptfds;
462     bsa.tv = args->timeout;
463 
464     /*
465      * Store current time for computation of the amount of
466      * time left.
467      */
468     if (args->timeout) {
469 	if ((error = copyin(args->timeout, &utv, sizeof(utv))))
470 	    goto select_out;
471 #ifdef DEBUG
472 	printf("Linux-emul(%ld): incoming timeout (%ld/%ld)\n",
473 	    (long)p->p_pid, utv.tv_sec, utv.tv_usec);
474 #endif
475 	if (itimerfix(&utv)) {
476 	    /*
477 	     * The timeval was invalid.  Convert it to something
478 	     * valid that will act as it does under Linux.
479 	     */
480 	    sg = stackgap_init();
481 	    tvp = stackgap_alloc(&sg, sizeof(utv));
482 	    utv.tv_sec += utv.tv_usec / 1000000;
483 	    utv.tv_usec %= 1000000;
484 	    if (utv.tv_usec < 0) {
485 		utv.tv_sec -= 1;
486 		utv.tv_usec += 1000000;
487 	    }
488 	    if (utv.tv_sec < 0)
489 		timevalclear(&utv);
490 	    if ((error = copyout(&utv, tvp, sizeof(utv))))
491 		goto select_out;
492 	    bsa.tv = tvp;
493 	}
494 	microtime(&tv0);
495     }
496 
497     error = select(p, &bsa);
498 #ifdef DEBUG
499     printf("Linux-emul(%d): real select returns %d\n",
500 	       p->p_pid, error);
501 #endif
502 
503     if (error) {
504 	/*
505 	 * See fs/select.c in the Linux kernel.  Without this,
506 	 * Maelstrom doesn't work.
507 	 */
508 	if (error == ERESTART)
509 	    error = EINTR;
510 	goto select_out;
511     }
512 
513     if (args->timeout) {
514 	if (p->p_retval[0]) {
515 	    /*
516 	     * Compute how much time was left of the timeout,
517 	     * by subtracting the current time and the time
518 	     * before we started the call, and subtracting
519 	     * that result from the user-supplied value.
520 	     */
521 	    microtime(&tv1);
522 	    timevalsub(&tv1, &tv0);
523 	    timevalsub(&utv, &tv1);
524 	    if (utv.tv_sec < 0)
525 		timevalclear(&utv);
526 	} else
527 	    timevalclear(&utv);
528 #ifdef DEBUG
529 	printf("Linux-emul(%ld): outgoing timeout (%ld/%ld)\n",
530 	    (long)p->p_pid, utv.tv_sec, utv.tv_usec);
531 #endif
532 	if ((error = copyout(&utv, args->timeout, sizeof(utv))))
533 	    goto select_out;
534     }
535 
536 select_out:
537 #ifdef DEBUG
538     printf("Linux-emul(%d): newselect_out -> %d\n",
539 	       p->p_pid, error);
540 #endif
541     return error;
542 }
543 
544 int
545 linux_getpgid(struct proc *p, struct linux_getpgid_args *args)
546 {
547     struct proc *curp;
548 
549 #ifdef DEBUG
550     printf("Linux-emul(%d): getpgid(%d)\n", p->p_pid, args->pid);
551 #endif
552     if (args->pid != p->p_pid) {
553 	if (!(curp = pfind(args->pid)))
554 	    return ESRCH;
555     }
556     else
557 	curp = p;
558     p->p_retval[0] = curp->p_pgid;
559     return 0;
560 }
561 
562 int
563 linux_fork(struct proc *p, struct linux_fork_args *args)
564 {
565     int error;
566 
567 #ifdef DEBUG
568     printf("Linux-emul(%d): fork()\n", p->p_pid);
569 #endif
570     if ((error = fork(p, (struct fork_args *)args)) != 0)
571 	return error;
572     if (p->p_retval[1] == 1)
573 	p->p_retval[0] = 0;
574     return 0;
575 }
576 
577 #define CLONE_VM	0x100
578 #define CLONE_FS	0x200
579 #define CLONE_FILES	0x400
580 #define CLONE_SIGHAND	0x800
581 #define CLONE_PID	0x1000
582 
583 int
584 linux_clone(struct proc *p, struct linux_clone_args *args)
585 {
586     int error, ff = RFPROC;
587     struct proc *p2;
588     int            exit_signal;
589     vm_offset_t    start;
590     struct rfork_args rf_args;
591 
592 #ifdef DEBUG
593     if (args->flags & CLONE_PID)
594 	printf("linux_clone(%d): CLONE_PID not yet supported\n", p->p_pid);
595     printf ("linux_clone(%d): invoked with flags %x and stack %x\n", p->p_pid,
596 	     (unsigned int)args->flags, (unsigned int)args->stack);
597 #endif
598 
599     if (!args->stack)
600         return (EINVAL);
601 
602     exit_signal = args->flags & 0x000000ff;
603     if (exit_signal >= LINUX_NSIG)
604 	return EINVAL;
605     exit_signal = linux_to_bsd_signal[exit_signal];
606 
607     /* RFTHREAD probably not necessary here, but it shouldn't hurt either */
608     ff |= RFTHREAD;
609 
610     if (args->flags & CLONE_VM)
611 	ff |= RFMEM;
612     if (args->flags & CLONE_SIGHAND)
613 	ff |= RFSIGSHARE;
614     if (!(args->flags & CLONE_FILES))
615 	ff |= RFFDG;
616 
617     error = 0;
618     start = 0;
619 
620     rf_args.flags = ff;
621     if ((error = rfork(p, &rf_args)) != 0)
622 	return error;
623 
624     p2 = pfind(p->p_retval[0]);
625     if (p2 == 0)
626  	return ESRCH;
627 
628     p2->p_sigparent = exit_signal;
629     p2->p_md.md_regs->tf_esp = (unsigned int)args->stack;
630 
631 #ifdef DEBUG
632     printf ("linux_clone(%d): successful rfork to %d\n", p->p_pid, p2->p_pid);
633 #endif
634     return 0;
635 }
636 
637 /* XXX move */
638 struct linux_mmap_argv {
639 	linux_caddr_t addr;
640 	int len;
641 	int prot;
642 	int flags;
643 	int fd;
644 	int pos;
645 };
646 
647 #define STACK_SIZE  (2 * 1024 * 1024)
648 #define GUARD_SIZE  (4 * PAGE_SIZE)
649 int
650 linux_mmap(struct proc *p, struct linux_mmap_args *args)
651 {
652     struct mmap_args /* {
653 	caddr_t addr;
654 	size_t len;
655 	int prot;
656 	int flags;
657 	int fd;
658 	long pad;
659 	off_t pos;
660     } */ bsd_args;
661     int error;
662     struct linux_mmap_argv linux_args;
663 
664     if ((error = copyin((caddr_t)args->ptr, (caddr_t)&linux_args,
665 			sizeof(linux_args))))
666 	return error;
667 #ifdef DEBUG
668     printf("Linux-emul(%ld): mmap(%p, %d, %d, %08x, %d, %d)\n",
669 	(long)p->p_pid, (void *)linux_args.addr, linux_args.len,
670 	linux_args.prot, linux_args.flags, linux_args.fd, linux_args.pos);
671 #endif
672     bsd_args.flags = 0;
673     if (linux_args.flags & LINUX_MAP_SHARED)
674 	bsd_args.flags |= MAP_SHARED;
675     if (linux_args.flags & LINUX_MAP_PRIVATE)
676 	bsd_args.flags |= MAP_PRIVATE;
677     if (linux_args.flags & LINUX_MAP_FIXED)
678 	bsd_args.flags |= MAP_FIXED;
679     if (linux_args.flags & LINUX_MAP_ANON)
680 	bsd_args.flags |= MAP_ANON;
681     if (linux_args.flags & LINUX_MAP_GROWSDOWN) {
682 	bsd_args.flags |= MAP_STACK;
683 
684 	/* The linux MAP_GROWSDOWN option does not limit auto
685 	 * growth of the region.  Linux mmap with this option
686 	 * takes as addr the inital BOS, and as len, the initial
687 	 * region size.  It can then grow down from addr without
688 	 * limit.  However, linux threads has an implicit internal
689 	 * limit to stack size of STACK_SIZE.  Its just not
690 	 * enforced explicitly in linux.  But, here we impose
691 	 * a limit of (STACK_SIZE - GUARD_SIZE) on the stack
692 	 * region, since we can do this with our mmap.
693 	 *
694 	 * Our mmap with MAP_STACK takes addr as the maximum
695 	 * downsize limit on BOS, and as len the max size of
696 	 * the region.  It them maps the top SGROWSIZ bytes,
697 	 * and autgrows the region down, up to the limit
698 	 * in addr.
699 	 *
700 	 * If we don't use the MAP_STACK option, the effect
701 	 * of this code is to allocate a stack region of a
702 	 * fixed size of (STACK_SIZE - GUARD_SIZE).
703 	 */
704 
705 	/* This gives us TOS */
706 	bsd_args.addr = linux_args.addr + linux_args.len;
707 
708 	/* This gives us our maximum stack size */
709 	if (linux_args.len > STACK_SIZE - GUARD_SIZE)
710 	    bsd_args.len = linux_args.len;
711 	else
712 	    bsd_args.len  = STACK_SIZE - GUARD_SIZE;
713 
714 	/* This gives us a new BOS.  If we're using VM_STACK, then
715 	 * mmap will just map the top SGROWSIZ bytes, and let
716 	 * the stack grow down to the limit at BOS.  If we're
717 	 * not using VM_STACK we map the full stack, since we
718 	 * don't have a way to autogrow it.
719 	 */
720 	bsd_args.addr -= bsd_args.len;
721 
722     } else {
723 	bsd_args.addr = linux_args.addr;
724 	bsd_args.len  = linux_args.len;
725     }
726 
727     bsd_args.prot = linux_args.prot | PROT_READ;	/* always required */
728     bsd_args.fd = linux_args.fd;
729     bsd_args.pos = linux_args.pos;
730     bsd_args.pad = 0;
731     return mmap(p, &bsd_args);
732 }
733 
734 int
735 linux_mremap(struct proc *p, struct linux_mremap_args *args)
736 {
737 	struct munmap_args /* {
738 		void *addr;
739 		size_t len;
740 	} */ bsd_args;
741 	int error = 0;
742 
743 #ifdef DEBUG
744 	printf("Linux-emul(%ld): mremap(%p, %08x, %08x, %08x)\n",
745 	    (long)p->p_pid, (void *)args->addr, args->old_len, args->new_len,
746 	    args->flags);
747 #endif
748 	args->new_len = round_page(args->new_len);
749 	args->old_len = round_page(args->old_len);
750 
751 	if (args->new_len > args->old_len) {
752 		p->p_retval[0] = 0;
753 		return ENOMEM;
754 	}
755 
756 	if (args->new_len < args->old_len) {
757 		bsd_args.addr = args->addr + args->new_len;
758 		bsd_args.len = args->old_len - args->new_len;
759 		error = munmap(p, &bsd_args);
760 	}
761 
762 	p->p_retval[0] = error ? 0 : (int)args->addr;
763 	return error;
764 }
765 
766 int
767 linux_msync(struct proc *p, struct linux_msync_args *args)
768 {
769 	struct msync_args bsd_args;
770 
771 	bsd_args.addr = args->addr;
772 	bsd_args.len = args->len;
773 	bsd_args.flags = 0;	/* XXX ignore */
774 
775 	return msync(p, &bsd_args);
776 }
777 
778 int
779 linux_pipe(struct proc *p, struct linux_pipe_args *args)
780 {
781     int error;
782     int reg_edx;
783 
784 #ifdef DEBUG
785     printf("Linux-emul(%d): pipe(*)\n", p->p_pid);
786 #endif
787     reg_edx = p->p_retval[1];
788     error = pipe(p, 0);
789     if (error) {
790 	p->p_retval[1] = reg_edx;
791 	return error;
792     }
793 
794     error = copyout(p->p_retval, args->pipefds, 2*sizeof(int));
795     if (error) {
796 	p->p_retval[1] = reg_edx;
797 	return error;
798     }
799 
800     p->p_retval[1] = reg_edx;
801     p->p_retval[0] = 0;
802     return 0;
803 }
804 
805 int
806 linux_time(struct proc *p, struct linux_time_args *args)
807 {
808     struct timeval tv;
809     linux_time_t tm;
810     int error;
811 
812 #ifdef DEBUG
813     printf("Linux-emul(%d): time(*)\n", p->p_pid);
814 #endif
815     microtime(&tv);
816     tm = tv.tv_sec;
817     if (args->tm && (error = copyout(&tm, args->tm, sizeof(linux_time_t))))
818 	return error;
819     p->p_retval[0] = tm;
820     return 0;
821 }
822 
823 struct linux_times_argv {
824     long    tms_utime;
825     long    tms_stime;
826     long    tms_cutime;
827     long    tms_cstime;
828 };
829 
830 #define CLK_TCK 100	/* Linux uses 100 */
831 #define CONVTCK(r)	(r.tv_sec * CLK_TCK + r.tv_usec / (1000000 / CLK_TCK))
832 
833 int
834 linux_times(struct proc *p, struct linux_times_args *args)
835 {
836     struct timeval tv;
837     struct linux_times_argv tms;
838     struct rusage ru;
839     int error;
840 
841 #ifdef DEBUG
842     printf("Linux-emul(%d): times(*)\n", p->p_pid);
843 #endif
844     calcru(p, &ru.ru_utime, &ru.ru_stime, NULL);
845 
846     tms.tms_utime = CONVTCK(ru.ru_utime);
847     tms.tms_stime = CONVTCK(ru.ru_stime);
848 
849     tms.tms_cutime = CONVTCK(p->p_stats->p_cru.ru_utime);
850     tms.tms_cstime = CONVTCK(p->p_stats->p_cru.ru_stime);
851 
852     if ((error = copyout((caddr_t)&tms, (caddr_t)args->buf,
853 	    	    sizeof(struct linux_times_argv))))
854 	return error;
855 
856     microuptime(&tv);
857     p->p_retval[0] = (int)CONVTCK(tv);
858     return 0;
859 }
860 
861 /* XXX move */
862 struct linux_newuname_t {
863     char sysname[65];
864     char nodename[65];
865     char release[65];
866     char version[65];
867     char machine[65];
868     char domainname[65];
869 };
870 
871 int
872 linux_newuname(struct proc *p, struct linux_newuname_args *args)
873 {
874     struct linux_newuname_t linux_newuname;
875 
876 #ifdef DEBUG
877     printf("Linux-emul(%d): newuname(*)\n", p->p_pid);
878 #endif
879     bzero(&linux_newuname, sizeof(struct linux_newuname_t));
880     strncpy(linux_newuname.sysname, "Linux",
881 	sizeof(linux_newuname.sysname) - 1);
882     strncpy(linux_newuname.nodename, hostname,
883 	sizeof(linux_newuname.nodename) - 1);
884     strncpy(linux_newuname.release, "2.0.36",
885 	sizeof(linux_newuname.release) - 1);
886     strncpy(linux_newuname.version, version,
887 	sizeof(linux_newuname.version) - 1);
888     strncpy(linux_newuname.machine, machine,
889 	sizeof(linux_newuname.machine) - 1);
890     strncpy(linux_newuname.domainname, domainname,
891 	sizeof(linux_newuname.domainname) - 1);
892     return (copyout((caddr_t)&linux_newuname, (caddr_t)args->buf,
893 	    	    sizeof(struct linux_newuname_t)));
894 }
895 
896 struct linux_utimbuf {
897 	linux_time_t l_actime;
898 	linux_time_t l_modtime;
899 };
900 
901 int
902 linux_utime(struct proc *p, struct linux_utime_args *args)
903 {
904     struct utimes_args /* {
905 	char	*path;
906 	struct	timeval *tptr;
907     } */ bsdutimes;
908     struct timeval tv[2], *tvp;
909     struct linux_utimbuf lut;
910     int error;
911     caddr_t sg;
912 
913     sg = stackgap_init();
914     CHECKALTEXIST(p, &sg, args->fname);
915 
916 #ifdef DEBUG
917     printf("Linux-emul(%d): utime(%s, *)\n", p->p_pid, args->fname);
918 #endif
919     if (args->times) {
920 	if ((error = copyin(args->times, &lut, sizeof lut)))
921 	    return error;
922 	tv[0].tv_sec = lut.l_actime;
923 	tv[0].tv_usec = 0;
924 	tv[1].tv_sec = lut.l_modtime;
925 	tv[1].tv_usec = 0;
926 	/* so that utimes can copyin */
927 	tvp = (struct timeval *)stackgap_alloc(&sg, sizeof(tv));
928 	if ((error = copyout(tv, tvp, sizeof(tv))))
929 	    return error;
930 	bsdutimes.tptr = tvp;
931     } else
932 	bsdutimes.tptr = NULL;
933 
934     bsdutimes.path = args->fname;
935     return utimes(p, &bsdutimes);
936 }
937 
938 #define __WCLONE 0x80000000
939 
940 int
941 linux_waitpid(struct proc *p, struct linux_waitpid_args *args)
942 {
943     struct wait_args /* {
944 	int pid;
945 	int *status;
946 	int options;
947 	struct	rusage *rusage;
948     } */ tmp;
949     int error, tmpstat;
950 
951 #ifdef DEBUG
952     printf("Linux-emul(%ld): waitpid(%d, %p, %d)\n",
953 	(long)p->p_pid, args->pid, (void *)args->status, args->options);
954 #endif
955     tmp.pid = args->pid;
956     tmp.status = args->status;
957     tmp.options = (args->options & (WNOHANG | WUNTRACED));
958     /* WLINUXCLONE should be equal to __WCLONE, but we make sure */
959     if (args->options & __WCLONE)
960 	tmp.options |= WLINUXCLONE;
961     tmp.rusage = NULL;
962 
963     if ((error = wait4(p, &tmp)) != 0)
964 	return error;
965 
966     if (args->status) {
967 	if ((error = copyin(args->status, &tmpstat, sizeof(int))) != 0)
968 	    return error;
969 	if (WIFSIGNALED(tmpstat))
970 	    tmpstat = (tmpstat & 0xffffff80) |
971 		      bsd_to_linux_signal[WTERMSIG(tmpstat)];
972 	else if (WIFSTOPPED(tmpstat))
973 	    tmpstat = (tmpstat & 0xffff00ff) |
974 		      (bsd_to_linux_signal[WSTOPSIG(tmpstat)]<<8);
975 	return copyout(&tmpstat, args->status, sizeof(int));
976     } else
977 	return 0;
978 }
979 
980 int
981 linux_wait4(struct proc *p, struct linux_wait4_args *args)
982 {
983     struct wait_args /* {
984 	int pid;
985 	int *status;
986 	int options;
987 	struct	rusage *rusage;
988     } */ tmp;
989     int error, tmpstat;
990 
991 #ifdef DEBUG
992     printf("Linux-emul(%ld): wait4(%d, %p, %d, %p)\n",
993 	(long)p->p_pid, args->pid, (void *)args->status, args->options,
994 	(void *)args->rusage);
995 #endif
996     tmp.pid = args->pid;
997     tmp.status = args->status;
998     tmp.options = (args->options & (WNOHANG | WUNTRACED));
999     /* WLINUXCLONE should be equal to __WCLONE, but we make sure */
1000     if (args->options & __WCLONE)
1001 	tmp.options |= WLINUXCLONE;
1002     tmp.rusage = args->rusage;
1003 
1004     if ((error = wait4(p, &tmp)) != 0)
1005 	return error;
1006 
1007     p->p_siglist &= ~sigmask(SIGCHLD);
1008 
1009     if (args->status) {
1010 	if ((error = copyin(args->status, &tmpstat, sizeof(int))) != 0)
1011 	    return error;
1012 	if (WIFSIGNALED(tmpstat))
1013 	    tmpstat = (tmpstat & 0xffffff80) |
1014 		  bsd_to_linux_signal[WTERMSIG(tmpstat)];
1015 	else if (WIFSTOPPED(tmpstat))
1016 	    tmpstat = (tmpstat & 0xffff00ff) |
1017 		  (bsd_to_linux_signal[WSTOPSIG(tmpstat)]<<8);
1018 	return copyout(&tmpstat, args->status, sizeof(int));
1019     } else
1020 	return 0;
1021 }
1022 
1023 int
1024 linux_mknod(struct proc *p, struct linux_mknod_args *args)
1025 {
1026 	caddr_t sg;
1027 	struct mknod_args bsd_mknod;
1028 	struct mkfifo_args bsd_mkfifo;
1029 
1030 	sg = stackgap_init();
1031 
1032 	CHECKALTCREAT(p, &sg, args->path);
1033 
1034 #ifdef DEBUG
1035 	printf("Linux-emul(%d): mknod(%s, %d, %d)\n",
1036 	   p->p_pid, args->path, args->mode, args->dev);
1037 #endif
1038 
1039 	if (args->mode & S_IFIFO) {
1040 		bsd_mkfifo.path = args->path;
1041 		bsd_mkfifo.mode = args->mode;
1042 		return mkfifo(p, &bsd_mkfifo);
1043 	} else {
1044 		bsd_mknod.path = args->path;
1045 		bsd_mknod.mode = args->mode;
1046 		bsd_mknod.dev = args->dev;
1047 		return mknod(p, &bsd_mknod);
1048 	}
1049 }
1050 
1051 /*
1052  * UGH! This is just about the dumbest idea I've ever heard!!
1053  */
1054 int
1055 linux_personality(struct proc *p, struct linux_personality_args *args)
1056 {
1057 #ifdef DEBUG
1058 	printf("Linux-emul(%d): personality(%d)\n",
1059 	   p->p_pid, args->per);
1060 #endif
1061 	if (args->per != 0)
1062 		return EINVAL;
1063 
1064 	/* Yes Jim, it's still a Linux... */
1065 	p->p_retval[0] = 0;
1066 	return 0;
1067 }
1068 
1069 /*
1070  * Wrappers for get/setitimer for debugging..
1071  */
1072 int
1073 linux_setitimer(struct proc *p, struct linux_setitimer_args *args)
1074 {
1075 	struct setitimer_args bsa;
1076 	struct itimerval foo;
1077 	int error;
1078 
1079 #ifdef DEBUG
1080 	printf("Linux-emul(%ld): setitimer(%p, %p)\n",
1081 	    (long)p->p_pid, (void *)args->itv, (void *)args->oitv);
1082 #endif
1083 	bsa.which = args->which;
1084 	bsa.itv = args->itv;
1085 	bsa.oitv = args->oitv;
1086 	if (args->itv) {
1087 	    if ((error = copyin((caddr_t)args->itv, (caddr_t)&foo,
1088 			sizeof(foo))))
1089 		return error;
1090 #ifdef DEBUG
1091 	    printf("setitimer: value: sec: %ld, usec: %ld\n",
1092 		foo.it_value.tv_sec, foo.it_value.tv_usec);
1093 	    printf("setitimer: interval: sec: %ld, usec: %ld\n",
1094 		foo.it_interval.tv_sec, foo.it_interval.tv_usec);
1095 #endif
1096 	}
1097 	return setitimer(p, &bsa);
1098 }
1099 
1100 int
1101 linux_getitimer(struct proc *p, struct linux_getitimer_args *args)
1102 {
1103 	struct getitimer_args bsa;
1104 #ifdef DEBUG
1105 	printf("Linux-emul(%ld): getitimer(%p)\n",
1106 	    (long)p->p_pid, (void *)args->itv);
1107 #endif
1108 	bsa.which = args->which;
1109 	bsa.itv = args->itv;
1110 	return getitimer(p, &bsa);
1111 }
1112 
1113 int
1114 linux_iopl(struct proc *p, struct linux_iopl_args *args)
1115 {
1116 	int error;
1117 
1118 	error = suser(p);
1119 	if (error != 0)
1120 		return error;
1121 	if (securelevel > 0)
1122 		return EPERM;
1123 	p->p_md.md_regs->tf_eflags |= PSL_IOPL;
1124 	return 0;
1125 }
1126 
1127 int
1128 linux_nice(struct proc *p, struct linux_nice_args *args)
1129 {
1130 	struct setpriority_args	bsd_args;
1131 
1132 	bsd_args.which = PRIO_PROCESS;
1133 	bsd_args.who = 0;	/* current process */
1134 	bsd_args.prio = args->inc;
1135 	return setpriority(p, &bsd_args);
1136 }
1137 
1138 int
1139 linux_setgroups(p, uap)
1140      struct proc *p;
1141      struct linux_setgroups_args *uap;
1142 {
1143   struct pcred *pc = p->p_cred;
1144   linux_gid_t linux_gidset[NGROUPS];
1145   gid_t *bsd_gidset;
1146   int ngrp, error;
1147 
1148   if ((error = suser(p)))
1149     return error;
1150 
1151   if (uap->gidsetsize > NGROUPS)
1152     return EINVAL;
1153 
1154   ngrp = uap->gidsetsize;
1155   pc->pc_ucred = crcopy(pc->pc_ucred);
1156   if (ngrp >= 1) {
1157     if ((error = copyin((caddr_t)uap->gidset,
1158                       (caddr_t)linux_gidset,
1159                         ngrp * sizeof(linux_gid_t))))
1160       return error;
1161 
1162     pc->pc_ucred->cr_ngroups = ngrp;
1163 
1164     bsd_gidset = pc->pc_ucred->cr_groups;
1165     ngrp--;
1166     while (ngrp >= 0) {
1167       bsd_gidset[ngrp] = linux_gidset[ngrp];
1168       ngrp--;
1169     }
1170   }
1171   else
1172     pc->pc_ucred->cr_ngroups = 1;
1173 
1174   setsugid(p);
1175   return 0;
1176 }
1177 
1178 int
1179 linux_getgroups(p, uap)
1180      struct proc *p;
1181      struct linux_getgroups_args *uap;
1182 {
1183   struct pcred *pc = p->p_cred;
1184   linux_gid_t linux_gidset[NGROUPS];
1185   gid_t *bsd_gidset;
1186   int ngrp, error;
1187 
1188   if ((ngrp = uap->gidsetsize) == 0) {
1189     p->p_retval[0] = pc->pc_ucred->cr_ngroups;
1190     return 0;
1191   }
1192 
1193   if (ngrp < pc->pc_ucred->cr_ngroups)
1194     return EINVAL;
1195 
1196   ngrp = 0;
1197   bsd_gidset = pc->pc_ucred->cr_groups;
1198   while (ngrp < pc->pc_ucred->cr_ngroups) {
1199     linux_gidset[ngrp] = bsd_gidset[ngrp];
1200     ngrp++;
1201   }
1202 
1203   if ((error = copyout((caddr_t)linux_gidset, (caddr_t)uap->gidset,
1204                        ngrp * sizeof(linux_gid_t))))
1205     return error;
1206 
1207   p->p_retval[0] = ngrp;
1208   return (0);
1209 }
1210 
1211 int
1212 linux_setrlimit(p, uap)
1213      struct proc *p;
1214      struct linux_setrlimit_args *uap;
1215 {
1216 #ifdef DEBUG
1217     printf("Linux-emul(%ld): setrlimit(%d, %p)\n",
1218 	   (long)p->p_pid, uap->resource, (void *)uap->rlim);
1219 #endif
1220 
1221     if (uap->resource >= LINUX_RLIM_NLIMITS)
1222 	return EINVAL;
1223 
1224     uap->resource = linux_to_bsd_resource[uap->resource];
1225 
1226     if (uap->resource == -1)
1227 	return EINVAL;
1228 
1229     return osetrlimit(p, uap);
1230 }
1231 
1232 int
1233 linux_getrlimit(p, uap)
1234      struct proc *p;
1235      struct linux_getrlimit_args *uap;
1236 {
1237 #ifdef DEBUG
1238     printf("Linux-emul(%ld): getrlimit(%d, %p)\n",
1239 	   (long)p->p_pid, uap->resource, (void *)uap->rlim);
1240 #endif
1241 
1242     if (uap->resource >= LINUX_RLIM_NLIMITS)
1243 	return EINVAL;
1244 
1245     uap->resource = linux_to_bsd_resource[uap->resource];
1246 
1247     if (uap->resource == -1)
1248 	return EINVAL;
1249 
1250     return ogetrlimit(p, uap);
1251 }
1252