|
|
1.1 root 1: /*
2: * linux/kernel/sched.c
3: *
4: * Copyright (C) 1991, 1992 Linus Torvalds
5: *
6: * 1996-04-21 Modified by Ulrich Windl to make NTP work
7: * 1996-12-23 Modified by Dave Grothe to fix bugs in semaphores and
8: * make semaphores SMP safe
9: * 1997-01-28 Modified by Finn Arne Gangstad to make timers scale better.
10: * 1997-09-10 Updated NTP code according to technical memorandum Jan '96
11: * "A Kernel Model for Precision Timekeeping" by Dave Mills
12: */
13:
14: /*
15: * 'sched.c' is the main kernel file. It contains scheduling primitives
16: * (sleep_on, wakeup, schedule etc) as well as a number of simple system
17: * call functions (type getpid()), which just extract a field from
18: * current-task
19: */
20:
21: #include <linux/signal.h>
22: #include <linux/sched.h>
23: #include <linux/timer.h>
24: #include <linux/kernel.h>
25: #include <linux/kernel_stat.h>
26: #include <linux/fdreg.h>
27: #include <linux/errno.h>
28: #include <linux/time.h>
29: #include <linux/ptrace.h>
30: #include <linux/delay.h>
31: #include <linux/interrupt.h>
32: #include <linux/tqueue.h>
33: #include <linux/resource.h>
34: #include <linux/mm.h>
35: #include <linux/smp.h>
36:
37: #include <asm/system.h>
38: #include <asm/io.h>
39: #include <asm/segment.h>
40: #include <asm/pgtable.h>
41: #include <asm/mmu_context.h>
42:
43: #include <linux/timex.h>
44:
45: /*
46: * kernel variables
47: */
48:
49: int securelevel = 0; /* system security level */
50:
51: long tick = (1000000 + HZ/2) / HZ; /* timer interrupt period */
52: volatile struct timeval xtime; /* The current time */
53: int tickadj = 500/HZ ? 500/HZ : 1; /* microsecs */
54:
55: DECLARE_TASK_QUEUE(tq_timer);
56: DECLARE_TASK_QUEUE(tq_immediate);
57: DECLARE_TASK_QUEUE(tq_scheduler);
58:
59: /*
60: * phase-lock loop variables
61: */
62: /* TIME_ERROR prevents overwriting the CMOS clock */
63: int time_state = TIME_ERROR; /* clock synchronization status */
64: int time_status = STA_UNSYNC; /* clock status bits */
65: long time_offset = 0; /* time adjustment (us) */
66: long time_constant = 2; /* pll time constant */
67: long time_tolerance = MAXFREQ; /* frequency tolerance (ppm) */
68: long time_precision = 1; /* clock precision (us) */
69: long time_maxerror = NTP_PHASE_LIMIT; /* maximum error (us) */
70: long time_esterror = NTP_PHASE_LIMIT; /* estimated error (us) */
71: long time_phase = 0; /* phase offset (scaled us) */
72: long time_freq = ((1000000 + HZ/2) % HZ - HZ/2) << SHIFT_USEC; /* frequency offset (scaled ppm) */
73: long time_adj = 0; /* tick adjust (scaled 1 / HZ) */
74: long time_reftime = 0; /* time at last adjustment (s) */
75:
76: long time_adjust = 0;
77: long time_adjust_step = 0;
78:
79: int need_resched = 0;
80: unsigned long event = 0;
81:
82: extern int _setitimer(int, struct itimerval *, struct itimerval *);
83: unsigned int * prof_buffer = NULL;
84: unsigned long prof_len = 0;
85: unsigned long prof_shift = 0;
86:
87: #define _S(nr) (1<<((nr)-1))
88:
89: extern void mem_use(void);
90: extern unsigned long get_wchan(struct task_struct *);
91:
92: static unsigned long init_kernel_stack[1024] = { STACK_MAGIC, };
93: unsigned long init_user_stack[1024] = { STACK_MAGIC, };
94: static struct vm_area_struct init_mmap = INIT_MMAP;
95: static struct fs_struct init_fs = INIT_FS;
96: static struct files_struct init_files = INIT_FILES;
97: static struct signal_struct init_signals = INIT_SIGNALS;
98:
99: struct mm_struct init_mm = INIT_MM;
100: struct task_struct init_task = INIT_TASK;
101:
102: unsigned long volatile jiffies=0;
103:
104: struct task_struct *current_set[NR_CPUS];
105: struct task_struct *last_task_used_math = NULL;
106:
107: struct task_struct * task[NR_TASKS] = {&init_task, };
108:
109: struct kernel_stat kstat = { 0 };
110:
111: static inline void add_to_runqueue(struct task_struct * p)
112: {
113: #ifdef __SMP__
114: int cpu=smp_processor_id();
115: #endif
116: #if 1 /* sanity tests */
117: if (p->next_run || p->prev_run) {
118: printk("task already on run-queue\n");
119: return;
120: }
121: #endif
122: if (p->policy != SCHED_OTHER || p->counter > current->counter + 3)
123: need_resched = 1;
124: nr_running++;
125: (p->prev_run = init_task.prev_run)->next_run = p;
126: p->next_run = &init_task;
127: init_task.prev_run = p;
128: #ifdef __SMP__
129: /* this is safe only if called with cli()*/
130: while(set_bit(31,&smp_process_available))
131: {
132: while(test_bit(31,&smp_process_available))
133: {
134: if(clear_bit(cpu,&smp_invalidate_needed))
135: {
136: local_flush_tlb();
137: set_bit(cpu,&cpu_callin_map[0]);
138: }
139: }
140: }
141: smp_process_available++;
142: clear_bit(31,&smp_process_available);
143: if ((0!=p->pid) && smp_threads_ready)
144: {
145: int i;
146: for (i=0;i<smp_num_cpus;i++)
147: {
148: if (0==current_set[cpu_logical_map[i]]->pid)
149: {
150: smp_message_pass(cpu_logical_map[i], MSG_RESCHEDULE, 0L, 0);
151: break;
152: }
153: }
154: }
155: #endif
156: }
157:
158: static inline void del_from_runqueue(struct task_struct * p)
159: {
160: struct task_struct *next = p->next_run;
161: struct task_struct *prev = p->prev_run;
162:
163: #if 1 /* sanity tests */
164: if (!next || !prev) {
165: printk("task not on run-queue\n");
166: return;
167: }
168: #endif
169: if (p == &init_task) {
170: static int nr = 0;
171: if (nr < 5) {
172: nr++;
173: printk("idle task may not sleep\n");
174: }
175: return;
176: }
177: nr_running--;
178: next->prev_run = prev;
179: prev->next_run = next;
180: p->next_run = NULL;
181: p->prev_run = NULL;
182: }
183:
184: static inline void move_last_runqueue(struct task_struct * p)
185: {
186: struct task_struct *next = p->next_run;
187: struct task_struct *prev = p->prev_run;
188:
189: /* remove from list */
190: next->prev_run = prev;
191: prev->next_run = next;
192: /* add back to list */
193: p->next_run = &init_task;
194: prev = init_task.prev_run;
195: init_task.prev_run = p;
196: p->prev_run = prev;
197: prev->next_run = p;
198: }
199:
200: /*
201: * Wake up a process. Put it on the run-queue if it's not
202: * already there. The "current" process is always on the
203: * run-queue (except when the actual re-schedule is in
204: * progress), and as such you're allowed to do the simpler
205: * "current->state = TASK_RUNNING" to mark yourself runnable
206: * without the overhead of this.
207: */
208: inline void wake_up_process(struct task_struct * p)
209: {
210: unsigned long flags;
211:
212: save_flags(flags);
213: cli();
214: p->state = TASK_RUNNING;
215: if (!p->next_run)
216: add_to_runqueue(p);
217: restore_flags(flags);
218: }
219:
220: static void process_timeout(unsigned long __data)
221: {
222: struct task_struct * p = (struct task_struct *) __data;
223:
224: p->timeout = 0;
225: wake_up_process(p);
226: }
227:
228: /*
229: * This is the function that decides how desirable a process is..
230: * You can weigh different processes against each other depending
231: * on what CPU they've run on lately etc to try to handle cache
232: * and TLB miss penalties.
233: *
234: * Return values:
235: * -1000: never select this
236: * 0: out of time, recalculate counters (but it might still be
237: * selected)
238: * +ve: "goodness" value (the larger, the better)
239: * +1000: realtime process, select this.
240: */
241: static inline int goodness(struct task_struct * p, struct task_struct * prev, int this_cpu)
242: {
243: int weight;
244:
245: #ifdef __SMP__
246: /* We are not permitted to run a task someone else is running */
247: if (p->processor != NO_PROC_ID)
248: return -1000;
249: #ifdef PAST_2_0
250: /* This process is locked to a processor group */
251: if (p->processor_mask && !(p->processor_mask & (1<<this_cpu))
252: return -1000;
253: #endif
254: #endif
255:
256: /*
257: * Realtime process, select the first one on the
258: * runqueue (taking priorities within processes
259: * into account).
260: */
261: if (p->policy != SCHED_OTHER)
262: return 1000 + p->rt_priority;
263:
264: /*
265: * Give the process a first-approximation goodness value
266: * according to the number of clock-ticks it has left.
267: *
268: * Don't do any other calculations if the time slice is
269: * over..
270: */
271: weight = p->counter;
272: if (weight) {
273:
274: #ifdef __SMP__
275: /* Give a largish advantage to the same processor... */
276: /* (this is equivalent to penalizing other processors) */
277: if (p->last_processor == this_cpu)
278: weight += PROC_CHANGE_PENALTY;
279: #endif
280:
281: /* .. and a slight advantage to the current process */
282: if (p == prev)
283: weight += 1;
284: }
285:
286: return weight;
287: }
288:
289:
290: /*
291: The following allow_interrupts function is used to workaround a rare but
292: nasty deadlock situation that is possible for 2.0.x Intel SMP because it uses
293: a single kernel lock and interrupts are only routed to the boot CPU. There
294: are two deadlock scenarios this code protects against.
295:
296: The first scenario is that if a CPU other than the boot CPU holds the kernel
297: lock and needs to wait for an operation to complete that itself requires an
298: interrupt, there is a deadlock since the boot CPU may be able to accept the
299: interrupt but will not be able to acquire the kernel lock to process it.
300:
301: The workaround for this deadlock requires adding calls to allow_interrupts to
302: places where this deadlock is possible. These places are known to be present
303: in buffer.c and keyboard.c. It is also possible that there are other such
304: places which have not been identified yet. In order to break the deadlock,
305: the code in allow_interrupts temporarily yields the kernel lock directly to
306: the boot CPU to allow the interrupt to be processed. The boot CPU interrupt
307: entry code indicates that it is spinning waiting for the kernel lock by
308: setting the smp_blocked_interrupt_pending variable. This code notices that
309: and manipulates the active_kernel_processor variable to yield the kernel lock
310: without ever clearing it. When the interrupt has been processed, the
311: saved_active_kernel_processor variable contains the value for the interrupt
312: exit code to restore, either the APICID of the CPU that granted it the kernel
313: lock, or NO_PROC_ID in the normal case where no yielding occurred. Restoring
314: active_kernel_processor from saved_active_kernel_processor returns the kernel
315: lock back to the CPU that yielded it.
316:
317: The second form of deadlock is even more insidious. Suppose the boot CPU
318: takes a page fault and then the previous scenario ensues. In this case, the
319: boot CPU would spin with interrupts disabled waiting to acquire the kernel
320: lock. To resolve this deadlock, the kernel lock acquisition code must enable
321: interrupts briefly so that the pending interrupt can be handled as in the
322: case above.
323:
324: An additional form of deadlock is where kernel code running on a non-boot CPU
325: waits for the jiffies variable to be incremented. This deadlock is avoided
326: by having the spin loops in ENTER_KERNEL increment jiffies approximately
327: every 10 milliseconds. Finally, if approximately 60 seconds elapse waiting
328: for the kernel lock, a message will be printed if possible to indicate that a
329: deadlock has been detected.
330:
331: Leonard N. Zubkoff
332: 4 August 1997
333: */
334:
335: #if defined(__SMP__) && defined(__i386__)
336:
337: volatile unsigned char smp_blocked_interrupt_pending = 0;
338:
339: volatile unsigned char saved_active_kernel_processor = NO_PROC_ID;
340:
341: void allow_interrupts(void)
342: {
343: if (smp_processor_id() == boot_cpu_id) return;
344: if (smp_blocked_interrupt_pending)
345: {
346: unsigned long saved_kernel_counter;
347: long timeout_counter;
348: saved_active_kernel_processor = active_kernel_processor;
349: saved_kernel_counter = kernel_counter;
350: kernel_counter = 0;
351: active_kernel_processor = boot_cpu_id;
352: timeout_counter = 6000000;
353: while (active_kernel_processor != saved_active_kernel_processor &&
354: --timeout_counter >= 0)
355: {
356: udelay(10);
357: barrier();
358: }
359: if (timeout_counter < 0)
360: panic("FORWARDED INTERRUPT TIMEOUT (AKP = %d, Saved AKP = %d)\n",
361: active_kernel_processor, saved_active_kernel_processor);
362: kernel_counter = saved_kernel_counter;
363: saved_active_kernel_processor = NO_PROC_ID;
364: }
365: }
366:
367: #else
368:
369: void allow_interrupts(void) {}
370:
371: #endif
372:
373:
374: /*
375: * 'schedule()' is the scheduler function. It's a very simple and nice
376: * scheduler: it's not perfect, but certainly works for most things.
377: *
378: * The goto is "interesting".
379: *
380: * NOTE!! Task 0 is the 'idle' task, which gets called when no other
381: * tasks can run. It can not be killed, and it cannot sleep. The 'state'
382: * information in task[0] is never used.
383: */
384: asmlinkage void schedule(void)
385: {
386: int c;
387: struct task_struct * p;
388: struct task_struct * prev, * next;
389: unsigned long timeout = 0;
390: int this_cpu=smp_processor_id();
391:
392: /* check alarm, wake up any interruptible tasks that have got a signal */
393:
394: allow_interrupts();
395:
396: if (intr_count)
397: goto scheduling_in_interrupt;
398:
399: if (bh_active & bh_mask) {
400: intr_count = 1;
401: do_bottom_half();
402: intr_count = 0;
403: }
404:
405: run_task_queue(&tq_scheduler);
406:
407: need_resched = 0;
408: prev = current;
409: cli();
410: /* move an exhausted RR process to be last.. */
411: if (!prev->counter && prev->policy == SCHED_RR) {
412: prev->counter = prev->priority;
413: move_last_runqueue(prev);
414: }
415: switch (prev->state) {
416: case TASK_INTERRUPTIBLE:
417: if (prev->signal & ~prev->blocked)
418: goto makerunnable;
419: timeout = prev->timeout;
420: if (timeout && (timeout <= jiffies)) {
421: prev->timeout = 0;
422: timeout = 0;
423: makerunnable:
424: prev->state = TASK_RUNNING;
425: break;
426: }
427: default:
428: del_from_runqueue(prev);
429: case TASK_RUNNING:
430: }
431: p = init_task.next_run;
432: sti();
433:
434: #ifdef __SMP__
435: /*
436: * This is safe as we do not permit re-entry of schedule()
437: */
438: prev->processor = NO_PROC_ID;
439: #define idle_task (task[cpu_number_map[this_cpu]])
440: #else
441: #define idle_task (&init_task)
442: #endif
443:
444: /*
445: * Note! there may appear new tasks on the run-queue during this, as
446: * interrupts are enabled. However, they will be put on front of the
447: * list, so our list starting at "p" is essentially fixed.
448: */
449: /* this is the scheduler proper: */
450: c = -1000;
451: next = idle_task;
452: while (p != &init_task) {
453: int weight = goodness(p, prev, this_cpu);
454: if (weight > c)
455: c = weight, next = p;
456: p = p->next_run;
457: }
458:
459: /* if all runnable processes have "counter == 0", re-calculate counters */
460: if (!c) {
461: for_each_task(p)
462: p->counter = (p->counter >> 1) + p->priority;
463: }
464: #ifdef __SMP__
465: /*
466: * Allocate process to CPU
467: */
468:
469: next->processor = this_cpu;
470: next->last_processor = this_cpu;
471: #endif
472: #ifdef __SMP_PROF__
473: /* mark processor running an idle thread */
474: if (0==next->pid)
475: set_bit(this_cpu,&smp_idle_map);
476: else
477: clear_bit(this_cpu,&smp_idle_map);
478: #endif
479: if (prev != next) {
480: struct timer_list timer;
481:
482: kstat.context_swtch++;
483: if (timeout) {
484: init_timer(&timer);
485: timer.expires = timeout;
486: timer.data = (unsigned long) prev;
487: timer.function = process_timeout;
488: add_timer(&timer);
489: }
490: get_mmu_context(next);
491: switch_to(prev,next);
492: if (timeout)
493: del_timer(&timer);
494: }
495: return;
496:
497: scheduling_in_interrupt:
498: printk("Aiee: scheduling in interrupt %p\n",
499: __builtin_return_address(0));
500: }
501:
502: #ifndef __alpha__
503:
504: /*
505: * For backwards compatibility? This can be done in libc so Alpha
506: * and all newer ports shouldn't need it.
507: */
508: asmlinkage int sys_pause(void)
509: {
510: current->state = TASK_INTERRUPTIBLE;
511: schedule();
512: return -ERESTARTNOHAND;
513: }
514:
515: #endif
516:
517: /*
518: * wake_up doesn't wake up stopped processes - they have to be awakened
519: * with signals or similar.
520: *
521: * Note that this doesn't need cli-sti pairs: interrupts may not change
522: * the wait-queue structures directly, but only call wake_up() to wake
523: * a process. The process itself must remove the queue once it has woken.
524: */
525: void wake_up(struct wait_queue **q)
526: {
527: struct wait_queue *next;
528: struct wait_queue *head;
529:
530: if (!q || !(next = *q))
531: return;
532: head = WAIT_QUEUE_HEAD(q);
533: while (next != head) {
534: struct task_struct *p = next->task;
535: next = next->next;
536: if (p != NULL) {
537: if ((p->state == TASK_UNINTERRUPTIBLE) ||
538: (p->state == TASK_INTERRUPTIBLE))
539: wake_up_process(p);
540: }
541: if (!next)
542: goto bad;
543: }
544: return;
545: bad:
546: printk("wait_queue is bad (eip = %p)\n",
547: __builtin_return_address(0));
548: printk(" q = %p\n",q);
549: printk(" *q = %p\n",*q);
550: }
551:
552: void wake_up_interruptible(struct wait_queue **q)
553: {
554: struct wait_queue *next;
555: struct wait_queue *head;
556:
557: if (!q || !(next = *q))
558: return;
559: head = WAIT_QUEUE_HEAD(q);
560: while (next != head) {
561: struct task_struct *p = next->task;
562: next = next->next;
563: if (p != NULL) {
564: if (p->state == TASK_INTERRUPTIBLE)
565: wake_up_process(p);
566: }
567: if (!next)
568: goto bad;
569: }
570: return;
571: bad:
572: printk("wait_queue is bad (eip = %p)\n",
573: __builtin_return_address(0));
574: printk(" q = %p\n",q);
575: printk(" *q = %p\n",*q);
576: }
577:
578:
579: /*
580: * Semaphores are implemented using a two-way counter:
581: * The "count" variable is decremented for each process
582: * that tries to sleep, while the "waking" variable is
583: * incremented when the "up()" code goes to wake up waiting
584: * processes.
585: *
586: * Notably, the inline "up()" and "down()" functions can
587: * efficiently test if they need to do any extra work (up
588: * needs to do something only if count was negative before
589: * the increment operation.
590: *
591: * This routine must execute atomically.
592: */
593: static inline int waking_non_zero(struct semaphore *sem)
594: {
595: int ret ;
596: long flags ;
597:
598: get_buzz_lock(&sem->lock) ;
599: save_flags(flags) ;
600: cli() ;
601:
602: if ((ret = (sem->waking > 0)))
603: sem->waking-- ;
604:
605: restore_flags(flags) ;
606: give_buzz_lock(&sem->lock) ;
607: return(ret) ;
608: }
609:
610: /*
611: * When __up() is called, the count was negative before
612: * incrementing it, and we need to wake up somebody.
613: *
614: * This routine adds one to the count of processes that need to
615: * wake up and exit. ALL waiting processes actually wake up but
616: * only the one that gets to the "waking" field first will gate
617: * through and acquire the semaphore. The others will go back
618: * to sleep.
619: *
620: * Note that these functions are only called when there is
621: * contention on the lock, and as such all this is the
622: * "non-critical" part of the whole semaphore business. The
623: * critical part is the inline stuff in <asm/semaphore.h>
624: * where we want to avoid any extra jumps and calls.
625: */
626: void __up(struct semaphore *sem)
627: {
628: atomic_inc(&sem->waking) ;
629: wake_up(&sem->wait);
630: }
631:
632: /*
633: * Perform the "down" function. Return zero for semaphore acquired,
634: * return negative for signalled out of the function.
635: *
636: * If called from __down, the return is ignored and the wait loop is
637: * not interruptible. This means that a task waiting on a semaphore
638: * using "down()" cannot be killed until someone does an "up()" on
639: * the semaphore.
640: *
641: * If called from __down_interruptible, the return value gets checked
642: * upon return. If the return value is negative then the task continues
643: * with the negative value in the return register (it can be tested by
644: * the caller).
645: *
646: * Either form may be used in conjunction with "up()".
647: *
648: */
649: int __do_down(struct semaphore * sem, int task_state)
650: {
651: struct task_struct *tsk = current;
652: struct wait_queue wait = { tsk, NULL };
653: int ret = 0 ;
654:
655: tsk->state = task_state;
656: add_wait_queue(&sem->wait, &wait);
657:
658: /*
659: * Ok, we're set up. sem->count is known to be less than zero
660: * so we must wait.
661: *
662: * We can let go the lock for purposes of waiting.
663: * We re-acquire it after awaking so as to protect
664: * all semaphore operations.
665: *
666: * If "up()" is called before we call waking_non_zero() then
667: * we will catch it right away. If it is called later then
668: * we will have to go through a wakeup cycle to catch it.
669: *
670: * Multiple waiters contend for the semaphore lock to see
671: * who gets to gate through and who has to wait some more.
672: */
673: for (;;)
674: {
675: if (waking_non_zero(sem)) /* are we waking up? */
676: break ; /* yes, exit loop */
677:
678: if ( task_state == TASK_INTERRUPTIBLE
679: && (tsk->signal & ~tsk->blocked) /* signalled */
680: )
681: {
682: ret = -EINTR ; /* interrupted */
683: atomic_inc(&sem->count) ; /* give up on down operation */
684: break ;
685: }
686:
687: schedule();
688: tsk->state = task_state;
689: }
690:
691: tsk->state = TASK_RUNNING;
692: remove_wait_queue(&sem->wait, &wait);
693: return(ret) ;
694:
695: } /* __do_down */
696:
697: void __down(struct semaphore * sem)
698: {
699: __do_down(sem,TASK_UNINTERRUPTIBLE) ;
700: }
701:
702: int __down_interruptible(struct semaphore * sem)
703: {
704: return(__do_down(sem,TASK_INTERRUPTIBLE)) ;
705: }
706:
707:
708: static inline void __sleep_on(struct wait_queue **p, int state)
709: {
710: unsigned long flags;
711: struct wait_queue wait = { current, NULL };
712:
713: if (!p)
714: return;
715: if (current == task[0])
716: panic("task[0] trying to sleep");
717: current->state = state;
718: save_flags(flags);
719: cli();
720: __add_wait_queue(p, &wait);
721: sti();
722: schedule();
723: cli();
724: __remove_wait_queue(p, &wait);
725: restore_flags(flags);
726: }
727:
728: void interruptible_sleep_on(struct wait_queue **p)
729: {
730: __sleep_on(p,TASK_INTERRUPTIBLE);
731: }
732:
733: void sleep_on(struct wait_queue **p)
734: {
735: __sleep_on(p,TASK_UNINTERRUPTIBLE);
736: }
737:
738: #define TVN_BITS 6
739: #define TVR_BITS 8
740: #define TVN_SIZE (1 << TVN_BITS)
741: #define TVR_SIZE (1 << TVR_BITS)
742: #define TVN_MASK (TVN_SIZE - 1)
743: #define TVR_MASK (TVR_SIZE - 1)
744:
745: #define SLOW_BUT_DEBUGGING_TIMERS 0
746:
747: struct timer_vec {
748: int index;
749: struct timer_list *vec[TVN_SIZE];
750: };
751:
752: struct timer_vec_root {
753: int index;
754: struct timer_list *vec[TVR_SIZE];
755: };
756:
757: static struct timer_vec tv5 = { 0 };
758: static struct timer_vec tv4 = { 0 };
759: static struct timer_vec tv3 = { 0 };
760: static struct timer_vec tv2 = { 0 };
761: static struct timer_vec_root tv1 = { 0 };
762:
763: static struct timer_vec * const tvecs[] = {
764: (struct timer_vec *)&tv1, &tv2, &tv3, &tv4, &tv5
765: };
766:
767: #define NOOF_TVECS (sizeof(tvecs) / sizeof(tvecs[0]))
768:
769: static unsigned long timer_jiffies = 0;
770:
771: static inline void insert_timer(struct timer_list *timer,
772: struct timer_list **vec, int idx)
773: {
774: if ((timer->next = vec[idx]))
775: vec[idx]->prev = timer;
776: vec[idx] = timer;
777: timer->prev = (struct timer_list *)&vec[idx];
778: }
779:
780: static inline void internal_add_timer(struct timer_list *timer)
781: {
782: /*
783: * must be cli-ed when calling this
784: */
785: unsigned long expires = timer->expires;
786: unsigned long idx = expires - timer_jiffies;
787:
788: if (idx < TVR_SIZE) {
789: int i = expires & TVR_MASK;
790: insert_timer(timer, tv1.vec, i);
791: } else if (idx < 1 << (TVR_BITS + TVN_BITS)) {
792: int i = (expires >> TVR_BITS) & TVN_MASK;
793: insert_timer(timer, tv2.vec, i);
794: } else if (idx < 1 << (TVR_BITS + 2 * TVN_BITS)) {
795: int i = (expires >> (TVR_BITS + TVN_BITS)) & TVN_MASK;
796: insert_timer(timer, tv3.vec, i);
797: } else if (idx < 1 << (TVR_BITS + 3 * TVN_BITS)) {
798: int i = (expires >> (TVR_BITS + 2 * TVN_BITS)) & TVN_MASK;
799: insert_timer(timer, tv4.vec, i);
800: } else if (expires < timer_jiffies) {
801: /* can happen if you add a timer with expires == jiffies,
802: * or you set a timer to go off in the past
803: */
804: insert_timer(timer, tv1.vec, tv1.index);
805: } else if (idx < 0xffffffffUL) {
806: int i = (expires >> (TVR_BITS + 3 * TVN_BITS)) & TVN_MASK;
807: insert_timer(timer, tv5.vec, i);
808: } else {
809: /* Can only get here on architectures with 64-bit jiffies */
810: timer->next = timer->prev = timer;
811: }
812: }
813:
814: void add_timer(struct timer_list *timer)
815: {
816: unsigned long flags;
817: save_flags(flags);
818: cli();
819: #if SLOW_BUT_DEBUGGING_TIMERS
820: if (timer->next || timer->prev) {
821: printk("add_timer() called with non-zero list from %p\n",
822: __builtin_return_address(0));
823: goto out;
824: }
825: #endif
826: internal_add_timer(timer);
827: #if SLOW_BUT_DEBUGGING_TIMERS
828: out:
829: #endif
830: restore_flags(flags);
831: }
832:
833: static inline int detach_timer(struct timer_list *timer)
834: {
835: int ret = 0;
836: struct timer_list *next, *prev;
837: next = timer->next;
838: prev = timer->prev;
839: if (next) {
840: next->prev = prev;
841: }
842: if (prev) {
843: ret = 1;
844: prev->next = next;
845: }
846: return ret;
847: }
848:
849:
850: int del_timer(struct timer_list * timer)
851: {
852: int ret;
853: unsigned long flags;
854: save_flags(flags);
855: cli();
856: ret = detach_timer(timer);
857: timer->next = timer->prev = 0;
858: restore_flags(flags);
859: return ret;
860: }
861:
862: static inline void cascade_timers(struct timer_vec *tv)
863: {
864: /* cascade all the timers from tv up one level */
865: struct timer_list *timer;
866: timer = tv->vec[tv->index];
867: /*
868: * We are removing _all_ timers from the list, so we don't have to
869: * detach them individually, just clear the list afterwards.
870: */
871: while (timer) {
872: struct timer_list *tmp = timer;
873: timer = timer->next;
874: internal_add_timer(tmp);
875: }
876: tv->vec[tv->index] = NULL;
877: tv->index = (tv->index + 1) & TVN_MASK;
878: }
879:
880: static inline void run_timer_list(void)
881: {
882: cli();
883: while ((long)(jiffies - timer_jiffies) >= 0) {
884: struct timer_list *timer;
885: if (!tv1.index) {
886: int n = 1;
887: do {
888: cascade_timers(tvecs[n]);
889: } while (tvecs[n]->index == 1 && ++n < NOOF_TVECS);
890: }
891: while ((timer = tv1.vec[tv1.index])) {
892: void (*fn)(unsigned long) = timer->function;
893: unsigned long data = timer->data;
894: detach_timer(timer);
895: timer->next = timer->prev = NULL;
896: sti();
897: fn(data);
898: cli();
899: }
900: ++timer_jiffies;
901: tv1.index = (tv1.index + 1) & TVR_MASK;
902: }
903: sti();
904: }
905:
906: static inline void run_old_timers(void)
907: {
908: struct timer_struct *tp;
909: unsigned long mask;
910:
911: for (mask = 1, tp = timer_table+0 ; mask ; tp++,mask += mask) {
912: if (mask > timer_active)
913: break;
914: if (!(mask & timer_active))
915: continue;
916: if (tp->expires > jiffies)
917: continue;
918: timer_active &= ~mask;
919: tp->fn();
920: sti();
921: }
922: }
923:
924: void tqueue_bh(void)
925: {
926: run_task_queue(&tq_timer);
927: }
928:
929: void immediate_bh(void)
930: {
931: run_task_queue(&tq_immediate);
932: }
933:
934: unsigned long timer_active = 0;
935: struct timer_struct timer_table[32];
936:
937: /*
938: * Hmm.. Changed this, as the GNU make sources (load.c) seems to
939: * imply that avenrun[] is the standard name for this kind of thing.
940: * Nothing else seems to be standardized: the fractional size etc
941: * all seem to differ on different machines.
942: */
943: unsigned long avenrun[3] = { 0,0,0 };
944:
945: /*
946: * Nr of active tasks - counted in fixed-point numbers
947: */
948: static unsigned long count_active_tasks(void)
949: {
950: struct task_struct **p;
951: unsigned long nr = 0;
952:
953: for(p = &LAST_TASK; p > &FIRST_TASK; --p)
954: if (*p && ((*p)->state == TASK_RUNNING ||
955: (*p)->state == TASK_UNINTERRUPTIBLE ||
956: (*p)->state == TASK_SWAPPING))
957: nr += FIXED_1;
958: #ifdef __SMP__
959: nr-=(smp_num_cpus-1)*FIXED_1;
960: #endif
961: return nr;
962: }
963:
964: static inline void calc_load(unsigned long ticks)
965: {
966: unsigned long active_tasks; /* fixed-point */
967: static int count = LOAD_FREQ;
968:
969: count -= ticks;
970: if (count < 0) {
971: count += LOAD_FREQ;
972: active_tasks = count_active_tasks();
973: CALC_LOAD(avenrun[0], EXP_1, active_tasks);
974: CALC_LOAD(avenrun[1], EXP_5, active_tasks);
975: CALC_LOAD(avenrun[2], EXP_15, active_tasks);
976: }
977: }
978:
979: /*
980: * this routine handles the overflow of the microsecond field
981: *
982: * The tricky bits of code to handle the accurate clock support
983: * were provided by Dave Mills ([email protected]) of NTP fame.
984: * They were originally developed for SUN and DEC kernels.
985: * All the kudos should go to Dave for this stuff.
986: *
987: */
988: static void second_overflow(void)
989: {
990: long ltemp;
991:
992: /* Bump the maxerror field */
993: time_maxerror += time_tolerance >> SHIFT_USEC;
994: if ( time_maxerror > NTP_PHASE_LIMIT ) {
995: time_maxerror = NTP_PHASE_LIMIT;
996: time_state = TIME_ERROR; /* p. 17, sect. 4.3, (b) */
997: time_status |= STA_UNSYNC;
998: }
999:
1000: /*
1001: * Leap second processing. If in leap-insert state at
1002: * the end of the day, the system clock is set back one
1003: * second; if in leap-delete state, the system clock is
1004: * set ahead one second. The microtime() routine or
1005: * external clock driver will insure that reported time
1006: * is always monotonic. The ugly divides should be
1007: * replaced.
1008: */
1009: switch (time_state) {
1010:
1011: case TIME_OK:
1012: if (time_status & STA_INS)
1013: time_state = TIME_INS;
1014: else if (time_status & STA_DEL)
1015: time_state = TIME_DEL;
1016: break;
1017:
1018: case TIME_INS:
1019: if (xtime.tv_sec % 86400 == 0) {
1020: xtime.tv_sec--;
1021: time_state = TIME_OOP;
1022: printk(KERN_NOTICE "Clock: inserting leap second 23:59:60 UTC\n");
1023: }
1024: break;
1025:
1026: case TIME_DEL:
1027: if ((xtime.tv_sec + 1) % 86400 == 0) {
1028: xtime.tv_sec++;
1029: time_state = TIME_WAIT;
1030: printk(KERN_NOTICE "Clock: deleting leap second 23:59:59 UTC\n");
1031: }
1032: break;
1033:
1034: case TIME_OOP:
1035: time_state = TIME_WAIT;
1036: break;
1037:
1038: case TIME_WAIT:
1039: if (!(time_status & (STA_INS | STA_DEL)))
1040: time_state = TIME_OK;
1041: }
1042:
1043: /*
1044: * Compute the phase adjustment for the next second. In
1045: * PLL mode, the offset is reduced by a fixed factor
1046: * times the time constant. In FLL mode the offset is
1047: * used directly. In either mode, the maximum phase
1048: * adjustment for each second is clamped so as to spread
1049: * the adjustment over not more than the number of
1050: * seconds between updates.
1051: */
1052: if (time_offset < 0) {
1053: ltemp = -time_offset;
1054: if (!(time_status & STA_FLL))
1055: ltemp >>= SHIFT_KG + time_constant;
1056: if (ltemp > (MAXPHASE / MINSEC) << SHIFT_UPDATE)
1057: ltemp = (MAXPHASE / MINSEC) << SHIFT_UPDATE;
1058: time_offset += ltemp;
1059: time_adj = -ltemp << (SHIFT_SCALE - SHIFT_HZ - SHIFT_UPDATE);
1060: } else {
1061: ltemp = time_offset;
1062: if (!(time_status & STA_FLL))
1063: ltemp >>= SHIFT_KG + time_constant;
1064: if (ltemp > (MAXPHASE / MINSEC) << SHIFT_UPDATE)
1065: ltemp = (MAXPHASE / MINSEC) << SHIFT_UPDATE;
1066: time_offset -= ltemp;
1067: time_adj = ltemp << (SHIFT_SCALE - SHIFT_HZ - SHIFT_UPDATE);
1068: }
1069:
1070: /*
1071: * Compute the frequency estimate and additional phase
1072: * adjustment due to frequency error for the next
1073: * second. When the PPS signal is engaged, gnaw on the
1074: * watchdog counter and update the frequency computed by
1075: * the pll and the PPS signal.
1076: */
1077: pps_valid++;
1078: if (pps_valid == PPS_VALID) { /* PPS signal lost */
1079: pps_jitter = MAXTIME;
1080: pps_stabil = MAXFREQ;
1081: time_status &= ~(STA_PPSSIGNAL | STA_PPSJITTER |
1082: STA_PPSWANDER | STA_PPSERROR);
1083: }
1084: ltemp = time_freq + pps_freq;
1085: if (ltemp < 0)
1086: time_adj -= -ltemp >> (SHIFT_USEC + SHIFT_HZ - SHIFT_SCALE);
1087: else
1088: time_adj += ltemp >> (SHIFT_USEC + SHIFT_HZ - SHIFT_SCALE);
1089:
1090: #if HZ == 100
1091: /* Compensate for (HZ==100) != (1 << SHIFT_HZ).
1092: * Add 25% and 3.125% to get 128.125; => only 0.125% error (p. 14)
1093: */
1094: if (time_adj < 0)
1095: time_adj -= (-time_adj >> 2) + (-time_adj >> 5);
1096: else
1097: time_adj += (time_adj >> 2) + (time_adj >> 5);
1098: #endif
1099: }
1100:
1101: /* in the NTP reference this is called "hardclock()" */
1102: static void update_wall_time_one_tick(void)
1103: {
1104: if ( (time_adjust_step = time_adjust) != 0 ) {
1105: /* We are doing an adjtime thing.
1106: *
1107: * Prepare time_adjust_step to be within bounds.
1108: * Note that a positive time_adjust means we want the clock
1109: * to run faster.
1110: *
1111: * Limit the amount of the step to be in the range
1112: * -tickadj .. +tickadj
1113: */
1114: if (time_adjust > tickadj)
1115: time_adjust_step = tickadj;
1116: else if (time_adjust < -tickadj)
1117: time_adjust_step = -tickadj;
1118:
1119: /* Reduce by this step the amount of time left */
1120: time_adjust -= time_adjust_step;
1121: }
1122: xtime.tv_usec += tick + time_adjust_step;
1123: /*
1124: * Advance the phase, once it gets to one microsecond, then
1125: * advance the tick more.
1126: */
1127: time_phase += time_adj;
1128: if (time_phase <= -FINEUSEC) {
1129: long ltemp = -time_phase >> SHIFT_SCALE;
1130: time_phase += ltemp << SHIFT_SCALE;
1131: xtime.tv_usec -= ltemp;
1132: }
1133: else if (time_phase >= FINEUSEC) {
1134: long ltemp = time_phase >> SHIFT_SCALE;
1135: time_phase -= ltemp << SHIFT_SCALE;
1136: xtime.tv_usec += ltemp;
1137: }
1138: }
1139:
1140: /*
1141: * Using a loop looks inefficient, but "ticks" is
1142: * usually just one (we shouldn't be losing ticks,
1143: * we're doing this this way mainly for interrupt
1144: * latency reasons, not because we think we'll
1145: * have lots of lost timer ticks
1146: */
1147: static void update_wall_time(unsigned long ticks)
1148: {
1149: do {
1150: ticks--;
1151: update_wall_time_one_tick();
1152: } while (ticks);
1153:
1154: if (xtime.tv_usec >= 1000000) {
1155: xtime.tv_usec -= 1000000;
1156: xtime.tv_sec++;
1157: second_overflow();
1158: }
1159: }
1160:
1161: static inline void do_process_times(struct task_struct *p,
1162: unsigned long user, unsigned long system)
1163: {
1164: long psecs;
1165:
1166: p->utime += user;
1167: p->stime += system;
1168:
1169: psecs = (p->stime + p->utime) / HZ;
1170: if (psecs > p->rlim[RLIMIT_CPU].rlim_cur) {
1171: /* Send SIGXCPU every second.. */
1172: if (psecs * HZ == p->stime + p->utime)
1173: send_sig(SIGXCPU, p, 1);
1174: /* and SIGKILL when we go over max.. */
1175: if (psecs > p->rlim[RLIMIT_CPU].rlim_max)
1176: send_sig(SIGKILL, p, 1);
1177: }
1178: }
1179:
1180: static inline void do_it_virt(struct task_struct * p, unsigned long ticks)
1181: {
1182: unsigned long it_virt = p->it_virt_value;
1183:
1184: if (it_virt) {
1185: if (it_virt <= ticks) {
1186: it_virt = ticks + p->it_virt_incr;
1187: send_sig(SIGVTALRM, p, 1);
1188: }
1189: p->it_virt_value = it_virt - ticks;
1190: }
1191: }
1192:
1193: static inline void do_it_prof(struct task_struct * p, unsigned long ticks)
1194: {
1195: unsigned long it_prof = p->it_prof_value;
1196:
1197: if (it_prof) {
1198: if (it_prof <= ticks) {
1199: it_prof = ticks + p->it_prof_incr;
1200: send_sig(SIGPROF, p, 1);
1201: }
1202: p->it_prof_value = it_prof - ticks;
1203: }
1204: }
1205:
1206: static __inline__ void update_one_process(struct task_struct *p,
1207: unsigned long ticks, unsigned long user, unsigned long system)
1208: {
1209: do_process_times(p, user, system);
1210: do_it_virt(p, user);
1211: do_it_prof(p, ticks);
1212: }
1213:
1214: static void update_process_times(unsigned long ticks, unsigned long system)
1215: {
1216: #ifndef __SMP__
1217: struct task_struct * p = current;
1218: unsigned long user = ticks - system;
1219: if (p->pid) {
1220: p->counter -= ticks;
1221: if (p->counter < 0) {
1222: p->counter = 0;
1223: need_resched = 1;
1224: }
1225: if (p->priority < DEF_PRIORITY)
1226: kstat.cpu_nice += user;
1227: else
1228: kstat.cpu_user += user;
1229: kstat.cpu_system += system;
1230: }
1231: update_one_process(p, ticks, user, system);
1232: #else
1233: int cpu,j;
1234: cpu = smp_processor_id();
1235: for (j=0;j<smp_num_cpus;j++)
1236: {
1237: int i = cpu_logical_map[j];
1238: struct task_struct *p;
1239:
1240: #ifdef __SMP_PROF__
1241: if (test_bit(i,&smp_idle_map))
1242: smp_idle_count[i]++;
1243: #endif
1244: p = current_set[i];
1245: /*
1246: * Do we have a real process?
1247: */
1248: if (p->pid) {
1249: /* assume user-mode process */
1250: unsigned long utime = ticks;
1251: unsigned long stime = 0;
1252: if (cpu == i) {
1253: utime = ticks-system;
1254: stime = system;
1255: } else if (smp_proc_in_lock[j]) {
1256: utime = 0;
1257: stime = ticks;
1258: }
1259: update_one_process(p, ticks, utime, stime);
1260:
1261: if (p->priority < DEF_PRIORITY)
1262: kstat.cpu_nice += utime;
1263: else
1264: kstat.cpu_user += utime;
1265: kstat.cpu_system += stime;
1266:
1267: p->counter -= ticks;
1268: if (p->counter >= 0)
1269: continue;
1270: p->counter = 0;
1271: } else {
1272: /*
1273: * Idle processor found, do we have anything
1274: * we could run?
1275: */
1276: if (!(0x7fffffff & smp_process_available))
1277: continue;
1278: }
1279: /* Ok, we should reschedule, do the magic */
1280: if (i==cpu)
1281: need_resched = 1;
1282: else
1283: smp_message_pass(i, MSG_RESCHEDULE, 0L, 0);
1284: }
1285: #endif
1286: }
1287:
1288: static unsigned long lost_ticks = 0;
1289: static unsigned long lost_ticks_system = 0;
1290:
1291: static inline void update_times(void)
1292: {
1293: unsigned long ticks;
1294:
1295: ticks = xchg(&lost_ticks, 0);
1296:
1297: if (ticks) {
1298: unsigned long system;
1299:
1300: system = xchg(&lost_ticks_system, 0);
1301: calc_load(ticks);
1302: update_wall_time(ticks);
1303: update_process_times(ticks, system);
1304: }
1305: }
1306:
1307: static void timer_bh(void)
1308: {
1309: update_times();
1310: run_old_timers();
1311: run_timer_list();
1312: }
1313:
1314: void do_timer(struct pt_regs * regs)
1315: {
1316: (*(unsigned long *)&jiffies)++;
1317: lost_ticks++;
1318: mark_bh(TIMER_BH);
1319: if (!user_mode(regs)) {
1320: lost_ticks_system++;
1321: if (prof_buffer && current->pid) {
1322: extern int _stext;
1323: unsigned long ip = instruction_pointer(regs);
1324: ip -= (unsigned long) &_stext;
1325: ip >>= prof_shift;
1326: if (ip < prof_len)
1327: prof_buffer[ip]++;
1328: }
1329: }
1330: if (tq_timer)
1331: mark_bh(TQUEUE_BH);
1332: }
1333:
1334: #ifndef __alpha__
1335:
1336: /*
1337: * For backwards compatibility? This can be done in libc so Alpha
1338: * and all newer ports shouldn't need it.
1339: */
1340: asmlinkage unsigned int sys_alarm(unsigned int seconds)
1341: {
1342: struct itimerval it_new, it_old;
1343: unsigned int oldalarm;
1344:
1345: it_new.it_interval.tv_sec = it_new.it_interval.tv_usec = 0;
1346: it_new.it_value.tv_sec = seconds;
1347: it_new.it_value.tv_usec = 0;
1348: _setitimer(ITIMER_REAL, &it_new, &it_old);
1349: oldalarm = it_old.it_value.tv_sec;
1350: /* ehhh.. We can't return 0 if we have an alarm pending.. */
1351: /* And we'd better return too much than too little anyway */
1352: if (it_old.it_value.tv_usec)
1353: oldalarm++;
1354: return oldalarm;
1355: }
1356:
1357: /*
1358: * The Alpha uses getxpid, getxuid, and getxgid instead. Maybe this
1359: * should be moved into arch/i386 instead?
1360: */
1361: asmlinkage int sys_getpid(void)
1362: {
1363: return current->pid;
1364: }
1365:
1366: asmlinkage int sys_getppid(void)
1367: {
1368: return current->p_opptr->pid;
1369: }
1370:
1371: asmlinkage int sys_getuid(void)
1372: {
1373: return current->uid;
1374: }
1375:
1376: asmlinkage int sys_geteuid(void)
1377: {
1378: return current->euid;
1379: }
1380:
1381: asmlinkage int sys_getgid(void)
1382: {
1383: return current->gid;
1384: }
1385:
1386: asmlinkage int sys_getegid(void)
1387: {
1388: return current->egid;
1389: }
1390:
1391: /*
1392: * This has been replaced by sys_setpriority. Maybe it should be
1393: * moved into the arch dependent tree for those ports that require
1394: * it for backward compatibility?
1395: */
1396: asmlinkage int sys_nice(int increment)
1397: {
1398: unsigned long newprio;
1399: int increase = 0;
1400:
1401: newprio = increment;
1402: if (increment < 0) {
1403: if (!suser())
1404: return -EPERM;
1405: newprio = -increment;
1406: increase = 1;
1407: }
1408: if (newprio > 40)
1409: newprio = 40;
1410: /*
1411: * do a "normalization" of the priority (traditionally
1412: * unix nice values are -20..20, linux doesn't really
1413: * use that kind of thing, but uses the length of the
1414: * timeslice instead (default 150 msec). The rounding is
1415: * why we want to avoid negative values.
1416: */
1417: newprio = (newprio * DEF_PRIORITY + 10) / 20;
1418: increment = newprio;
1419: if (increase)
1420: increment = -increment;
1421: newprio = current->priority - increment;
1422: if ((signed) newprio < 1)
1423: newprio = 1;
1424: if (newprio > DEF_PRIORITY*2)
1425: newprio = DEF_PRIORITY*2;
1426: current->priority = newprio;
1427: return 0;
1428: }
1429:
1430: #endif
1431:
1432: static struct task_struct *find_process_by_pid(pid_t pid) {
1433: struct task_struct *p, *q;
1434:
1435: if (pid == 0)
1436: p = current;
1437: else {
1438: p = 0;
1439: for_each_task(q) {
1440: if (q && q->pid == pid) {
1441: p = q;
1442: break;
1443: }
1444: }
1445: }
1446: return p;
1447: }
1448:
1449: static int setscheduler(pid_t pid, int policy,
1450: struct sched_param *param)
1451: {
1452: int error;
1453: struct sched_param lp;
1454: struct task_struct *p;
1455:
1456: if (!param || pid < 0)
1457: return -EINVAL;
1458:
1459: error = verify_area(VERIFY_READ, param, sizeof(struct sched_param));
1460: if (error)
1461: return error;
1462: memcpy_fromfs(&lp, param, sizeof(struct sched_param));
1463:
1464: p = find_process_by_pid(pid);
1465: if (!p)
1466: return -ESRCH;
1467:
1468: if (policy < 0)
1469: policy = p->policy;
1470: else if (policy != SCHED_FIFO && policy != SCHED_RR &&
1471: policy != SCHED_OTHER)
1472: return -EINVAL;
1473:
1474: /*
1475: * Valid priorities for SCHED_FIFO and SCHED_RR are 1..99, valid
1476: * priority for SCHED_OTHER is 0.
1477: */
1478: if (lp.sched_priority < 0 || lp.sched_priority > 99)
1479: return -EINVAL;
1480: if ((policy == SCHED_OTHER) != (lp.sched_priority == 0))
1481: return -EINVAL;
1482:
1483: if ((policy == SCHED_FIFO || policy == SCHED_RR) && !suser())
1484: return -EPERM;
1485: if ((current->euid != p->euid) && (current->euid != p->uid) &&
1486: !suser())
1487: return -EPERM;
1488:
1489: p->policy = policy;
1490: p->rt_priority = lp.sched_priority;
1491: cli();
1492: if (p->next_run)
1493: move_last_runqueue(p);
1494: sti();
1495: need_resched = 1;
1496: return 0;
1497: }
1498:
1499: asmlinkage int sys_sched_setscheduler(pid_t pid, int policy,
1500: struct sched_param *param)
1501: {
1502: return setscheduler(pid, policy, param);
1503: }
1504:
1505: asmlinkage int sys_sched_setparam(pid_t pid, struct sched_param *param)
1506: {
1507: return setscheduler(pid, -1, param);
1508: }
1509:
1510: asmlinkage int sys_sched_getscheduler(pid_t pid)
1511: {
1512: struct task_struct *p;
1513:
1514: if (pid < 0)
1515: return -EINVAL;
1516:
1517: p = find_process_by_pid(pid);
1518: if (!p)
1519: return -ESRCH;
1520:
1521: return p->policy;
1522: }
1523:
1524: asmlinkage int sys_sched_getparam(pid_t pid, struct sched_param *param)
1525: {
1526: int error;
1527: struct task_struct *p;
1528: struct sched_param lp;
1529:
1530: if (!param || pid < 0)
1531: return -EINVAL;
1532:
1533: error = verify_area(VERIFY_WRITE, param, sizeof(struct sched_param));
1534: if (error)
1535: return error;
1536:
1537: p = find_process_by_pid(pid);
1538: if (!p)
1539: return -ESRCH;
1540:
1541: lp.sched_priority = p->rt_priority;
1542: memcpy_tofs(param, &lp, sizeof(struct sched_param));
1543:
1544: return 0;
1545: }
1546:
1547: asmlinkage int sys_sched_yield(void)
1548: {
1549: cli();
1550: move_last_runqueue(current);
1551: current->counter = 0;
1552: need_resched = 1;
1553: sti();
1554: return 0;
1555: }
1556:
1557: asmlinkage int sys_sched_get_priority_max(int policy)
1558: {
1559: switch (policy) {
1560: case SCHED_FIFO:
1561: case SCHED_RR:
1562: return 99;
1563: case SCHED_OTHER:
1564: return 0;
1565: }
1566:
1567: return -EINVAL;
1568: }
1569:
1570: asmlinkage int sys_sched_get_priority_min(int policy)
1571: {
1572: switch (policy) {
1573: case SCHED_FIFO:
1574: case SCHED_RR:
1575: return 1;
1576: case SCHED_OTHER:
1577: return 0;
1578: }
1579:
1580: return -EINVAL;
1581: }
1582:
1583: asmlinkage int sys_sched_rr_get_interval(pid_t pid, struct timespec *interval)
1584: {
1585: int error;
1586: struct timespec t;
1587:
1588: error = verify_area(VERIFY_WRITE, interval, sizeof(struct timespec));
1589: if (error)
1590: return error;
1591:
1592: /* Values taken from 2.1.38 */
1593: t.tv_sec = 0;
1594: t.tv_nsec = 150000; /* is this right for non-intel architecture too?*/
1595: memcpy_tofs(interval, &t, sizeof(struct timespec));
1596:
1597: return 0;
1598: }
1599:
1600: /*
1601: * change timeval to jiffies, trying to avoid the
1602: * most obvious overflows..
1603: */
1604: static unsigned long timespectojiffies(struct timespec *value)
1605: {
1606: unsigned long sec = (unsigned) value->tv_sec;
1607: long nsec = value->tv_nsec;
1608:
1609: if (sec > (LONG_MAX / HZ))
1610: return LONG_MAX;
1611: nsec += 1000000000L / HZ - 1;
1612: nsec /= 1000000000L / HZ;
1613: return HZ * sec + nsec;
1614: }
1615:
1616: static void jiffiestotimespec(unsigned long jiffies, struct timespec *value)
1617: {
1618: value->tv_nsec = (jiffies % HZ) * (1000000000L / HZ);
1619: value->tv_sec = jiffies / HZ;
1620: return;
1621: }
1622:
1623: asmlinkage int sys_nanosleep(struct timespec *rqtp, struct timespec *rmtp)
1624: {
1625: int error;
1626: struct timespec t;
1627: unsigned long expire;
1628:
1629: error = verify_area(VERIFY_READ, rqtp, sizeof(struct timespec));
1630: if (error)
1631: return error;
1632: memcpy_fromfs(&t, rqtp, sizeof(struct timespec));
1633: if (rmtp) {
1634: error = verify_area(VERIFY_WRITE, rmtp,
1635: sizeof(struct timespec));
1636: if (error)
1637: return error;
1638: }
1639:
1640: if (t.tv_nsec >= 1000000000L || t.tv_nsec < 0 || t.tv_sec < 0)
1641: return -EINVAL;
1642:
1643: if (t.tv_sec == 0 && t.tv_nsec <= 2000000L &&
1644: current->policy != SCHED_OTHER) {
1645: /*
1646: * Short delay requests up to 2 ms will be handled with
1647: * high precision by a busy wait for all real-time processes.
1648: */
1649: udelay((t.tv_nsec + 999) / 1000);
1650: return 0;
1651: }
1652:
1653: expire = timespectojiffies(&t) + (t.tv_sec || t.tv_nsec) + jiffies;
1654: current->timeout = expire;
1655: current->state = TASK_INTERRUPTIBLE;
1656: schedule();
1657:
1658: if (expire > jiffies) {
1659: if (rmtp) {
1660: jiffiestotimespec(expire - jiffies -
1661: (expire > jiffies + 1), &t);
1662: memcpy_tofs(rmtp, &t, sizeof(struct timespec));
1663: }
1664: return -EINTR;
1665: }
1666:
1667: return 0;
1668: }
1669:
1670: static void show_task(int nr,struct task_struct * p)
1671: {
1672: unsigned long free;
1673: static const char * stat_nam[] = { "R", "S", "D", "Z", "T", "W" };
1674:
1675: printk("%-8s %3d ", p->comm, (p == current) ? -nr : nr);
1676: if (((unsigned) p->state) < sizeof(stat_nam)/sizeof(char *))
1677: printk(stat_nam[p->state]);
1678: else
1679: printk(" ");
1680: #if ((~0UL) == 0xffffffff)
1681: if (p == current)
1682: printk(" current ");
1683: else
1684: printk(" %08lX ", thread_saved_pc(&p->tss));
1685: printk("%08lX ", get_wchan(p));
1686: #else
1687: if (p == current)
1688: printk(" current task ");
1689: else
1690: printk(" %016lx ", thread_saved_pc(&p->tss));
1691: printk("%08lX ", get_wchan(p) & 0xffffffffL);
1692: #endif
1693: for (free = 1; free < PAGE_SIZE/sizeof(long) ; free++) {
1694: if (((unsigned long *)p->kernel_stack_page)[free])
1695: break;
1696: }
1697: printk("%5lu %5d %6d ", free*sizeof(long), p->pid, p->p_pptr->pid);
1698: if (p->p_cptr)
1699: printk("%5d ", p->p_cptr->pid);
1700: else
1701: printk(" ");
1702: if (p->p_ysptr)
1703: printk("%7d", p->p_ysptr->pid);
1704: else
1705: printk(" ");
1706: if (p->p_osptr)
1707: printk(" %5d\n", p->p_osptr->pid);
1708: else
1709: printk("\n");
1710: }
1711:
1712: void show_state(void)
1713: {
1714: int i;
1715:
1716: #if ((~0UL) == 0xffffffff)
1717: printk("\n"
1718: " free sibling\n");
1719: printk(" task PC wchan stack pid father child younger older\n");
1720: #else
1721: printk("\n"
1722: " free sibling\n");
1723: printk(" task PC wchan stack pid father child younger older\n");
1724: #endif
1725: for (i=0 ; i<NR_TASKS ; i++)
1726: if (task[i])
1727: show_task(i,task[i]);
1728: }
1729:
1730: void sched_init(void)
1731: {
1732: /*
1733: * We have to do a little magic to get the first
1734: * process right in SMP mode.
1735: */
1736: int cpu=smp_processor_id();
1737: #ifndef __SMP__
1738: current_set[cpu]=&init_task;
1739: #else
1740: init_task.processor=cpu;
1741: for(cpu = 0; cpu < NR_CPUS; cpu++)
1742: current_set[cpu] = &init_task;
1743: #endif
1744: init_bh(TIMER_BH, timer_bh);
1745: init_bh(TQUEUE_BH, tqueue_bh);
1746: init_bh(IMMEDIATE_BH, immediate_bh);
1747: }
This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.