diff --git a/include/linux/sched/sysctl.h b/include/linux/sched/sysctl.h index 82fd187bb282..aac23569af5f 100644 --- a/include/linux/sched/sysctl.h +++ b/include/linux/sched/sysctl.h @@ -74,7 +74,7 @@ sched_ravg_window_handler(struct ctl_table *table, int write, loff_t *ppos); #endif -#if defined(CONFIG_PREEMPT_TRACER) || defined(CONFIG_DEBUG_PREEMPT) +#if defined(CONFIG_PREEMPTIRQ_EVENTS) || defined(CONFIG_PREEMPT_TRACER) extern unsigned int sysctl_preemptoff_tracing_threshold_ns; #endif #if defined(CONFIG_PREEMPTIRQ_EVENTS) && defined(CONFIG_IRQSOFF_TRACER) diff --git a/include/trace/events/preemptirq.h b/include/trace/events/preemptirq.h index 83bca4dc6334..d65e59731990 100644 --- a/include/trace/events/preemptirq.h +++ b/include/trace/events/preemptirq.h @@ -90,6 +90,38 @@ TRACE_EVENT(irqs_disable, __entry->caddr2, __entry->caddr3) ); +TRACE_EVENT(sched_preempt_disable, + + TP_PROTO(u64 delta, bool irqs_disabled, + unsigned long caddr0, unsigned long caddr1, + unsigned long caddr2, unsigned long caddr3), + + TP_ARGS(delta, irqs_disabled, caddr0, caddr1, caddr2, caddr3), + + TP_STRUCT__entry( + __field(u64, delta) + __field(bool, irqs_disabled) + __field(void*, caddr0) + __field(void*, caddr1) + __field(void*, caddr2) + __field(void*, caddr3) + ), + + TP_fast_assign( + __entry->delta = delta; + __entry->irqs_disabled = irqs_disabled; + __entry->caddr0 = (void *)caddr0; + __entry->caddr1 = (void *)caddr1; + __entry->caddr2 = (void *)caddr2; + __entry->caddr3 = (void *)caddr3; + ), + + TP_printk("delta=%llu(ns) irqs_d=%d Callers:(%ps<-%ps<-%ps<-%ps)", + __entry->delta, __entry->irqs_disabled, + __entry->caddr0, __entry->caddr1, + __entry->caddr2, __entry->caddr3) +); + #endif /* _TRACE_PREEMPTIRQ_H */ #include diff --git a/include/trace/events/sched.h b/include/trace/events/sched.h index e6fa9fe233e9..0e92f30fa061 100644 --- a/include/trace/events/sched.h +++ b/include/trace/events/sched.h @@ -1137,38 +1137,6 @@ TRACE_EVENT_CONDITION(sched_overutilized, ); #endif -TRACE_EVENT(sched_preempt_disable, - - TP_PROTO(u64 delta, bool irqs_disabled, - unsigned long caddr0, unsigned long caddr1, - unsigned long caddr2, unsigned long caddr3), - - TP_ARGS(delta, irqs_disabled, caddr0, caddr1, caddr2, caddr3), - - TP_STRUCT__entry( - __field(u64, delta) - __field(bool, irqs_disabled) - __field(void*, caddr0) - __field(void*, caddr1) - __field(void*, caddr2) - __field(void*, caddr3) - ), - - TP_fast_assign( - __entry->delta = delta; - __entry->irqs_disabled = irqs_disabled; - __entry->caddr0 = (void *)caddr0; - __entry->caddr1 = (void *)caddr1; - __entry->caddr2 = (void *)caddr2; - __entry->caddr3 = (void *)caddr3; - ), - - TP_printk("delta=%llu(ns) irqs_d=%d Callers:(%ps<-%ps<-%ps<-%ps)", - __entry->delta, __entry->irqs_disabled, - __entry->caddr0, __entry->caddr1, - __entry->caddr2, __entry->caddr3) -); - #endif /* _TRACE_SCHED_H */ /* This part must be outside protection */ diff --git a/kernel/sched/core.c b/kernel/sched/core.c index ddc508634516..b2fa01eecaf2 100644 --- a/kernel/sched/core.c +++ b/kernel/sched/core.c @@ -3870,55 +3870,17 @@ static inline void sched_tick_stop(int cpu) { } #if defined(CONFIG_PREEMPTION) && (defined(CONFIG_DEBUG_PREEMPT) || \ defined(CONFIG_TRACE_PREEMPT_TOGGLE)) -/* - * preemptoff stack tracing threshold in ns. - * default: 1ms - */ -unsigned int sysctl_preemptoff_tracing_threshold_ns = 1000000UL; - -struct preempt_store { - u64 ts; - unsigned long caddr[4]; - bool irqs_disabled; -}; - -DEFINE_PER_CPU(struct preempt_store, the_ps); - -/* - * This is only called from __schedule() upon context switch. - * - * schedule() calls __schedule() with preemption disabled. - * if we had entered idle and exiting idle now, reset the preemption - * tracking otherwise we may think preemption is disabled the whole time - * when the non idle task re-enables the preemption in schedule(). - */ -static inline void preempt_latency_reset(void) -{ - if (is_idle_task(this_rq()->curr)) - this_cpu_ptr(&the_ps)->ts = 0; -} - /* * If the value passed in is equal to the current preempt count * then we just disabled preemption. Start timing the latency. */ static inline void preempt_latency_start(int val) { - int cpu = raw_smp_processor_id(); - struct preempt_store *ps = &per_cpu(the_ps, cpu); - if (preempt_count() == val) { unsigned long ip = get_lock_parent_ip(); #ifdef CONFIG_DEBUG_PREEMPT current->preempt_disable_ip = ip; #endif - ps->ts = sched_clock(); - ps->caddr[0] = CALLER_ADDR0; - ps->caddr[1] = CALLER_ADDR1; - ps->caddr[2] = CALLER_ADDR2; - ps->caddr[3] = CALLER_ADDR3; - ps->irqs_disabled = irqs_disabled(); - trace_preempt_off(CALLER_ADDR0, ip); } } @@ -3951,22 +3913,8 @@ NOKPROBE_SYMBOL(preempt_count_add); */ static inline void preempt_latency_stop(int val) { - if (preempt_count() == val) { - struct preempt_store *ps = &per_cpu(the_ps, - raw_smp_processor_id()); - u64 delta = ps->ts ? (sched_clock() - ps->ts) : 0; - - /* - * Trace preempt disable stack if preemption - * is disabled for more than the threshold. - */ - if (delta > sysctl_preemptoff_tracing_threshold_ns) - trace_sched_preempt_disable(delta, ps->irqs_disabled, - ps->caddr[0], ps->caddr[1], - ps->caddr[2], ps->caddr[3]); - ps->ts = 0; + if (preempt_count() == val) trace_preempt_on(CALLER_ADDR0, get_lock_parent_ip()); - } } void preempt_count_sub(int val) @@ -3994,7 +3942,6 @@ NOKPROBE_SYMBOL(preempt_count_sub); #else static inline void preempt_latency_start(int val) { } static inline void preempt_latency_stop(int val) { } -static inline void preempt_latency_reset(void) { } #endif static inline unsigned long get_preempt_disable_ip(struct task_struct *p) @@ -4229,7 +4176,6 @@ static void __sched notrace __schedule(bool preempt) prev->last_sleep_ts = wallclock; #endif - preempt_latency_reset(); walt_update_task_ravg(prev, rq, PUT_PREV_TASK, wallclock, 0); walt_update_task_ravg(next, rq, PICK_NEXT_TASK, wallclock, 0); rq->nr_switches++; diff --git a/kernel/sysctl.c b/kernel/sysctl.c index 31b90a5a9a45..88986dded472 100644 --- a/kernel/sysctl.c +++ b/kernel/sysctl.c @@ -340,7 +340,7 @@ static struct ctl_table kern_table[] = { .mode = 0644, .proc_handler = proc_dointvec, }, -#if defined(CONFIG_PREEMPT_TRACER) || defined(CONFIG_DEBUG_PREEMPT) +#if defined(CONFIG_PREEMPT_TRACER) && defined(CONFIG_PREEMPTIRQ_EVENTS) { .procname = "preemptoff_tracing_threshold_ns", .data = &sysctl_preemptoff_tracing_threshold_ns, diff --git a/kernel/trace/trace_irqsoff.c b/kernel/trace/trace_irqsoff.c index 741c94432797..bd251354be5d 100644 --- a/kernel/trace/trace_irqsoff.c +++ b/kernel/trace/trace_irqsoff.c @@ -619,7 +619,7 @@ struct irqsoff_store { unsigned long caddr[4]; }; -DEFINE_PER_CPU(struct irqsoff_store, the_irqsoff); +static DEFINE_PER_CPU(struct irqsoff_store, the_irqsoff); #endif /* CONFIG_PREEMPTIRQ_EVENTS */ /* @@ -704,9 +704,57 @@ static struct tracer irqsoff_tracer __read_mostly = #endif /* CONFIG_IRQSOFF_TRACER */ #ifdef CONFIG_PREEMPT_TRACER +#ifdef CONFIG_PREEMPTIRQ_EVENTS +/* + * preemptoff stack tracing threshold in ns. + * default: 1ms + */ +unsigned int sysctl_preemptoff_tracing_threshold_ns = 1000000UL; + +struct preempt_store { + u64 ts; + unsigned long caddr[4]; + bool irqs_disabled; + int pid; + unsigned long ncsw; +}; + +static DEFINE_PER_CPU(struct preempt_store, the_ps); +#endif /* CONFIG_PREEMPTIRQ_EVENTS */ + void tracer_preempt_on(unsigned long a0, unsigned long a1) { int pc = preempt_count(); +#ifdef CONFIG_PREEMPTIRQ_EVENTS + struct preempt_store *ps; + u64 delta = 0; + + lockdep_off(); + ps = &per_cpu(the_ps, raw_smp_processor_id()); + + /* + * schedule() calls __schedule() with preemption disabled. + * if we had entered idle and exiting idle now, we think + * preemption is disabled the whole time. Detect this by + * checking if the preemption is disabled across the same + * task. There is a possiblity that the same task is scheduled + * after idle. To rule out this possibility, compare the + * context switch count also. + */ + if (ps->ts && ps->pid == current->pid && (ps->ncsw == + current->nvcsw + current->nivcsw)) + delta = sched_clock() - ps->ts; + /* + * Trace preempt disable stack if preemption + * is disabled for more than the threshold. + */ + if (delta > sysctl_preemptoff_tracing_threshold_ns) + trace_sched_preempt_disable(delta, ps->irqs_disabled, + ps->caddr[0], ps->caddr[1], + ps->caddr[2], ps->caddr[3]); + ps->ts = 0; + lockdep_on(); +#endif /* CONFIG_PREEMPTIRQ_EVENTS */ if (preempt_trace(pc) && !irq_trace()) stop_critical_timing(a0, a1, pc); @@ -715,6 +763,21 @@ void tracer_preempt_on(unsigned long a0, unsigned long a1) void tracer_preempt_off(unsigned long a0, unsigned long a1) { int pc = preempt_count(); +#ifdef CONFIG_PREEMPTIRQ_EVENTS + struct preempt_store *ps; + + lockdep_off(); + ps = &per_cpu(the_ps, raw_smp_processor_id()); + ps->ts = sched_clock(); + ps->caddr[0] = CALLER_ADDR0; + ps->caddr[1] = CALLER_ADDR1; + ps->caddr[2] = CALLER_ADDR2; + ps->caddr[3] = CALLER_ADDR3; + ps->irqs_disabled = irqs_disabled(); + ps->pid = current->pid; + ps->ncsw = current->nvcsw + current->nivcsw; + lockdep_on(); +#endif /* CONFIG_PREEMPTIRQ_EVENTS */ if (preempt_trace(pc) && !irq_trace()) start_critical_timing(a0, a1, pc);