2010-05-08 05:11:44 +08:00
|
|
|
/*
|
|
|
|
* Detect hard and soft lockups on a system
|
|
|
|
*
|
|
|
|
* started by Don Zickus, Copyright (C) 2010 Red Hat, Inc.
|
|
|
|
*
|
2012-02-10 06:42:22 +08:00
|
|
|
* Note: Most of this code is borrowed heavily from the original softlockup
|
|
|
|
* detector, so thanks to Ingo for the initial implementation.
|
|
|
|
* Some chunks also taken from the old x86-specific nmi watchdog code, thanks
|
2010-05-08 05:11:44 +08:00
|
|
|
* to those contributors as well.
|
|
|
|
*/
|
|
|
|
|
2012-03-24 06:01:55 +08:00
|
|
|
#define pr_fmt(fmt) "NMI watchdog: " fmt
|
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
#include <linux/mm.h>
|
|
|
|
#include <linux/cpu.h>
|
|
|
|
#include <linux/nmi.h>
|
|
|
|
#include <linux/init.h>
|
|
|
|
#include <linux/delay.h>
|
|
|
|
#include <linux/freezer.h>
|
|
|
|
#include <linux/kthread.h>
|
|
|
|
#include <linux/lockdep.h>
|
|
|
|
#include <linux/notifier.h>
|
|
|
|
#include <linux/module.h>
|
|
|
|
#include <linux/sysctl.h>
|
2012-07-16 18:42:38 +08:00
|
|
|
#include <linux/smpboot.h>
|
2013-02-07 23:47:07 +08:00
|
|
|
#include <linux/sched/rt.h>
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
#include <asm/irq_regs.h>
|
2012-03-11 03:37:28 +08:00
|
|
|
#include <linux/kvm_para.h>
|
2010-05-08 05:11:44 +08:00
|
|
|
#include <linux/perf_event.h>
|
|
|
|
|
2013-05-20 02:45:15 +08:00
|
|
|
int watchdog_user_enabled = 1;
|
2011-05-23 13:10:23 +08:00
|
|
|
int __read_mostly watchdog_thresh = 10;
|
2013-05-20 02:45:15 +08:00
|
|
|
static int __read_mostly watchdog_running;
|
2012-12-18 07:59:50 +08:00
|
|
|
static u64 __read_mostly sample_period;
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
static DEFINE_PER_CPU(unsigned long, watchdog_touch_ts);
|
|
|
|
static DEFINE_PER_CPU(struct task_struct *, softlockup_watchdog);
|
|
|
|
static DEFINE_PER_CPU(struct hrtimer, watchdog_hrtimer);
|
|
|
|
static DEFINE_PER_CPU(bool, softlockup_touch_sync);
|
|
|
|
static DEFINE_PER_CPU(bool, soft_watchdog_warn);
|
2012-07-16 18:42:38 +08:00
|
|
|
static DEFINE_PER_CPU(unsigned long, hrtimer_interrupts);
|
|
|
|
static DEFINE_PER_CPU(unsigned long, soft_lockup_hrtimer_cnt);
|
2010-05-16 05:15:20 +08:00
|
|
|
#ifdef CONFIG_HARDLOCKUP_DETECTOR
|
2010-05-14 23:11:21 +08:00
|
|
|
static DEFINE_PER_CPU(bool, hard_watchdog_warn);
|
|
|
|
static DEFINE_PER_CPU(bool, watchdog_nmi_touch);
|
2010-05-08 05:11:44 +08:00
|
|
|
static DEFINE_PER_CPU(unsigned long, hrtimer_interrupts_saved);
|
|
|
|
static DEFINE_PER_CPU(struct perf_event *, watchdog_ev);
|
|
|
|
#endif
|
|
|
|
|
|
|
|
/* boot commands */
|
|
|
|
/*
|
|
|
|
* Should we panic when a soft-lockup or hard-lockup occurs:
|
|
|
|
*/
|
2010-05-16 05:15:20 +08:00
|
|
|
#ifdef CONFIG_HARDLOCKUP_DETECTOR
|
2011-03-23 07:34:16 +08:00
|
|
|
static int hardlockup_panic =
|
|
|
|
CONFIG_BOOTPARAM_HARDLOCKUP_PANIC_VALUE;
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
static int __init hardlockup_panic_setup(char *str)
|
|
|
|
{
|
|
|
|
if (!strncmp(str, "panic", 5))
|
|
|
|
hardlockup_panic = 1;
|
2011-03-23 07:34:16 +08:00
|
|
|
else if (!strncmp(str, "nopanic", 7))
|
|
|
|
hardlockup_panic = 0;
|
2010-11-30 06:07:17 +08:00
|
|
|
else if (!strncmp(str, "0", 1))
|
2013-05-20 02:45:15 +08:00
|
|
|
watchdog_user_enabled = 0;
|
2010-05-08 05:11:44 +08:00
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
__setup("nmi_watchdog=", hardlockup_panic_setup);
|
|
|
|
#endif
|
|
|
|
|
|
|
|
unsigned int __read_mostly softlockup_panic =
|
|
|
|
CONFIG_BOOTPARAM_SOFTLOCKUP_PANIC_VALUE;
|
|
|
|
|
|
|
|
static int __init softlockup_panic_setup(char *str)
|
|
|
|
{
|
|
|
|
softlockup_panic = simple_strtoul(str, NULL, 0);
|
|
|
|
|
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
__setup("softlockup_panic=", softlockup_panic_setup);
|
|
|
|
|
|
|
|
static int __init nowatchdog_setup(char *str)
|
|
|
|
{
|
2013-05-20 02:45:15 +08:00
|
|
|
watchdog_user_enabled = 0;
|
2010-05-08 05:11:44 +08:00
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
__setup("nowatchdog", nowatchdog_setup);
|
|
|
|
|
|
|
|
/* deprecated */
|
|
|
|
static int __init nosoftlockup_setup(char *str)
|
|
|
|
{
|
2013-05-20 02:45:15 +08:00
|
|
|
watchdog_user_enabled = 0;
|
2010-05-08 05:11:44 +08:00
|
|
|
return 1;
|
|
|
|
}
|
|
|
|
__setup("nosoftlockup", nosoftlockup_setup);
|
|
|
|
/* */
|
|
|
|
|
2011-05-23 13:10:23 +08:00
|
|
|
/*
|
|
|
|
* Hard-lockup warnings should be triggered after just a few seconds. Soft-
|
|
|
|
* lockups can have false positives under extreme conditions. So we generally
|
|
|
|
* want a higher threshold for soft lockups than for hard lockups. So we couple
|
|
|
|
* the thresholds with a factor: we make the soft threshold twice the amount of
|
|
|
|
* time the hard threshold is.
|
|
|
|
*/
|
2011-05-24 11:43:18 +08:00
|
|
|
static int get_softlockup_thresh(void)
|
2011-05-23 13:10:23 +08:00
|
|
|
{
|
|
|
|
return watchdog_thresh * 2;
|
|
|
|
}
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
/*
|
|
|
|
* Returns seconds, approximately. We don't need nanosecond
|
|
|
|
* resolution, and we don't need to waste time with a big divide when
|
|
|
|
* 2^30ns == 1.074s.
|
|
|
|
*/
|
2012-12-27 10:49:44 +08:00
|
|
|
static unsigned long get_timestamp(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2012-12-27 10:49:44 +08:00
|
|
|
return local_clock() >> 30LL; /* 2^30 ~= 10^9 */
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
|
2012-12-18 07:59:50 +08:00
|
|
|
static void set_sample_period(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
|
|
|
/*
|
2011-05-23 13:10:22 +08:00
|
|
|
* convert watchdog_thresh from seconds to ns
|
2012-02-10 06:42:22 +08:00
|
|
|
* the divide by 5 is to give hrtimer several chances (two
|
|
|
|
* or three with the current relation between the soft
|
|
|
|
* and hard thresholds) to increment before the
|
|
|
|
* hardlockup detector generates a warning
|
2010-05-08 05:11:44 +08:00
|
|
|
*/
|
2012-12-18 07:59:50 +08:00
|
|
|
sample_period = get_softlockup_thresh() * ((u64)NSEC_PER_SEC / 5);
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
/* Commands for resetting the watchdog */
|
|
|
|
static void __touch_watchdog(void)
|
|
|
|
{
|
2012-12-27 10:49:44 +08:00
|
|
|
__this_cpu_write(watchdog_touch_ts, get_timestamp());
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
|
2010-05-08 05:11:45 +08:00
|
|
|
void touch_softlockup_watchdog(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(watchdog_touch_ts, 0);
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
2010-05-13 14:53:33 +08:00
|
|
|
EXPORT_SYMBOL(touch_softlockup_watchdog);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2010-05-08 05:11:45 +08:00
|
|
|
void touch_all_softlockup_watchdogs(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
|
|
|
int cpu;
|
|
|
|
|
|
|
|
/*
|
|
|
|
* this is done lockless
|
|
|
|
* do we care if a 0 races with a timestamp?
|
|
|
|
* all it means is the softlock check starts one cycle later
|
|
|
|
*/
|
|
|
|
for_each_online_cpu(cpu)
|
|
|
|
per_cpu(watchdog_touch_ts, cpu) = 0;
|
|
|
|
}
|
|
|
|
|
2010-05-14 23:11:21 +08:00
|
|
|
#ifdef CONFIG_HARDLOCKUP_DETECTOR
|
2010-05-08 05:11:44 +08:00
|
|
|
void touch_nmi_watchdog(void)
|
|
|
|
{
|
2013-05-20 02:45:15 +08:00
|
|
|
if (watchdog_user_enabled) {
|
2010-09-01 11:00:07 +08:00
|
|
|
unsigned cpu;
|
|
|
|
|
|
|
|
for_each_present_cpu(cpu) {
|
|
|
|
if (per_cpu(watchdog_nmi_touch, cpu) != true)
|
|
|
|
per_cpu(watchdog_nmi_touch, cpu) = true;
|
|
|
|
}
|
|
|
|
}
|
2010-05-08 05:11:45 +08:00
|
|
|
touch_softlockup_watchdog();
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
EXPORT_SYMBOL(touch_nmi_watchdog);
|
|
|
|
|
2010-05-14 23:11:21 +08:00
|
|
|
#endif
|
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
void touch_softlockup_watchdog_sync(void)
|
|
|
|
{
|
|
|
|
__raw_get_cpu_var(softlockup_touch_sync) = true;
|
|
|
|
__raw_get_cpu_var(watchdog_touch_ts) = 0;
|
|
|
|
}
|
|
|
|
|
2010-05-16 05:15:20 +08:00
|
|
|
#ifdef CONFIG_HARDLOCKUP_DETECTOR
|
2010-05-08 05:11:44 +08:00
|
|
|
/* watchdog detector functions */
|
2010-05-18 06:06:04 +08:00
|
|
|
static int is_hardlockup(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2010-12-08 23:22:55 +08:00
|
|
|
unsigned long hrint = __this_cpu_read(hrtimer_interrupts);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2010-12-08 23:22:55 +08:00
|
|
|
if (__this_cpu_read(hrtimer_interrupts_saved) == hrint)
|
2010-05-08 05:11:44 +08:00
|
|
|
return 1;
|
|
|
|
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(hrtimer_interrupts_saved, hrint);
|
2010-05-08 05:11:44 +08:00
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
#endif
|
|
|
|
|
2010-05-18 06:06:04 +08:00
|
|
|
static int is_softlockup(unsigned long touch_ts)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2012-12-27 10:49:44 +08:00
|
|
|
unsigned long now = get_timestamp();
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
/* Warn about unreasonable delays: */
|
2011-05-23 13:10:23 +08:00
|
|
|
if (time_after(now, touch_ts + get_softlockup_thresh()))
|
2010-05-08 05:11:44 +08:00
|
|
|
return now - touch_ts;
|
|
|
|
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
2010-05-16 05:15:20 +08:00
|
|
|
#ifdef CONFIG_HARDLOCKUP_DETECTOR
|
2011-06-23 20:49:18 +08:00
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
static struct perf_event_attr wd_hw_attr = {
|
|
|
|
.type = PERF_TYPE_HARDWARE,
|
|
|
|
.config = PERF_COUNT_HW_CPU_CYCLES,
|
|
|
|
.size = sizeof(struct perf_event_attr),
|
|
|
|
.pinned = 1,
|
|
|
|
.disabled = 1,
|
|
|
|
};
|
|
|
|
|
|
|
|
/* Callback function for perf event subsystem */
|
2011-06-27 20:41:57 +08:00
|
|
|
static void watchdog_overflow_callback(struct perf_event *event,
|
2010-05-08 05:11:44 +08:00
|
|
|
struct perf_sample_data *data,
|
|
|
|
struct pt_regs *regs)
|
|
|
|
{
|
2010-08-20 17:49:15 +08:00
|
|
|
/* Ensure the watchdog never gets throttled */
|
|
|
|
event->hw.interrupts = 0;
|
|
|
|
|
2010-12-08 23:22:55 +08:00
|
|
|
if (__this_cpu_read(watchdog_nmi_touch) == true) {
|
|
|
|
__this_cpu_write(watchdog_nmi_touch, false);
|
2010-05-08 05:11:44 +08:00
|
|
|
return;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* check for a hardlockup
|
|
|
|
* This is done by making sure our timer interrupt
|
|
|
|
* is incrementing. The timer interrupt should have
|
|
|
|
* fired multiple times before we overflow'd. If it hasn't
|
|
|
|
* then this is a good indication the cpu is stuck
|
|
|
|
*/
|
2010-05-18 06:06:04 +08:00
|
|
|
if (is_hardlockup()) {
|
|
|
|
int this_cpu = smp_processor_id();
|
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
/* only print hardlockups once */
|
2010-12-08 23:22:55 +08:00
|
|
|
if (__this_cpu_read(hard_watchdog_warn) == true)
|
2010-05-08 05:11:44 +08:00
|
|
|
return;
|
|
|
|
|
|
|
|
if (hardlockup_panic)
|
|
|
|
panic("Watchdog detected hard LOCKUP on cpu %d", this_cpu);
|
|
|
|
else
|
|
|
|
WARN(1, "Watchdog detected hard LOCKUP on cpu %d", this_cpu);
|
|
|
|
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(hard_watchdog_warn, true);
|
2010-05-08 05:11:44 +08:00
|
|
|
return;
|
|
|
|
}
|
|
|
|
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(hard_watchdog_warn, false);
|
2010-05-08 05:11:44 +08:00
|
|
|
return;
|
|
|
|
}
|
2012-07-16 18:42:38 +08:00
|
|
|
#endif /* CONFIG_HARDLOCKUP_DETECTOR */
|
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
static void watchdog_interrupt_count(void)
|
|
|
|
{
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_inc(hrtimer_interrupts);
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
2012-07-16 18:42:38 +08:00
|
|
|
|
|
|
|
static int watchdog_nmi_enable(unsigned int cpu);
|
|
|
|
static void watchdog_nmi_disable(unsigned int cpu);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
/* watchdog kicker functions */
|
|
|
|
static enum hrtimer_restart watchdog_timer_fn(struct hrtimer *hrtimer)
|
|
|
|
{
|
2010-12-08 23:22:55 +08:00
|
|
|
unsigned long touch_ts = __this_cpu_read(watchdog_touch_ts);
|
2010-05-08 05:11:44 +08:00
|
|
|
struct pt_regs *regs = get_irq_regs();
|
|
|
|
int duration;
|
|
|
|
|
|
|
|
/* kick the hardlockup detector */
|
|
|
|
watchdog_interrupt_count();
|
|
|
|
|
|
|
|
/* kick the softlockup detector */
|
2010-12-08 23:22:55 +08:00
|
|
|
wake_up_process(__this_cpu_read(softlockup_watchdog));
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
/* .. and repeat */
|
2012-12-18 07:59:50 +08:00
|
|
|
hrtimer_forward_now(hrtimer, ns_to_ktime(sample_period));
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
if (touch_ts == 0) {
|
2010-12-08 23:22:55 +08:00
|
|
|
if (unlikely(__this_cpu_read(softlockup_touch_sync))) {
|
2010-05-08 05:11:44 +08:00
|
|
|
/*
|
|
|
|
* If the time stamp was touched atomically
|
|
|
|
* make sure the scheduler tick is up to date.
|
|
|
|
*/
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(softlockup_touch_sync, false);
|
2010-05-08 05:11:44 +08:00
|
|
|
sched_clock_tick();
|
|
|
|
}
|
2012-03-11 03:37:28 +08:00
|
|
|
|
|
|
|
/* Clear the guest paused flag on watchdog reset */
|
|
|
|
kvm_check_and_clear_guest_paused();
|
2010-05-08 05:11:44 +08:00
|
|
|
__touch_watchdog();
|
|
|
|
return HRTIMER_RESTART;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* check for a softlockup
|
|
|
|
* This is done by making sure a high priority task is
|
|
|
|
* being scheduled. The task touches the watchdog to
|
|
|
|
* indicate it is getting cpu time. If it hasn't then
|
|
|
|
* this is a good indication some task is hogging the cpu
|
|
|
|
*/
|
2010-05-18 06:06:04 +08:00
|
|
|
duration = is_softlockup(touch_ts);
|
2010-05-08 05:11:44 +08:00
|
|
|
if (unlikely(duration)) {
|
2012-03-11 03:37:28 +08:00
|
|
|
/*
|
|
|
|
* If a virtual machine is stopped by the host it can look to
|
|
|
|
* the watchdog like a soft lockup, check to see if the host
|
|
|
|
* stopped the vm before we issue the warning
|
|
|
|
*/
|
|
|
|
if (kvm_check_and_clear_guest_paused())
|
|
|
|
return HRTIMER_RESTART;
|
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
/* only warn once */
|
2010-12-08 23:22:55 +08:00
|
|
|
if (__this_cpu_read(soft_watchdog_warn) == true)
|
2010-05-08 05:11:44 +08:00
|
|
|
return HRTIMER_RESTART;
|
|
|
|
|
bugs, x86: Fix printk levels for panic, softlockups and stack dumps
rsyslog will display KERN_EMERG messages on a connected
terminal. However, these messages are useless/undecipherable
for a general user.
For example, after a softlockup we get:
Message from syslogd@intel-s3e37-04 at Jan 25 14:18:06 ...
kernel:Stack:
Message from syslogd@intel-s3e37-04 at Jan 25 14:18:06 ...
kernel:Call Trace:
Message from syslogd@intel-s3e37-04 at Jan 25 14:18:06 ...
kernel:Code: ff ff a8 08 75 25 31 d2 48 8d 86 38 e0 ff ff 48 89
d1 0f 01 c8 0f ae f0 48 8b 86 38 e0 ff ff a8 08 75 08 b1 01 4c 89 e0 0f 01 c9 <e8> ea 69 dd ff 4c 29 e8 48 89 c7 e8 0f bc da ff 49 89 c4 49 89
This happens because the printk levels for these messages are
incorrect. Only an informational message should be displayed on
a terminal.
I modified the printk levels for various messages in the kernel
and tested the output by using the drivers/misc/lkdtm.c kernel
modules (ie, softlockups, panics, hard lockups, etc.) and
confirmed that the console output was still the same and that
the output to the terminals was correct.
For example, in the case of a softlockup we now see the much
more informative:
Message from syslogd@intel-s3e37-04 at Jan 25 10:18:06 ...
BUG: soft lockup - CPU4 stuck for 60s!
instead of the above confusing messages.
AFAICT, the messages no longer have to be KERN_EMERG. In the
most important case of a panic we set console_verbose(). As for
the other less severe cases the correct data is output to the
console and /var/log/messages.
Successfully tested by me using the drivers/misc/lkdtm.c module.
Signed-off-by: Prarit Bhargava <prarit@redhat.com>
Cc: dzickus@redhat.com
Cc: Linus Torvalds <torvalds@linux-foundation.org>
Cc: Andrew Morton <akpm@linux-foundation.org>
Link: http://lkml.kernel.org/r/1327586134-11926-1-git-send-email-prarit@redhat.com
Signed-off-by: Ingo Molnar <mingo@elte.hu>
2012-01-26 21:55:34 +08:00
|
|
|
printk(KERN_EMERG "BUG: soft lockup - CPU#%d stuck for %us! [%s:%d]\n",
|
2010-05-18 06:06:04 +08:00
|
|
|
smp_processor_id(), duration,
|
2010-05-08 05:11:44 +08:00
|
|
|
current->comm, task_pid_nr(current));
|
|
|
|
print_modules();
|
|
|
|
print_irqtrace_events(current);
|
|
|
|
if (regs)
|
|
|
|
show_regs(regs);
|
|
|
|
else
|
|
|
|
dump_stack();
|
|
|
|
|
|
|
|
if (softlockup_panic)
|
|
|
|
panic("softlockup: hung tasks");
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(soft_watchdog_warn, true);
|
2010-05-08 05:11:44 +08:00
|
|
|
} else
|
2010-12-08 23:22:55 +08:00
|
|
|
__this_cpu_write(soft_watchdog_warn, false);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
return HRTIMER_RESTART;
|
|
|
|
}
|
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
static void watchdog_set_prio(unsigned int policy, unsigned int prio)
|
|
|
|
{
|
|
|
|
struct sched_param param = { .sched_priority = prio };
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
sched_setscheduler(current, policy, ¶m);
|
|
|
|
}
|
|
|
|
|
|
|
|
static void watchdog_enable(unsigned int cpu)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2010-05-18 06:06:04 +08:00
|
|
|
struct hrtimer *hrtimer = &__raw_get_cpu_var(watchdog_hrtimer);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2012-12-20 03:51:31 +08:00
|
|
|
/* kick off the timer for the hardlockup detector */
|
|
|
|
hrtimer_init(hrtimer, CLOCK_MONOTONIC, HRTIMER_MODE_REL);
|
|
|
|
hrtimer->function = watchdog_timer_fn;
|
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
/* Enable the perf event */
|
|
|
|
watchdog_nmi_enable(cpu);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
/* done here because hrtimer_start can only pin to smp_processor_id() */
|
2012-12-18 07:59:50 +08:00
|
|
|
hrtimer_start(hrtimer, ns_to_ktime(sample_period),
|
2010-05-08 05:11:44 +08:00
|
|
|
HRTIMER_MODE_REL_PINNED);
|
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
/* initialize timestamp */
|
|
|
|
watchdog_set_prio(SCHED_FIFO, MAX_RT_PRIO - 1);
|
|
|
|
__touch_watchdog();
|
|
|
|
}
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
static void watchdog_disable(unsigned int cpu)
|
|
|
|
{
|
|
|
|
struct hrtimer *hrtimer = &__raw_get_cpu_var(watchdog_hrtimer);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
watchdog_set_prio(SCHED_NORMAL, 0);
|
|
|
|
hrtimer_cancel(hrtimer);
|
|
|
|
/* disable the perf event */
|
|
|
|
watchdog_nmi_disable(cpu);
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
|
2013-06-06 21:42:53 +08:00
|
|
|
static void watchdog_cleanup(unsigned int cpu, bool online)
|
|
|
|
{
|
|
|
|
watchdog_disable(cpu);
|
|
|
|
}
|
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
static int watchdog_should_run(unsigned int cpu)
|
|
|
|
{
|
|
|
|
return __this_cpu_read(hrtimer_interrupts) !=
|
|
|
|
__this_cpu_read(soft_lockup_hrtimer_cnt);
|
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
|
|
|
* The watchdog thread function - touches the timestamp.
|
|
|
|
*
|
2012-12-18 07:59:50 +08:00
|
|
|
* It only runs once every sample_period seconds (4 seconds by
|
2012-07-16 18:42:38 +08:00
|
|
|
* default) to reset the softlockup timestamp. If this gets delayed
|
|
|
|
* for more than 2*watchdog_thresh seconds then the debug-printout
|
|
|
|
* triggers in watchdog_timer_fn().
|
|
|
|
*/
|
|
|
|
static void watchdog(unsigned int cpu)
|
|
|
|
{
|
|
|
|
__this_cpu_write(soft_lockup_hrtimer_cnt,
|
|
|
|
__this_cpu_read(hrtimer_interrupts));
|
|
|
|
__touch_watchdog();
|
|
|
|
}
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2010-05-16 05:15:20 +08:00
|
|
|
#ifdef CONFIG_HARDLOCKUP_DETECTOR
|
watchdog: Quiet down the boot messages
A bunch of bugzillas have complained how noisy the nmi_watchdog
is during boot-up especially with its expected failure cases
(like virt and bios resource contention).
This is my attempt to quiet them down and keep it less confusing
for the end user. What I did is print the message for cpu0 and
save it for future comparisons. If future cpus have an
identical message as cpu0, then don't print the redundant info.
However, if a future cpu has a different message, happily print
that loudly.
Before the change, you would see something like:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog enabled, takes one hw-pmu counter.
Booting Node 0, Processors #1
NMI watchdog enabled, takes one hw-pmu counter.
#2
NMI watchdog enabled, takes one hw-pmu counter.
#3 Ok.
NMI watchdog enabled, takes one hw-pmu counter.
Brought up 4 CPUs
Total of 4 processors activated (22607.24 BogoMIPS).
After the change, it is simplified to:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog: enabled on all CPUs, permanently consumes one hw-PMU counter.
Booting Node 0, Processors #1 #2 #3 Ok.
Brought up 4 CPUs
V2: little changes based on Joe Perches' feedback
V3: printk cleanup based on Ingo's feedback; checkpatch fix
V4: keep printk as one long line
V5: Ingo fix ups
Reported-and-tested-by: Nathan Zimmer <nzimmer@sgi.com>
Signed-off-by: Don Zickus <dzickus@redhat.com>
Cc: nzimmer@sgi.com
Cc: joe@perches.com
Link: http://lkml.kernel.org/r/1339594548-17227-1-git-send-email-dzickus@redhat.com
Signed-off-by: Ingo Molnar <mingo@kernel.org>
2012-06-13 21:35:48 +08:00
|
|
|
/*
|
|
|
|
* People like the simple clean cpu node info on boot.
|
|
|
|
* Reduce the watchdog noise by only printing messages
|
|
|
|
* that are different from what cpu0 displayed.
|
|
|
|
*/
|
|
|
|
static unsigned long cpu0_err;
|
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
static int watchdog_nmi_enable(unsigned int cpu)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
|
|
|
struct perf_event_attr *wd_attr;
|
|
|
|
struct perf_event *event = per_cpu(watchdog_ev, cpu);
|
|
|
|
|
|
|
|
/* is it already setup and enabled? */
|
|
|
|
if (event && event->state > PERF_EVENT_STATE_OFF)
|
|
|
|
goto out;
|
|
|
|
|
|
|
|
/* it is setup but not enabled */
|
|
|
|
if (event != NULL)
|
|
|
|
goto out_enable;
|
|
|
|
|
|
|
|
wd_attr = &wd_hw_attr;
|
2011-05-23 13:10:23 +08:00
|
|
|
wd_attr->sample_period = hw_nmi_get_sample_period(watchdog_thresh);
|
2011-06-23 20:49:18 +08:00
|
|
|
|
|
|
|
/* Try to register using hardware perf events */
|
2011-06-29 23:42:35 +08:00
|
|
|
event = perf_event_create_kernel_counter(wd_attr, cpu, NULL, watchdog_overflow_callback, NULL);
|
watchdog: Quiet down the boot messages
A bunch of bugzillas have complained how noisy the nmi_watchdog
is during boot-up especially with its expected failure cases
(like virt and bios resource contention).
This is my attempt to quiet them down and keep it less confusing
for the end user. What I did is print the message for cpu0 and
save it for future comparisons. If future cpus have an
identical message as cpu0, then don't print the redundant info.
However, if a future cpu has a different message, happily print
that loudly.
Before the change, you would see something like:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog enabled, takes one hw-pmu counter.
Booting Node 0, Processors #1
NMI watchdog enabled, takes one hw-pmu counter.
#2
NMI watchdog enabled, takes one hw-pmu counter.
#3 Ok.
NMI watchdog enabled, takes one hw-pmu counter.
Brought up 4 CPUs
Total of 4 processors activated (22607.24 BogoMIPS).
After the change, it is simplified to:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog: enabled on all CPUs, permanently consumes one hw-PMU counter.
Booting Node 0, Processors #1 #2 #3 Ok.
Brought up 4 CPUs
V2: little changes based on Joe Perches' feedback
V3: printk cleanup based on Ingo's feedback; checkpatch fix
V4: keep printk as one long line
V5: Ingo fix ups
Reported-and-tested-by: Nathan Zimmer <nzimmer@sgi.com>
Signed-off-by: Don Zickus <dzickus@redhat.com>
Cc: nzimmer@sgi.com
Cc: joe@perches.com
Link: http://lkml.kernel.org/r/1339594548-17227-1-git-send-email-dzickus@redhat.com
Signed-off-by: Ingo Molnar <mingo@kernel.org>
2012-06-13 21:35:48 +08:00
|
|
|
|
|
|
|
/* save cpu0 error for future comparision */
|
|
|
|
if (cpu == 0 && IS_ERR(event))
|
|
|
|
cpu0_err = PTR_ERR(event);
|
|
|
|
|
2010-05-08 05:11:44 +08:00
|
|
|
if (!IS_ERR(event)) {
|
watchdog: Quiet down the boot messages
A bunch of bugzillas have complained how noisy the nmi_watchdog
is during boot-up especially with its expected failure cases
(like virt and bios resource contention).
This is my attempt to quiet them down and keep it less confusing
for the end user. What I did is print the message for cpu0 and
save it for future comparisons. If future cpus have an
identical message as cpu0, then don't print the redundant info.
However, if a future cpu has a different message, happily print
that loudly.
Before the change, you would see something like:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog enabled, takes one hw-pmu counter.
Booting Node 0, Processors #1
NMI watchdog enabled, takes one hw-pmu counter.
#2
NMI watchdog enabled, takes one hw-pmu counter.
#3 Ok.
NMI watchdog enabled, takes one hw-pmu counter.
Brought up 4 CPUs
Total of 4 processors activated (22607.24 BogoMIPS).
After the change, it is simplified to:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog: enabled on all CPUs, permanently consumes one hw-PMU counter.
Booting Node 0, Processors #1 #2 #3 Ok.
Brought up 4 CPUs
V2: little changes based on Joe Perches' feedback
V3: printk cleanup based on Ingo's feedback; checkpatch fix
V4: keep printk as one long line
V5: Ingo fix ups
Reported-and-tested-by: Nathan Zimmer <nzimmer@sgi.com>
Signed-off-by: Don Zickus <dzickus@redhat.com>
Cc: nzimmer@sgi.com
Cc: joe@perches.com
Link: http://lkml.kernel.org/r/1339594548-17227-1-git-send-email-dzickus@redhat.com
Signed-off-by: Ingo Molnar <mingo@kernel.org>
2012-06-13 21:35:48 +08:00
|
|
|
/* only print for cpu0 or different than cpu0 */
|
|
|
|
if (cpu == 0 || cpu0_err)
|
|
|
|
pr_info("enabled on all CPUs, permanently consumes one hw-PMU counter.\n");
|
2010-05-08 05:11:44 +08:00
|
|
|
goto out_save;
|
|
|
|
}
|
|
|
|
|
watchdog: Quiet down the boot messages
A bunch of bugzillas have complained how noisy the nmi_watchdog
is during boot-up especially with its expected failure cases
(like virt and bios resource contention).
This is my attempt to quiet them down and keep it less confusing
for the end user. What I did is print the message for cpu0 and
save it for future comparisons. If future cpus have an
identical message as cpu0, then don't print the redundant info.
However, if a future cpu has a different message, happily print
that loudly.
Before the change, you would see something like:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog enabled, takes one hw-pmu counter.
Booting Node 0, Processors #1
NMI watchdog enabled, takes one hw-pmu counter.
#2
NMI watchdog enabled, takes one hw-pmu counter.
#3 Ok.
NMI watchdog enabled, takes one hw-pmu counter.
Brought up 4 CPUs
Total of 4 processors activated (22607.24 BogoMIPS).
After the change, it is simplified to:
..TIMER: vector=0x30 apic1=0 pin1=2 apic2=-1 pin2=-1
CPU0: Intel(R) Core(TM)2 Quad CPU Q9550 @ 2.83GHz stepping 0a
Performance Events: PEBS fmt0+, Core2 events, Intel PMU driver.
... version: 2
... bit width: 40
... generic registers: 2
... value mask: 000000ffffffffff
... max period: 000000007fffffff
... fixed-purpose events: 3
... event mask: 0000000700000003
NMI watchdog: enabled on all CPUs, permanently consumes one hw-PMU counter.
Booting Node 0, Processors #1 #2 #3 Ok.
Brought up 4 CPUs
V2: little changes based on Joe Perches' feedback
V3: printk cleanup based on Ingo's feedback; checkpatch fix
V4: keep printk as one long line
V5: Ingo fix ups
Reported-and-tested-by: Nathan Zimmer <nzimmer@sgi.com>
Signed-off-by: Don Zickus <dzickus@redhat.com>
Cc: nzimmer@sgi.com
Cc: joe@perches.com
Link: http://lkml.kernel.org/r/1339594548-17227-1-git-send-email-dzickus@redhat.com
Signed-off-by: Ingo Molnar <mingo@kernel.org>
2012-06-13 21:35:48 +08:00
|
|
|
/* skip displaying the same error again */
|
|
|
|
if (cpu > 0 && (PTR_ERR(event) == cpu0_err))
|
|
|
|
return PTR_ERR(event);
|
2011-02-10 03:02:33 +08:00
|
|
|
|
|
|
|
/* vary the KERN level based on the returned errno */
|
|
|
|
if (PTR_ERR(event) == -EOPNOTSUPP)
|
2012-03-24 06:01:55 +08:00
|
|
|
pr_info("disabled (cpu%i): not supported (no LAPIC?)\n", cpu);
|
2011-02-10 03:02:33 +08:00
|
|
|
else if (PTR_ERR(event) == -ENOENT)
|
2012-03-24 06:01:55 +08:00
|
|
|
pr_warning("disabled (cpu%i): hardware events not enabled\n",
|
|
|
|
cpu);
|
2011-02-10 03:02:33 +08:00
|
|
|
else
|
2012-03-24 06:01:55 +08:00
|
|
|
pr_err("disabled (cpu%i): unable to create perf event: %ld\n",
|
|
|
|
cpu, PTR_ERR(event));
|
2010-09-01 11:00:08 +08:00
|
|
|
return PTR_ERR(event);
|
2010-05-08 05:11:44 +08:00
|
|
|
|
|
|
|
/* success path */
|
|
|
|
out_save:
|
|
|
|
per_cpu(watchdog_ev, cpu) = event;
|
|
|
|
out_enable:
|
|
|
|
perf_event_enable(per_cpu(watchdog_ev, cpu));
|
|
|
|
out:
|
|
|
|
return 0;
|
|
|
|
}
|
|
|
|
|
2012-07-16 18:42:38 +08:00
|
|
|
static void watchdog_nmi_disable(unsigned int cpu)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
|
|
|
struct perf_event *event = per_cpu(watchdog_ev, cpu);
|
|
|
|
|
|
|
|
if (event) {
|
|
|
|
perf_event_disable(event);
|
|
|
|
per_cpu(watchdog_ev, cpu) = NULL;
|
|
|
|
|
|
|
|
/* should be in cleanup, but blocks oprofile */
|
|
|
|
perf_event_release_kernel(event);
|
|
|
|
}
|
|
|
|
return;
|
|
|
|
}
|
|
|
|
#else
|
2012-07-16 18:42:38 +08:00
|
|
|
static int watchdog_nmi_enable(unsigned int cpu) { return 0; }
|
|
|
|
static void watchdog_nmi_disable(unsigned int cpu) { return; }
|
2010-05-16 05:15:20 +08:00
|
|
|
#endif /* CONFIG_HARDLOCKUP_DETECTOR */
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2013-06-06 21:42:53 +08:00
|
|
|
static struct smp_hotplug_thread watchdog_threads = {
|
|
|
|
.store = &softlockup_watchdog,
|
|
|
|
.thread_should_run = watchdog_should_run,
|
|
|
|
.thread_fn = watchdog,
|
|
|
|
.thread_comm = "watchdog/%u",
|
|
|
|
.setup = watchdog_enable,
|
|
|
|
.cleanup = watchdog_cleanup,
|
|
|
|
.park = watchdog_disable,
|
|
|
|
.unpark = watchdog_enable,
|
|
|
|
};
|
|
|
|
|
|
|
|
static int watchdog_enable_all_cpus(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2013-06-06 21:42:53 +08:00
|
|
|
int err = 0;
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2013-05-20 02:45:15 +08:00
|
|
|
if (!watchdog_running) {
|
2013-06-06 21:42:53 +08:00
|
|
|
err = smpboot_register_percpu_thread(&watchdog_threads);
|
|
|
|
if (err)
|
|
|
|
pr_err("Failed to create watchdog threads, disabled\n");
|
|
|
|
else
|
2013-05-20 02:45:15 +08:00
|
|
|
watchdog_running = 1;
|
2012-07-16 18:42:38 +08:00
|
|
|
}
|
2013-06-06 21:42:53 +08:00
|
|
|
|
|
|
|
return err;
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
|
2013-06-06 21:42:53 +08:00
|
|
|
/* prepare/enable/disable routines */
|
|
|
|
/* sysctl functions */
|
|
|
|
#ifdef CONFIG_SYSCTL
|
2010-05-08 05:11:44 +08:00
|
|
|
static void watchdog_disable_all_cpus(void)
|
|
|
|
{
|
2013-05-20 02:45:15 +08:00
|
|
|
if (watchdog_running) {
|
|
|
|
watchdog_running = 0;
|
2013-06-06 21:42:53 +08:00
|
|
|
smpboot_unregister_percpu_thread(&watchdog_threads);
|
2012-07-16 18:42:38 +08:00
|
|
|
}
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
2011-05-23 13:10:22 +08:00
|
|
|
* proc handler for /proc/sys/kernel/nmi_watchdog,watchdog_thresh
|
2010-05-08 05:11:44 +08:00
|
|
|
*/
|
|
|
|
|
2011-05-23 13:10:22 +08:00
|
|
|
int proc_dowatchdog(struct ctl_table *table, int write,
|
|
|
|
void __user *buffer, size_t *lenp, loff_t *ppos)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2013-06-06 21:42:53 +08:00
|
|
|
int err, old_thresh, old_enabled;
|
2010-05-08 05:11:44 +08:00
|
|
|
|
2013-06-06 21:42:53 +08:00
|
|
|
old_thresh = ACCESS_ONCE(watchdog_thresh);
|
2013-05-20 02:45:15 +08:00
|
|
|
old_enabled = ACCESS_ONCE(watchdog_user_enabled);
|
2012-07-16 18:42:38 +08:00
|
|
|
|
2013-06-06 21:42:53 +08:00
|
|
|
err = proc_dointvec_minmax(table, write, buffer, lenp, ppos);
|
|
|
|
if (err || !write)
|
|
|
|
return err;
|
2011-05-23 13:10:21 +08:00
|
|
|
|
2012-12-18 07:59:50 +08:00
|
|
|
set_sample_period();
|
2013-03-13 02:44:08 +08:00
|
|
|
/*
|
|
|
|
* Watchdog threads shouldn't be enabled if they are
|
2013-05-20 02:45:15 +08:00
|
|
|
* disabled. The 'watchdog_running' variable check in
|
2013-03-13 02:44:08 +08:00
|
|
|
* watchdog_*_all_cpus() function takes care of this.
|
|
|
|
*/
|
2013-05-20 02:45:15 +08:00
|
|
|
if (watchdog_user_enabled && watchdog_thresh)
|
2013-06-06 21:42:53 +08:00
|
|
|
err = watchdog_enable_all_cpus();
|
2011-05-23 13:10:21 +08:00
|
|
|
else
|
|
|
|
watchdog_disable_all_cpus();
|
|
|
|
|
2013-06-06 21:42:53 +08:00
|
|
|
/* Restore old values on failure */
|
|
|
|
if (err) {
|
|
|
|
watchdog_thresh = old_thresh;
|
2013-05-20 02:45:15 +08:00
|
|
|
watchdog_user_enabled = old_enabled;
|
2013-06-06 21:42:53 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
return err;
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|
|
|
|
#endif /* CONFIG_SYSCTL */
|
|
|
|
|
2010-11-26 01:38:29 +08:00
|
|
|
void __init lockup_detector_init(void)
|
2010-05-08 05:11:44 +08:00
|
|
|
{
|
2012-12-18 07:59:50 +08:00
|
|
|
set_sample_period();
|
2013-06-06 21:42:53 +08:00
|
|
|
|
2013-05-20 02:45:15 +08:00
|
|
|
if (watchdog_user_enabled)
|
2013-06-06 21:42:53 +08:00
|
|
|
watchdog_enable_all_cpus();
|
2010-05-08 05:11:44 +08:00
|
|
|
}
|