hung_task.c 9.8 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401
  1. // SPDX-License-Identifier: GPL-2.0-only
  2. /*
  3. * Detect Hung Task
  4. *
  5. * kernel/hung_task.c - kernel thread for detecting tasks stuck in D state
  6. *
  7. */
  8. #include <linux/mm.h>
  9. #include <linux/cpu.h>
  10. #include <linux/nmi.h>
  11. #include <linux/init.h>
  12. #include <linux/delay.h>
  13. #include <linux/freezer.h>
  14. #include <linux/kthread.h>
  15. #include <linux/lockdep.h>
  16. #include <linux/export.h>
  17. #include <linux/panic_notifier.h>
  18. #include <linux/sysctl.h>
  19. #include <linux/suspend.h>
  20. #include <linux/utsname.h>
  21. #include <linux/sched/signal.h>
  22. #include <linux/sched/debug.h>
  23. #include <linux/sched/sysctl.h>
  24. #include <trace/events/sched.h>
  25. /*
  26. * The number of tasks checked:
  27. */
  28. static int __read_mostly sysctl_hung_task_check_count = PID_MAX_LIMIT;
  29. /*
  30. * Limit number of tasks checked in a batch.
  31. *
  32. * This value controls the preemptibility of khungtaskd since preemption
  33. * is disabled during the critical section. It also controls the size of
  34. * the RCU grace period. So it needs to be upper-bound.
  35. */
  36. #define HUNG_TASK_LOCK_BREAK (HZ / 10)
  37. /*
  38. * Zero means infinite timeout - no checking done:
  39. */
  40. unsigned long __read_mostly sysctl_hung_task_timeout_secs = CONFIG_DEFAULT_HUNG_TASK_TIMEOUT;
  41. EXPORT_SYMBOL_GPL(sysctl_hung_task_timeout_secs);
  42. /*
  43. * Zero (default value) means use sysctl_hung_task_timeout_secs:
  44. */
  45. static unsigned long __read_mostly sysctl_hung_task_check_interval_secs;
  46. static int __read_mostly sysctl_hung_task_warnings = 10;
  47. static int __read_mostly did_panic;
  48. static bool hung_task_show_lock;
  49. static bool hung_task_call_panic;
  50. static bool hung_task_show_all_bt;
  51. static struct task_struct *watchdog_task;
  52. #ifdef CONFIG_SMP
  53. /*
  54. * Should we dump all CPUs backtraces in a hung task event?
  55. * Defaults to 0, can be changed via sysctl.
  56. */
  57. static unsigned int __read_mostly sysctl_hung_task_all_cpu_backtrace;
  58. #else
  59. #define sysctl_hung_task_all_cpu_backtrace 0
  60. #endif /* CONFIG_SMP */
  61. /*
  62. * Should we panic (and reboot, if panic_timeout= is set) when a
  63. * hung task is detected:
  64. */
  65. static unsigned int __read_mostly sysctl_hung_task_panic =
  66. IS_ENABLED(CONFIG_BOOTPARAM_HUNG_TASK_PANIC);
  67. static int
  68. hung_task_panic(struct notifier_block *this, unsigned long event, void *ptr)
  69. {
  70. did_panic = 1;
  71. return NOTIFY_DONE;
  72. }
  73. static struct notifier_block panic_block = {
  74. .notifier_call = hung_task_panic,
  75. };
  76. static void check_hung_task(struct task_struct *t, unsigned long timeout)
  77. {
  78. unsigned long switch_count = t->nvcsw + t->nivcsw;
  79. /*
  80. * Ensure the task is not frozen.
  81. * Also, skip vfork and any other user process that freezer should skip.
  82. */
  83. if (unlikely(READ_ONCE(t->__state) & TASK_FROZEN))
  84. return;
  85. /*
  86. * When a freshly created task is scheduled once, changes its state to
  87. * TASK_UNINTERRUPTIBLE without having ever been switched out once, it
  88. * musn't be checked.
  89. */
  90. if (unlikely(!switch_count))
  91. return;
  92. if (switch_count != t->last_switch_count) {
  93. t->last_switch_count = switch_count;
  94. t->last_switch_time = jiffies;
  95. return;
  96. }
  97. if (time_is_after_jiffies(t->last_switch_time + timeout * HZ))
  98. return;
  99. trace_sched_process_hang(t);
  100. if (sysctl_hung_task_panic) {
  101. console_verbose();
  102. hung_task_show_lock = true;
  103. hung_task_call_panic = true;
  104. }
  105. /*
  106. * Ok, the task did not get scheduled for more than 2 minutes,
  107. * complain:
  108. */
  109. if (sysctl_hung_task_warnings || hung_task_call_panic) {
  110. if (sysctl_hung_task_warnings > 0)
  111. sysctl_hung_task_warnings--;
  112. pr_err("INFO: task %s:%d blocked for more than %ld seconds.\n",
  113. t->comm, t->pid, (jiffies - t->last_switch_time) / HZ);
  114. pr_err(" %s %s %.*s\n",
  115. print_tainted(), init_utsname()->release,
  116. (int)strcspn(init_utsname()->version, " "),
  117. init_utsname()->version);
  118. pr_err("\"echo 0 > /proc/sys/kernel/hung_task_timeout_secs\""
  119. " disables this message.\n");
  120. sched_show_task(t);
  121. hung_task_show_lock = true;
  122. if (sysctl_hung_task_all_cpu_backtrace)
  123. hung_task_show_all_bt = true;
  124. if (!sysctl_hung_task_warnings)
  125. pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n");
  126. }
  127. touch_nmi_watchdog();
  128. }
  129. /*
  130. * To avoid extending the RCU grace period for an unbounded amount of time,
  131. * periodically exit the critical section and enter a new one.
  132. *
  133. * For preemptible RCU it is sufficient to call rcu_read_unlock in order
  134. * to exit the grace period. For classic RCU, a reschedule is required.
  135. */
  136. static bool rcu_lock_break(struct task_struct *g, struct task_struct *t)
  137. {
  138. bool can_cont;
  139. get_task_struct(g);
  140. get_task_struct(t);
  141. rcu_read_unlock();
  142. cond_resched();
  143. rcu_read_lock();
  144. can_cont = pid_alive(g) && pid_alive(t);
  145. put_task_struct(t);
  146. put_task_struct(g);
  147. return can_cont;
  148. }
  149. /*
  150. * Check whether a TASK_UNINTERRUPTIBLE does not get woken up for
  151. * a really long time (120 seconds). If that happens, print out
  152. * a warning.
  153. */
  154. static void check_hung_uninterruptible_tasks(unsigned long timeout)
  155. {
  156. int max_count = sysctl_hung_task_check_count;
  157. unsigned long last_break = jiffies;
  158. struct task_struct *g, *t;
  159. /*
  160. * If the system crashed already then all bets are off,
  161. * do not report extra hung tasks:
  162. */
  163. if (test_taint(TAINT_DIE) || did_panic)
  164. return;
  165. hung_task_show_lock = false;
  166. rcu_read_lock();
  167. for_each_process_thread(g, t) {
  168. unsigned int state;
  169. if (!max_count--)
  170. goto unlock;
  171. if (time_after(jiffies, last_break + HUNG_TASK_LOCK_BREAK)) {
  172. if (!rcu_lock_break(g, t))
  173. goto unlock;
  174. last_break = jiffies;
  175. }
  176. /*
  177. * skip the TASK_KILLABLE tasks -- these can be killed
  178. * skip the TASK_IDLE tasks -- those are genuinely idle
  179. */
  180. state = READ_ONCE(t->__state);
  181. if ((state & TASK_UNINTERRUPTIBLE) &&
  182. !(state & TASK_WAKEKILL) &&
  183. !(state & TASK_NOLOAD))
  184. check_hung_task(t, timeout);
  185. }
  186. unlock:
  187. rcu_read_unlock();
  188. if (hung_task_show_lock)
  189. debug_show_all_locks();
  190. if (hung_task_show_all_bt) {
  191. hung_task_show_all_bt = false;
  192. trigger_all_cpu_backtrace();
  193. }
  194. if (hung_task_call_panic)
  195. panic("hung_task: blocked tasks");
  196. }
  197. static long hung_timeout_jiffies(unsigned long last_checked,
  198. unsigned long timeout)
  199. {
  200. /* timeout of 0 will disable the watchdog */
  201. return timeout ? last_checked - jiffies + timeout * HZ :
  202. MAX_SCHEDULE_TIMEOUT;
  203. }
  204. #ifdef CONFIG_SYSCTL
  205. /*
  206. * Process updating of timeout sysctl
  207. */
  208. static int proc_dohung_task_timeout_secs(const struct ctl_table *table, int write,
  209. void *buffer,
  210. size_t *lenp, loff_t *ppos)
  211. {
  212. int ret;
  213. ret = proc_doulongvec_minmax(table, write, buffer, lenp, ppos);
  214. if (ret || !write)
  215. goto out;
  216. wake_up_process(watchdog_task);
  217. out:
  218. return ret;
  219. }
  220. /*
  221. * This is needed for proc_doulongvec_minmax of sysctl_hung_task_timeout_secs
  222. * and hung_task_check_interval_secs
  223. */
  224. static const unsigned long hung_task_timeout_max = (LONG_MAX / HZ);
  225. static struct ctl_table hung_task_sysctls[] = {
  226. #ifdef CONFIG_SMP
  227. {
  228. .procname = "hung_task_all_cpu_backtrace",
  229. .data = &sysctl_hung_task_all_cpu_backtrace,
  230. .maxlen = sizeof(int),
  231. .mode = 0644,
  232. .proc_handler = proc_dointvec_minmax,
  233. .extra1 = SYSCTL_ZERO,
  234. .extra2 = SYSCTL_ONE,
  235. },
  236. #endif /* CONFIG_SMP */
  237. {
  238. .procname = "hung_task_panic",
  239. .data = &sysctl_hung_task_panic,
  240. .maxlen = sizeof(int),
  241. .mode = 0644,
  242. .proc_handler = proc_dointvec_minmax,
  243. .extra1 = SYSCTL_ZERO,
  244. .extra2 = SYSCTL_ONE,
  245. },
  246. {
  247. .procname = "hung_task_check_count",
  248. .data = &sysctl_hung_task_check_count,
  249. .maxlen = sizeof(int),
  250. .mode = 0644,
  251. .proc_handler = proc_dointvec_minmax,
  252. .extra1 = SYSCTL_ZERO,
  253. },
  254. {
  255. .procname = "hung_task_timeout_secs",
  256. .data = &sysctl_hung_task_timeout_secs,
  257. .maxlen = sizeof(unsigned long),
  258. .mode = 0644,
  259. .proc_handler = proc_dohung_task_timeout_secs,
  260. .extra2 = (void *)&hung_task_timeout_max,
  261. },
  262. {
  263. .procname = "hung_task_check_interval_secs",
  264. .data = &sysctl_hung_task_check_interval_secs,
  265. .maxlen = sizeof(unsigned long),
  266. .mode = 0644,
  267. .proc_handler = proc_dohung_task_timeout_secs,
  268. .extra2 = (void *)&hung_task_timeout_max,
  269. },
  270. {
  271. .procname = "hung_task_warnings",
  272. .data = &sysctl_hung_task_warnings,
  273. .maxlen = sizeof(int),
  274. .mode = 0644,
  275. .proc_handler = proc_dointvec_minmax,
  276. .extra1 = SYSCTL_NEG_ONE,
  277. },
  278. };
  279. static void __init hung_task_sysctl_init(void)
  280. {
  281. register_sysctl_init("kernel", hung_task_sysctls);
  282. }
  283. #else
  284. #define hung_task_sysctl_init() do { } while (0)
  285. #endif /* CONFIG_SYSCTL */
  286. static atomic_t reset_hung_task = ATOMIC_INIT(0);
  287. void reset_hung_task_detector(void)
  288. {
  289. atomic_set(&reset_hung_task, 1);
  290. }
  291. EXPORT_SYMBOL_GPL(reset_hung_task_detector);
  292. static bool hung_detector_suspended;
  293. static int hungtask_pm_notify(struct notifier_block *self,
  294. unsigned long action, void *hcpu)
  295. {
  296. switch (action) {
  297. case PM_SUSPEND_PREPARE:
  298. case PM_HIBERNATION_PREPARE:
  299. case PM_RESTORE_PREPARE:
  300. hung_detector_suspended = true;
  301. break;
  302. case PM_POST_SUSPEND:
  303. case PM_POST_HIBERNATION:
  304. case PM_POST_RESTORE:
  305. hung_detector_suspended = false;
  306. break;
  307. default:
  308. break;
  309. }
  310. return NOTIFY_OK;
  311. }
  312. /*
  313. * kthread which checks for tasks stuck in D state
  314. */
  315. static int watchdog(void *dummy)
  316. {
  317. unsigned long hung_last_checked = jiffies;
  318. set_user_nice(current, 0);
  319. for ( ; ; ) {
  320. unsigned long timeout = sysctl_hung_task_timeout_secs;
  321. unsigned long interval = sysctl_hung_task_check_interval_secs;
  322. long t;
  323. if (interval == 0)
  324. interval = timeout;
  325. interval = min_t(unsigned long, interval, timeout);
  326. t = hung_timeout_jiffies(hung_last_checked, interval);
  327. if (t <= 0) {
  328. if (!atomic_xchg(&reset_hung_task, 0) &&
  329. !hung_detector_suspended)
  330. check_hung_uninterruptible_tasks(timeout);
  331. hung_last_checked = jiffies;
  332. continue;
  333. }
  334. schedule_timeout_interruptible(t);
  335. }
  336. return 0;
  337. }
  338. static int __init hung_task_init(void)
  339. {
  340. atomic_notifier_chain_register(&panic_notifier_list, &panic_block);
  341. /* Disable hung task detector on suspend */
  342. pm_notifier(hungtask_pm_notify, 0);
  343. watchdog_task = kthread_run(watchdog, NULL, "khungtaskd");
  344. hung_task_sysctl_init();
  345. return 0;
  346. }
  347. subsys_initcall(hung_task_init);