error.c 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485
  1. // SPDX-License-Identifier: GPL-2.0
  2. #include "bcachefs.h"
  3. #include "btree_iter.h"
  4. #include "error.h"
  5. #include "journal.h"
  6. #include "recovery_passes.h"
  7. #include "super.h"
  8. #include "thread_with_file.h"
  9. #define FSCK_ERR_RATELIMIT_NR 10
  10. bool bch2_inconsistent_error(struct bch_fs *c)
  11. {
  12. set_bit(BCH_FS_error, &c->flags);
  13. switch (c->opts.errors) {
  14. case BCH_ON_ERROR_continue:
  15. return false;
  16. case BCH_ON_ERROR_fix_safe:
  17. case BCH_ON_ERROR_ro:
  18. if (bch2_fs_emergency_read_only(c))
  19. bch_err(c, "inconsistency detected - emergency read only at journal seq %llu",
  20. journal_cur_seq(&c->journal));
  21. return true;
  22. case BCH_ON_ERROR_panic:
  23. panic(bch2_fmt(c, "panic after error"));
  24. return true;
  25. default:
  26. BUG();
  27. }
  28. }
  29. int bch2_topology_error(struct bch_fs *c)
  30. {
  31. set_bit(BCH_FS_topology_error, &c->flags);
  32. if (!test_bit(BCH_FS_fsck_running, &c->flags)) {
  33. bch2_inconsistent_error(c);
  34. return -BCH_ERR_btree_need_topology_repair;
  35. } else {
  36. return bch2_run_explicit_recovery_pass(c, BCH_RECOVERY_PASS_check_topology) ?:
  37. -BCH_ERR_btree_node_read_validate_error;
  38. }
  39. }
  40. void bch2_fatal_error(struct bch_fs *c)
  41. {
  42. if (bch2_fs_emergency_read_only(c))
  43. bch_err(c, "fatal error - emergency read only");
  44. }
  45. void bch2_io_error_work(struct work_struct *work)
  46. {
  47. struct bch_dev *ca = container_of(work, struct bch_dev, io_error_work);
  48. struct bch_fs *c = ca->fs;
  49. bool dev;
  50. down_write(&c->state_lock);
  51. dev = bch2_dev_state_allowed(c, ca, BCH_MEMBER_STATE_ro,
  52. BCH_FORCE_IF_DEGRADED);
  53. if (dev
  54. ? __bch2_dev_set_state(c, ca, BCH_MEMBER_STATE_ro,
  55. BCH_FORCE_IF_DEGRADED)
  56. : bch2_fs_emergency_read_only(c))
  57. bch_err(ca,
  58. "too many IO errors, setting %s RO",
  59. dev ? "device" : "filesystem");
  60. up_write(&c->state_lock);
  61. }
  62. void bch2_io_error(struct bch_dev *ca, enum bch_member_error_type type)
  63. {
  64. atomic64_inc(&ca->errors[type]);
  65. //queue_work(system_long_wq, &ca->io_error_work);
  66. }
  67. enum ask_yn {
  68. YN_NO,
  69. YN_YES,
  70. YN_ALLNO,
  71. YN_ALLYES,
  72. };
  73. static enum ask_yn parse_yn_response(char *buf)
  74. {
  75. buf = strim(buf);
  76. if (strlen(buf) == 1)
  77. switch (buf[0]) {
  78. case 'n':
  79. return YN_NO;
  80. case 'y':
  81. return YN_YES;
  82. case 'N':
  83. return YN_ALLNO;
  84. case 'Y':
  85. return YN_ALLYES;
  86. }
  87. return -1;
  88. }
  89. #ifdef __KERNEL__
  90. static enum ask_yn bch2_fsck_ask_yn(struct bch_fs *c, struct btree_trans *trans)
  91. {
  92. struct stdio_redirect *stdio = c->stdio;
  93. if (c->stdio_filter && c->stdio_filter != current)
  94. stdio = NULL;
  95. if (!stdio)
  96. return YN_NO;
  97. if (trans)
  98. bch2_trans_unlock(trans);
  99. unsigned long unlock_long_at = trans ? jiffies + HZ * 2 : 0;
  100. darray_char line = {};
  101. int ret;
  102. do {
  103. unsigned long t;
  104. bch2_print(c, " (y,n, or Y,N for all errors of this type) ");
  105. rewait:
  106. t = unlock_long_at
  107. ? max_t(long, unlock_long_at - jiffies, 0)
  108. : MAX_SCHEDULE_TIMEOUT;
  109. int r = bch2_stdio_redirect_readline_timeout(stdio, &line, t);
  110. if (r == -ETIME) {
  111. bch2_trans_unlock_long(trans);
  112. unlock_long_at = 0;
  113. goto rewait;
  114. }
  115. if (r < 0) {
  116. ret = YN_NO;
  117. break;
  118. }
  119. darray_last(line) = '\0';
  120. } while ((ret = parse_yn_response(line.data)) < 0);
  121. darray_exit(&line);
  122. return ret;
  123. }
  124. #else
  125. #include "tools-util.h"
  126. static enum ask_yn bch2_fsck_ask_yn(struct bch_fs *c, struct btree_trans *trans)
  127. {
  128. char *buf = NULL;
  129. size_t buflen = 0;
  130. int ret;
  131. do {
  132. fputs(" (y,n, or Y,N for all errors of this type) ", stdout);
  133. fflush(stdout);
  134. if (getline(&buf, &buflen, stdin) < 0)
  135. die("error reading from standard input");
  136. } while ((ret = parse_yn_response(buf)) < 0);
  137. free(buf);
  138. return ret;
  139. }
  140. #endif
  141. static struct fsck_err_state *fsck_err_get(struct bch_fs *c, const char *fmt)
  142. {
  143. struct fsck_err_state *s;
  144. if (!test_bit(BCH_FS_fsck_running, &c->flags))
  145. return NULL;
  146. list_for_each_entry(s, &c->fsck_error_msgs, list)
  147. if (s->fmt == fmt) {
  148. /*
  149. * move it to the head of the list: repeated fsck errors
  150. * are common
  151. */
  152. list_move(&s->list, &c->fsck_error_msgs);
  153. return s;
  154. }
  155. s = kzalloc(sizeof(*s), GFP_NOFS);
  156. if (!s) {
  157. if (!c->fsck_alloc_msgs_err)
  158. bch_err(c, "kmalloc err, cannot ratelimit fsck errs");
  159. c->fsck_alloc_msgs_err = true;
  160. return NULL;
  161. }
  162. INIT_LIST_HEAD(&s->list);
  163. s->fmt = fmt;
  164. list_add(&s->list, &c->fsck_error_msgs);
  165. return s;
  166. }
  167. /* s/fix?/fixing/ s/recreate?/recreating/ */
  168. static void prt_actioning(struct printbuf *out, const char *action)
  169. {
  170. unsigned len = strlen(action);
  171. BUG_ON(action[len - 1] != '?');
  172. --len;
  173. if (action[len - 1] == 'e')
  174. --len;
  175. prt_bytes(out, action, len);
  176. prt_str(out, "ing");
  177. }
  178. static const u8 fsck_flags_extra[] = {
  179. #define x(t, n, flags) [BCH_FSCK_ERR_##t] = flags,
  180. BCH_SB_ERRS()
  181. #undef x
  182. };
  183. int __bch2_fsck_err(struct bch_fs *c,
  184. struct btree_trans *trans,
  185. enum bch_fsck_flags flags,
  186. enum bch_sb_error_id err,
  187. const char *fmt, ...)
  188. {
  189. struct fsck_err_state *s = NULL;
  190. va_list args;
  191. bool print = true, suppressing = false, inconsistent = false;
  192. struct printbuf buf = PRINTBUF, *out = &buf;
  193. int ret = -BCH_ERR_fsck_ignore;
  194. const char *action_orig = "fix?", *action = action_orig;
  195. might_sleep();
  196. if (!WARN_ON(err >= ARRAY_SIZE(fsck_flags_extra)))
  197. flags |= fsck_flags_extra[err];
  198. if (!c)
  199. c = trans->c;
  200. /*
  201. * Ugly: if there's a transaction in the current task it has to be
  202. * passed in to unlock if we prompt for user input.
  203. *
  204. * But, plumbing a transaction and transaction restarts into
  205. * bkey_validate() is problematic.
  206. *
  207. * So:
  208. * - make all bkey errors AUTOFIX, they're simple anyways (we just
  209. * delete the key)
  210. * - and we don't need to warn if we're not prompting
  211. */
  212. WARN_ON((flags & FSCK_CAN_FIX) &&
  213. !(flags & FSCK_AUTOFIX) &&
  214. !trans &&
  215. bch2_current_has_btree_trans(c));
  216. if ((flags & FSCK_CAN_FIX) &&
  217. test_bit(err, c->sb.errors_silent))
  218. return -BCH_ERR_fsck_fix;
  219. bch2_sb_error_count(c, err);
  220. va_start(args, fmt);
  221. prt_vprintf(out, fmt, args);
  222. va_end(args);
  223. /* Custom fix/continue/recreate/etc.? */
  224. if (out->buf[out->pos - 1] == '?') {
  225. const char *p = strrchr(out->buf, ',');
  226. if (p) {
  227. out->pos = p - out->buf;
  228. action = kstrdup(p + 2, GFP_KERNEL);
  229. if (!action) {
  230. ret = -ENOMEM;
  231. goto err;
  232. }
  233. }
  234. }
  235. mutex_lock(&c->fsck_error_msgs_lock);
  236. s = fsck_err_get(c, fmt);
  237. if (s) {
  238. /*
  239. * We may be called multiple times for the same error on
  240. * transaction restart - this memoizes instead of asking the user
  241. * multiple times for the same error:
  242. */
  243. if (s->last_msg && !strcmp(buf.buf, s->last_msg)) {
  244. ret = s->ret;
  245. mutex_unlock(&c->fsck_error_msgs_lock);
  246. goto err;
  247. }
  248. kfree(s->last_msg);
  249. s->last_msg = kstrdup(buf.buf, GFP_KERNEL);
  250. if (!s->last_msg) {
  251. mutex_unlock(&c->fsck_error_msgs_lock);
  252. ret = -ENOMEM;
  253. goto err;
  254. }
  255. if (c->opts.ratelimit_errors &&
  256. !(flags & FSCK_NO_RATELIMIT) &&
  257. s->nr >= FSCK_ERR_RATELIMIT_NR) {
  258. if (s->nr == FSCK_ERR_RATELIMIT_NR)
  259. suppressing = true;
  260. else
  261. print = false;
  262. }
  263. s->nr++;
  264. }
  265. #ifdef BCACHEFS_LOG_PREFIX
  266. if (!strncmp(fmt, "bcachefs:", 9))
  267. prt_printf(out, bch2_log_msg(c, ""));
  268. #endif
  269. if ((flags & FSCK_CAN_FIX) &&
  270. (flags & FSCK_AUTOFIX) &&
  271. (c->opts.errors == BCH_ON_ERROR_continue ||
  272. c->opts.errors == BCH_ON_ERROR_fix_safe)) {
  273. prt_str(out, ", ");
  274. prt_actioning(out, action);
  275. ret = -BCH_ERR_fsck_fix;
  276. } else if (!test_bit(BCH_FS_fsck_running, &c->flags)) {
  277. if (c->opts.errors != BCH_ON_ERROR_continue ||
  278. !(flags & (FSCK_CAN_FIX|FSCK_CAN_IGNORE))) {
  279. prt_str(out, ", shutting down");
  280. inconsistent = true;
  281. ret = -BCH_ERR_fsck_errors_not_fixed;
  282. } else if (flags & FSCK_CAN_FIX) {
  283. prt_str(out, ", ");
  284. prt_actioning(out, action);
  285. ret = -BCH_ERR_fsck_fix;
  286. } else {
  287. prt_str(out, ", continuing");
  288. ret = -BCH_ERR_fsck_ignore;
  289. }
  290. } else if (c->opts.fix_errors == FSCK_FIX_exit) {
  291. prt_str(out, ", exiting");
  292. ret = -BCH_ERR_fsck_errors_not_fixed;
  293. } else if (flags & FSCK_CAN_FIX) {
  294. int fix = s && s->fix
  295. ? s->fix
  296. : c->opts.fix_errors;
  297. if (fix == FSCK_FIX_ask) {
  298. prt_str(out, ", ");
  299. prt_str(out, action);
  300. if (bch2_fs_stdio_redirect(c))
  301. bch2_print(c, "%s", out->buf);
  302. else
  303. bch2_print_string_as_lines(KERN_ERR, out->buf);
  304. print = false;
  305. int ask = bch2_fsck_ask_yn(c, trans);
  306. if (trans) {
  307. ret = bch2_trans_relock(trans);
  308. if (ret) {
  309. mutex_unlock(&c->fsck_error_msgs_lock);
  310. goto err;
  311. }
  312. }
  313. if (ask >= YN_ALLNO && s)
  314. s->fix = ask == YN_ALLNO
  315. ? FSCK_FIX_no
  316. : FSCK_FIX_yes;
  317. ret = ask & 1
  318. ? -BCH_ERR_fsck_fix
  319. : -BCH_ERR_fsck_ignore;
  320. } else if (fix == FSCK_FIX_yes ||
  321. (c->opts.nochanges &&
  322. !(flags & FSCK_CAN_IGNORE))) {
  323. prt_str(out, ", ");
  324. prt_actioning(out, action);
  325. ret = -BCH_ERR_fsck_fix;
  326. } else {
  327. prt_str(out, ", not ");
  328. prt_actioning(out, action);
  329. }
  330. } else if (flags & FSCK_NEED_FSCK) {
  331. prt_str(out, " (run fsck to correct)");
  332. } else {
  333. prt_str(out, " (repair unimplemented)");
  334. }
  335. if (ret == -BCH_ERR_fsck_ignore &&
  336. (c->opts.fix_errors == FSCK_FIX_exit ||
  337. !(flags & FSCK_CAN_IGNORE)))
  338. ret = -BCH_ERR_fsck_errors_not_fixed;
  339. bool exiting =
  340. test_bit(BCH_FS_fsck_running, &c->flags) &&
  341. (ret != -BCH_ERR_fsck_fix &&
  342. ret != -BCH_ERR_fsck_ignore);
  343. if (exiting)
  344. print = true;
  345. if (print) {
  346. if (bch2_fs_stdio_redirect(c))
  347. bch2_print(c, "%s\n", out->buf);
  348. else
  349. bch2_print_string_as_lines(KERN_ERR, out->buf);
  350. }
  351. if (exiting)
  352. bch_err(c, "Unable to continue, halting");
  353. else if (suppressing)
  354. bch_err(c, "Ratelimiting new instances of previous error");
  355. if (s)
  356. s->ret = ret;
  357. mutex_unlock(&c->fsck_error_msgs_lock);
  358. if (inconsistent)
  359. bch2_inconsistent_error(c);
  360. if (ret == -BCH_ERR_fsck_fix) {
  361. set_bit(BCH_FS_errors_fixed, &c->flags);
  362. } else {
  363. set_bit(BCH_FS_errors_not_fixed, &c->flags);
  364. set_bit(BCH_FS_error, &c->flags);
  365. }
  366. err:
  367. if (action != action_orig)
  368. kfree(action);
  369. printbuf_exit(&buf);
  370. return ret;
  371. }
  372. int __bch2_bkey_fsck_err(struct bch_fs *c,
  373. struct bkey_s_c k,
  374. enum bch_validate_flags validate_flags,
  375. enum bch_sb_error_id err,
  376. const char *fmt, ...)
  377. {
  378. if (validate_flags & BCH_VALIDATE_silent)
  379. return -BCH_ERR_fsck_delete_bkey;
  380. unsigned fsck_flags = 0;
  381. if (!(validate_flags & (BCH_VALIDATE_write|BCH_VALIDATE_commit)))
  382. fsck_flags |= FSCK_AUTOFIX|FSCK_CAN_FIX;
  383. struct printbuf buf = PRINTBUF;
  384. va_list args;
  385. prt_str(&buf, "invalid bkey ");
  386. bch2_bkey_val_to_text(&buf, c, k);
  387. prt_str(&buf, "\n ");
  388. va_start(args, fmt);
  389. prt_vprintf(&buf, fmt, args);
  390. va_end(args);
  391. prt_str(&buf, ": delete?");
  392. int ret = __bch2_fsck_err(c, NULL, fsck_flags, err, "%s", buf.buf);
  393. printbuf_exit(&buf);
  394. return ret;
  395. }
  396. void bch2_flush_fsck_errs(struct bch_fs *c)
  397. {
  398. struct fsck_err_state *s, *n;
  399. mutex_lock(&c->fsck_error_msgs_lock);
  400. list_for_each_entry_safe(s, n, &c->fsck_error_msgs, list) {
  401. if (s->ratelimited && s->last_msg)
  402. bch_err(c, "Saw %llu errors like:\n %s", s->nr, s->last_msg);
  403. list_del(&s->list);
  404. kfree(s->last_msg);
  405. kfree(s);
  406. }
  407. mutex_unlock(&c->fsck_error_msgs_lock);
  408. }