alloc_foreground.c 47 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020102110221023102410251026102710281029103010311032103310341035103610371038103910401041104210431044104510461047104810491050105110521053105410551056105710581059106010611062106310641065106610671068106910701071107210731074107510761077107810791080108110821083108410851086108710881089109010911092109310941095109610971098109911001101110211031104110511061107110811091110111111121113111411151116111711181119112011211122112311241125112611271128112911301131113211331134113511361137113811391140114111421143114411451146114711481149115011511152115311541155115611571158115911601161116211631164116511661167116811691170117111721173117411751176117711781179118011811182118311841185118611871188118911901191119211931194119511961197119811991200120112021203120412051206120712081209121012111212121312141215121612171218121912201221122212231224122512261227122812291230123112321233123412351236123712381239124012411242124312441245124612471248124912501251125212531254125512561257125812591260126112621263126412651266126712681269127012711272127312741275127612771278127912801281128212831284128512861287128812891290129112921293129412951296129712981299130013011302130313041305130613071308130913101311131213131314131513161317131813191320132113221323132413251326132713281329133013311332133313341335133613371338133913401341134213431344134513461347134813491350135113521353135413551356135713581359136013611362136313641365136613671368136913701371137213731374137513761377137813791380138113821383138413851386138713881389139013911392139313941395139613971398139914001401140214031404140514061407140814091410141114121413141414151416141714181419142014211422142314241425142614271428142914301431143214331434143514361437143814391440144114421443144414451446144714481449145014511452145314541455145614571458145914601461146214631464146514661467146814691470147114721473147414751476147714781479148014811482148314841485148614871488148914901491149214931494149514961497149814991500150115021503150415051506150715081509151015111512151315141515151615171518151915201521152215231524152515261527152815291530153115321533153415351536153715381539154015411542154315441545154615471548154915501551155215531554155515561557155815591560156115621563156415651566156715681569157015711572157315741575157615771578157915801581158215831584158515861587158815891590159115921593159415951596159715981599160016011602160316041605160616071608160916101611161216131614161516161617161816191620162116221623162416251626162716281629163016311632163316341635163616371638163916401641164216431644164516461647164816491650165116521653165416551656165716581659166016611662166316641665166616671668166916701671167216731674167516761677167816791680168116821683168416851686168716881689169016911692169316941695169616971698169917001701170217031704170517061707170817091710171117121713171417151716171717181719172017211722172317241725172617271728172917301731173217331734173517361737173817391740174117421743174417451746174717481749175017511752175317541755175617571758175917601761176217631764176517661767176817691770177117721773177417751776177717781779178017811782178317841785178617871788178917901791179217931794179517961797179817991800180118021803180418051806180718081809181018111812181318141815181618171818
  1. // SPDX-License-Identifier: GPL-2.0
  2. /*
  3. * Copyright 2012 Google, Inc.
  4. *
  5. * Foreground allocator code: allocate buckets from freelist, and allocate in
  6. * sector granularity from writepoints.
  7. *
  8. * bch2_bucket_alloc() allocates a single bucket from a specific device.
  9. *
  10. * bch2_bucket_alloc_set() allocates one or more buckets from different devices
  11. * in a given filesystem.
  12. */
  13. #include "bcachefs.h"
  14. #include "alloc_background.h"
  15. #include "alloc_foreground.h"
  16. #include "backpointers.h"
  17. #include "btree_iter.h"
  18. #include "btree_update.h"
  19. #include "btree_gc.h"
  20. #include "buckets.h"
  21. #include "buckets_waiting_for_journal.h"
  22. #include "clock.h"
  23. #include "debug.h"
  24. #include "disk_groups.h"
  25. #include "ec.h"
  26. #include "error.h"
  27. #include "io_write.h"
  28. #include "journal.h"
  29. #include "movinggc.h"
  30. #include "nocow_locking.h"
  31. #include "trace.h"
  32. #include <linux/math64.h>
  33. #include <linux/rculist.h>
  34. #include <linux/rcupdate.h>
  35. static void bch2_trans_mutex_lock_norelock(struct btree_trans *trans,
  36. struct mutex *lock)
  37. {
  38. if (!mutex_trylock(lock)) {
  39. bch2_trans_unlock(trans);
  40. mutex_lock(lock);
  41. }
  42. }
  43. const char * const bch2_watermarks[] = {
  44. #define x(t) #t,
  45. BCH_WATERMARKS()
  46. #undef x
  47. NULL
  48. };
  49. /*
  50. * Open buckets represent a bucket that's currently being allocated from. They
  51. * serve two purposes:
  52. *
  53. * - They track buckets that have been partially allocated, allowing for
  54. * sub-bucket sized allocations - they're used by the sector allocator below
  55. *
  56. * - They provide a reference to the buckets they own that mark and sweep GC
  57. * can find, until the new allocation has a pointer to it inserted into the
  58. * btree
  59. *
  60. * When allocating some space with the sector allocator, the allocation comes
  61. * with a reference to an open bucket - the caller is required to put that
  62. * reference _after_ doing the index update that makes its allocation reachable.
  63. */
  64. void bch2_reset_alloc_cursors(struct bch_fs *c)
  65. {
  66. rcu_read_lock();
  67. for_each_member_device_rcu(c, ca, NULL)
  68. memset(ca->alloc_cursor, 0, sizeof(ca->alloc_cursor));
  69. rcu_read_unlock();
  70. }
  71. static void bch2_open_bucket_hash_add(struct bch_fs *c, struct open_bucket *ob)
  72. {
  73. open_bucket_idx_t idx = ob - c->open_buckets;
  74. open_bucket_idx_t *slot = open_bucket_hashslot(c, ob->dev, ob->bucket);
  75. ob->hash = *slot;
  76. *slot = idx;
  77. }
  78. static void bch2_open_bucket_hash_remove(struct bch_fs *c, struct open_bucket *ob)
  79. {
  80. open_bucket_idx_t idx = ob - c->open_buckets;
  81. open_bucket_idx_t *slot = open_bucket_hashslot(c, ob->dev, ob->bucket);
  82. while (*slot != idx) {
  83. BUG_ON(!*slot);
  84. slot = &c->open_buckets[*slot].hash;
  85. }
  86. *slot = ob->hash;
  87. ob->hash = 0;
  88. }
  89. void __bch2_open_bucket_put(struct bch_fs *c, struct open_bucket *ob)
  90. {
  91. struct bch_dev *ca = ob_dev(c, ob);
  92. if (ob->ec) {
  93. ec_stripe_new_put(c, ob->ec, STRIPE_REF_io);
  94. return;
  95. }
  96. percpu_down_read(&c->mark_lock);
  97. spin_lock(&ob->lock);
  98. ob->valid = false;
  99. ob->data_type = 0;
  100. spin_unlock(&ob->lock);
  101. percpu_up_read(&c->mark_lock);
  102. spin_lock(&c->freelist_lock);
  103. bch2_open_bucket_hash_remove(c, ob);
  104. ob->freelist = c->open_buckets_freelist;
  105. c->open_buckets_freelist = ob - c->open_buckets;
  106. c->open_buckets_nr_free++;
  107. ca->nr_open_buckets--;
  108. spin_unlock(&c->freelist_lock);
  109. closure_wake_up(&c->open_buckets_wait);
  110. }
  111. void bch2_open_bucket_write_error(struct bch_fs *c,
  112. struct open_buckets *obs,
  113. unsigned dev)
  114. {
  115. struct open_bucket *ob;
  116. unsigned i;
  117. open_bucket_for_each(c, obs, ob, i)
  118. if (ob->dev == dev && ob->ec)
  119. bch2_ec_bucket_cancel(c, ob);
  120. }
  121. static struct open_bucket *bch2_open_bucket_alloc(struct bch_fs *c)
  122. {
  123. struct open_bucket *ob;
  124. BUG_ON(!c->open_buckets_freelist || !c->open_buckets_nr_free);
  125. ob = c->open_buckets + c->open_buckets_freelist;
  126. c->open_buckets_freelist = ob->freelist;
  127. atomic_set(&ob->pin, 1);
  128. ob->data_type = 0;
  129. c->open_buckets_nr_free--;
  130. return ob;
  131. }
  132. static void open_bucket_free_unused(struct bch_fs *c, struct open_bucket *ob)
  133. {
  134. BUG_ON(c->open_buckets_partial_nr >=
  135. ARRAY_SIZE(c->open_buckets_partial));
  136. spin_lock(&c->freelist_lock);
  137. rcu_read_lock();
  138. bch2_dev_rcu(c, ob->dev)->nr_partial_buckets++;
  139. rcu_read_unlock();
  140. ob->on_partial_list = true;
  141. c->open_buckets_partial[c->open_buckets_partial_nr++] =
  142. ob - c->open_buckets;
  143. spin_unlock(&c->freelist_lock);
  144. closure_wake_up(&c->open_buckets_wait);
  145. closure_wake_up(&c->freelist_wait);
  146. }
  147. /* _only_ for allocating the journal on a new device: */
  148. long bch2_bucket_alloc_new_fs(struct bch_dev *ca)
  149. {
  150. while (ca->new_fs_bucket_idx < ca->mi.nbuckets) {
  151. u64 b = ca->new_fs_bucket_idx++;
  152. if (!is_superblock_bucket(ca, b) &&
  153. (!ca->buckets_nouse || !test_bit(b, ca->buckets_nouse)))
  154. return b;
  155. }
  156. return -1;
  157. }
  158. static inline unsigned open_buckets_reserved(enum bch_watermark watermark)
  159. {
  160. switch (watermark) {
  161. case BCH_WATERMARK_interior_updates:
  162. return 0;
  163. case BCH_WATERMARK_reclaim:
  164. return OPEN_BUCKETS_COUNT / 6;
  165. case BCH_WATERMARK_btree:
  166. case BCH_WATERMARK_btree_copygc:
  167. return OPEN_BUCKETS_COUNT / 4;
  168. case BCH_WATERMARK_copygc:
  169. return OPEN_BUCKETS_COUNT / 3;
  170. default:
  171. return OPEN_BUCKETS_COUNT / 2;
  172. }
  173. }
  174. static struct open_bucket *__try_alloc_bucket(struct bch_fs *c, struct bch_dev *ca,
  175. u64 bucket,
  176. enum bch_watermark watermark,
  177. const struct bch_alloc_v4 *a,
  178. struct bucket_alloc_state *s,
  179. struct closure *cl)
  180. {
  181. struct open_bucket *ob;
  182. if (unlikely(ca->buckets_nouse && test_bit(bucket, ca->buckets_nouse))) {
  183. s->skipped_nouse++;
  184. return NULL;
  185. }
  186. if (bch2_bucket_is_open(c, ca->dev_idx, bucket)) {
  187. s->skipped_open++;
  188. return NULL;
  189. }
  190. if (bch2_bucket_needs_journal_commit(&c->buckets_waiting_for_journal,
  191. c->journal.flushed_seq_ondisk, ca->dev_idx, bucket)) {
  192. s->skipped_need_journal_commit++;
  193. return NULL;
  194. }
  195. if (bch2_bucket_nocow_is_locked(&c->nocow_locks, POS(ca->dev_idx, bucket))) {
  196. s->skipped_nocow++;
  197. return NULL;
  198. }
  199. spin_lock(&c->freelist_lock);
  200. if (unlikely(c->open_buckets_nr_free <= open_buckets_reserved(watermark))) {
  201. if (cl)
  202. closure_wait(&c->open_buckets_wait, cl);
  203. track_event_change(&c->times[BCH_TIME_blocked_allocate_open_bucket], true);
  204. spin_unlock(&c->freelist_lock);
  205. return ERR_PTR(-BCH_ERR_open_buckets_empty);
  206. }
  207. /* Recheck under lock: */
  208. if (bch2_bucket_is_open(c, ca->dev_idx, bucket)) {
  209. spin_unlock(&c->freelist_lock);
  210. s->skipped_open++;
  211. return NULL;
  212. }
  213. ob = bch2_open_bucket_alloc(c);
  214. spin_lock(&ob->lock);
  215. ob->valid = true;
  216. ob->sectors_free = ca->mi.bucket_size;
  217. ob->dev = ca->dev_idx;
  218. ob->gen = a->gen;
  219. ob->bucket = bucket;
  220. spin_unlock(&ob->lock);
  221. ca->nr_open_buckets++;
  222. bch2_open_bucket_hash_add(c, ob);
  223. track_event_change(&c->times[BCH_TIME_blocked_allocate_open_bucket], false);
  224. track_event_change(&c->times[BCH_TIME_blocked_allocate], false);
  225. spin_unlock(&c->freelist_lock);
  226. return ob;
  227. }
  228. static struct open_bucket *try_alloc_bucket(struct btree_trans *trans, struct bch_dev *ca,
  229. enum bch_watermark watermark, u64 free_entry,
  230. struct bucket_alloc_state *s,
  231. struct bkey_s_c freespace_k,
  232. struct closure *cl)
  233. {
  234. struct bch_fs *c = trans->c;
  235. struct btree_iter iter = { NULL };
  236. struct bkey_s_c k;
  237. struct open_bucket *ob;
  238. struct bch_alloc_v4 a_convert;
  239. const struct bch_alloc_v4 *a;
  240. u64 b = free_entry & ~(~0ULL << 56);
  241. unsigned genbits = free_entry >> 56;
  242. struct printbuf buf = PRINTBUF;
  243. int ret;
  244. if (b < ca->mi.first_bucket || b >= ca->mi.nbuckets) {
  245. prt_printf(&buf, "freespace btree has bucket outside allowed range %u-%llu\n"
  246. " freespace key ",
  247. ca->mi.first_bucket, ca->mi.nbuckets);
  248. bch2_bkey_val_to_text(&buf, c, freespace_k);
  249. bch2_trans_inconsistent(trans, "%s", buf.buf);
  250. ob = ERR_PTR(-EIO);
  251. goto err;
  252. }
  253. k = bch2_bkey_get_iter(trans, &iter,
  254. BTREE_ID_alloc, POS(ca->dev_idx, b),
  255. BTREE_ITER_cached);
  256. ret = bkey_err(k);
  257. if (ret) {
  258. ob = ERR_PTR(ret);
  259. goto err;
  260. }
  261. a = bch2_alloc_to_v4(k, &a_convert);
  262. if (a->data_type != BCH_DATA_free) {
  263. if (c->curr_recovery_pass <= BCH_RECOVERY_PASS_check_alloc_info) {
  264. ob = NULL;
  265. goto err;
  266. }
  267. prt_printf(&buf, "non free bucket in freespace btree\n"
  268. " freespace key ");
  269. bch2_bkey_val_to_text(&buf, c, freespace_k);
  270. prt_printf(&buf, "\n ");
  271. bch2_bkey_val_to_text(&buf, c, k);
  272. bch2_trans_inconsistent(trans, "%s", buf.buf);
  273. ob = ERR_PTR(-EIO);
  274. goto err;
  275. }
  276. if (genbits != (alloc_freespace_genbits(*a) >> 56) &&
  277. c->curr_recovery_pass > BCH_RECOVERY_PASS_check_alloc_info) {
  278. prt_printf(&buf, "bucket in freespace btree with wrong genbits (got %u should be %llu)\n"
  279. " freespace key ",
  280. genbits, alloc_freespace_genbits(*a) >> 56);
  281. bch2_bkey_val_to_text(&buf, c, freespace_k);
  282. prt_printf(&buf, "\n ");
  283. bch2_bkey_val_to_text(&buf, c, k);
  284. bch2_trans_inconsistent(trans, "%s", buf.buf);
  285. ob = ERR_PTR(-EIO);
  286. goto err;
  287. }
  288. if (c->curr_recovery_pass <= BCH_RECOVERY_PASS_check_extents_to_backpointers) {
  289. struct bch_backpointer bp;
  290. struct bpos bp_pos = POS_MIN;
  291. ret = bch2_get_next_backpointer(trans, ca, POS(ca->dev_idx, b), -1,
  292. &bp_pos, &bp,
  293. BTREE_ITER_nopreserve);
  294. if (ret) {
  295. ob = ERR_PTR(ret);
  296. goto err;
  297. }
  298. if (!bkey_eq(bp_pos, POS_MAX)) {
  299. /*
  300. * Bucket may have data in it - we don't call
  301. * bc2h_trans_inconnsistent() because fsck hasn't
  302. * finished yet
  303. */
  304. ob = NULL;
  305. goto err;
  306. }
  307. }
  308. ob = __try_alloc_bucket(c, ca, b, watermark, a, s, cl);
  309. if (!ob)
  310. bch2_set_btree_iter_dontneed(&iter);
  311. err:
  312. if (iter.path)
  313. bch2_set_btree_iter_dontneed(&iter);
  314. bch2_trans_iter_exit(trans, &iter);
  315. printbuf_exit(&buf);
  316. return ob;
  317. }
  318. /*
  319. * This path is for before the freespace btree is initialized:
  320. *
  321. * If ca->new_fs_bucket_idx is nonzero, we haven't yet marked superblock &
  322. * journal buckets - journal buckets will be < ca->new_fs_bucket_idx
  323. */
  324. static noinline struct open_bucket *
  325. bch2_bucket_alloc_early(struct btree_trans *trans,
  326. struct bch_dev *ca,
  327. enum bch_watermark watermark,
  328. struct bucket_alloc_state *s,
  329. struct closure *cl)
  330. {
  331. struct btree_iter iter, citer;
  332. struct bkey_s_c k, ck;
  333. struct open_bucket *ob = NULL;
  334. u64 first_bucket = max_t(u64, ca->mi.first_bucket, ca->new_fs_bucket_idx);
  335. u64 *dev_alloc_cursor = &ca->alloc_cursor[s->btree_bitmap];
  336. u64 alloc_start = max(first_bucket, *dev_alloc_cursor);
  337. u64 alloc_cursor = alloc_start;
  338. int ret;
  339. /*
  340. * Scan with an uncached iterator to avoid polluting the key cache. An
  341. * uncached iter will return a cached key if one exists, but if not
  342. * there is no other underlying protection for the associated key cache
  343. * slot. To avoid racing bucket allocations, look up the cached key slot
  344. * of any likely allocation candidate before attempting to proceed with
  345. * the allocation. This provides proper exclusion on the associated
  346. * bucket.
  347. */
  348. again:
  349. for_each_btree_key_norestart(trans, iter, BTREE_ID_alloc, POS(ca->dev_idx, alloc_cursor),
  350. BTREE_ITER_slots, k, ret) {
  351. u64 bucket = k.k->p.offset;
  352. if (bkey_ge(k.k->p, POS(ca->dev_idx, ca->mi.nbuckets)))
  353. break;
  354. if (ca->new_fs_bucket_idx &&
  355. is_superblock_bucket(ca, k.k->p.offset))
  356. continue;
  357. if (s->btree_bitmap != BTREE_BITMAP_ANY &&
  358. s->btree_bitmap != bch2_dev_btree_bitmap_marked_sectors(ca,
  359. bucket_to_sector(ca, bucket), ca->mi.bucket_size)) {
  360. if (s->btree_bitmap == BTREE_BITMAP_YES &&
  361. bucket_to_sector(ca, bucket) > 64ULL << ca->mi.btree_bitmap_shift)
  362. break;
  363. bucket = sector_to_bucket(ca,
  364. round_up(bucket_to_sector(ca, bucket) + 1,
  365. 1ULL << ca->mi.btree_bitmap_shift));
  366. bch2_btree_iter_set_pos(&iter, POS(ca->dev_idx, bucket));
  367. s->buckets_seen++;
  368. s->skipped_mi_btree_bitmap++;
  369. continue;
  370. }
  371. struct bch_alloc_v4 a_convert;
  372. const struct bch_alloc_v4 *a = bch2_alloc_to_v4(k, &a_convert);
  373. if (a->data_type != BCH_DATA_free)
  374. continue;
  375. /* now check the cached key to serialize concurrent allocs of the bucket */
  376. ck = bch2_bkey_get_iter(trans, &citer, BTREE_ID_alloc, k.k->p, BTREE_ITER_cached);
  377. ret = bkey_err(ck);
  378. if (ret)
  379. break;
  380. a = bch2_alloc_to_v4(ck, &a_convert);
  381. if (a->data_type != BCH_DATA_free)
  382. goto next;
  383. s->buckets_seen++;
  384. ob = __try_alloc_bucket(trans->c, ca, k.k->p.offset, watermark, a, s, cl);
  385. next:
  386. bch2_set_btree_iter_dontneed(&citer);
  387. bch2_trans_iter_exit(trans, &citer);
  388. if (ob)
  389. break;
  390. }
  391. bch2_trans_iter_exit(trans, &iter);
  392. alloc_cursor = iter.pos.offset;
  393. if (!ob && ret)
  394. ob = ERR_PTR(ret);
  395. if (!ob && alloc_start > first_bucket) {
  396. alloc_cursor = alloc_start = first_bucket;
  397. goto again;
  398. }
  399. *dev_alloc_cursor = alloc_cursor;
  400. return ob;
  401. }
  402. static struct open_bucket *bch2_bucket_alloc_freelist(struct btree_trans *trans,
  403. struct bch_dev *ca,
  404. enum bch_watermark watermark,
  405. struct bucket_alloc_state *s,
  406. struct closure *cl)
  407. {
  408. struct btree_iter iter;
  409. struct bkey_s_c k;
  410. struct open_bucket *ob = NULL;
  411. u64 *dev_alloc_cursor = &ca->alloc_cursor[s->btree_bitmap];
  412. u64 alloc_start = max_t(u64, ca->mi.first_bucket, READ_ONCE(*dev_alloc_cursor));
  413. u64 alloc_cursor = alloc_start;
  414. int ret;
  415. BUG_ON(ca->new_fs_bucket_idx);
  416. again:
  417. for_each_btree_key_norestart(trans, iter, BTREE_ID_freespace,
  418. POS(ca->dev_idx, alloc_cursor), 0, k, ret) {
  419. if (k.k->p.inode != ca->dev_idx)
  420. break;
  421. for (alloc_cursor = max(alloc_cursor, bkey_start_offset(k.k));
  422. alloc_cursor < k.k->p.offset;
  423. alloc_cursor++) {
  424. s->buckets_seen++;
  425. u64 bucket = alloc_cursor & ~(~0ULL << 56);
  426. if (s->btree_bitmap != BTREE_BITMAP_ANY &&
  427. s->btree_bitmap != bch2_dev_btree_bitmap_marked_sectors(ca,
  428. bucket_to_sector(ca, bucket), ca->mi.bucket_size)) {
  429. if (s->btree_bitmap == BTREE_BITMAP_YES &&
  430. bucket_to_sector(ca, bucket) > 64ULL << ca->mi.btree_bitmap_shift)
  431. goto fail;
  432. bucket = sector_to_bucket(ca,
  433. round_up(bucket_to_sector(ca, bucket) + 1,
  434. 1ULL << ca->mi.btree_bitmap_shift));
  435. u64 genbits = alloc_cursor >> 56;
  436. alloc_cursor = bucket | (genbits << 56);
  437. if (alloc_cursor > k.k->p.offset)
  438. bch2_btree_iter_set_pos(&iter, POS(ca->dev_idx, alloc_cursor));
  439. s->skipped_mi_btree_bitmap++;
  440. continue;
  441. }
  442. ob = try_alloc_bucket(trans, ca, watermark,
  443. alloc_cursor, s, k, cl);
  444. if (ob) {
  445. bch2_set_btree_iter_dontneed(&iter);
  446. break;
  447. }
  448. }
  449. if (ob || ret)
  450. break;
  451. }
  452. fail:
  453. bch2_trans_iter_exit(trans, &iter);
  454. if (!ob && ret)
  455. ob = ERR_PTR(ret);
  456. if (!ob && alloc_start > ca->mi.first_bucket) {
  457. alloc_cursor = alloc_start = ca->mi.first_bucket;
  458. goto again;
  459. }
  460. *dev_alloc_cursor = alloc_cursor;
  461. return ob;
  462. }
  463. static noinline void trace_bucket_alloc2(struct bch_fs *c, struct bch_dev *ca,
  464. enum bch_watermark watermark,
  465. enum bch_data_type data_type,
  466. struct closure *cl,
  467. struct bch_dev_usage *usage,
  468. struct bucket_alloc_state *s,
  469. struct open_bucket *ob)
  470. {
  471. struct printbuf buf = PRINTBUF;
  472. printbuf_tabstop_push(&buf, 24);
  473. prt_printf(&buf, "dev\t%s (%u)\n", ca->name, ca->dev_idx);
  474. prt_printf(&buf, "watermark\t%s\n", bch2_watermarks[watermark]);
  475. prt_printf(&buf, "data type\t%s\n", __bch2_data_types[data_type]);
  476. prt_printf(&buf, "blocking\t%u\n", cl != NULL);
  477. prt_printf(&buf, "free\t%llu\n", usage->d[BCH_DATA_free].buckets);
  478. prt_printf(&buf, "avail\t%llu\n", dev_buckets_free(ca, *usage, watermark));
  479. prt_printf(&buf, "copygc_wait\t%lu/%lli\n",
  480. bch2_copygc_wait_amount(c),
  481. c->copygc_wait - atomic64_read(&c->io_clock[WRITE].now));
  482. prt_printf(&buf, "seen\t%llu\n", s->buckets_seen);
  483. prt_printf(&buf, "open\t%llu\n", s->skipped_open);
  484. prt_printf(&buf, "need journal commit\t%llu\n", s->skipped_need_journal_commit);
  485. prt_printf(&buf, "nocow\t%llu\n", s->skipped_nocow);
  486. prt_printf(&buf, "nouse\t%llu\n", s->skipped_nouse);
  487. prt_printf(&buf, "mi_btree_bitmap\t%llu\n", s->skipped_mi_btree_bitmap);
  488. if (!IS_ERR(ob)) {
  489. prt_printf(&buf, "allocated\t%llu\n", ob->bucket);
  490. trace_bucket_alloc(c, buf.buf);
  491. } else {
  492. prt_printf(&buf, "err\t%s\n", bch2_err_str(PTR_ERR(ob)));
  493. trace_bucket_alloc_fail(c, buf.buf);
  494. }
  495. printbuf_exit(&buf);
  496. }
  497. /**
  498. * bch2_bucket_alloc_trans - allocate a single bucket from a specific device
  499. * @trans: transaction object
  500. * @ca: device to allocate from
  501. * @watermark: how important is this allocation?
  502. * @data_type: BCH_DATA_journal, btree, user...
  503. * @cl: if not NULL, closure to be used to wait if buckets not available
  504. * @usage: for secondarily also returning the current device usage
  505. *
  506. * Returns: an open_bucket on success, or an ERR_PTR() on failure.
  507. */
  508. static struct open_bucket *bch2_bucket_alloc_trans(struct btree_trans *trans,
  509. struct bch_dev *ca,
  510. enum bch_watermark watermark,
  511. enum bch_data_type data_type,
  512. struct closure *cl,
  513. bool nowait,
  514. struct bch_dev_usage *usage)
  515. {
  516. struct bch_fs *c = trans->c;
  517. struct open_bucket *ob = NULL;
  518. bool freespace = READ_ONCE(ca->mi.freespace_initialized);
  519. u64 avail;
  520. struct bucket_alloc_state s = {
  521. .btree_bitmap = data_type == BCH_DATA_btree,
  522. };
  523. bool waiting = nowait;
  524. again:
  525. bch2_dev_usage_read_fast(ca, usage);
  526. avail = dev_buckets_free(ca, *usage, watermark);
  527. if (usage->d[BCH_DATA_need_discard].buckets > avail)
  528. bch2_dev_do_discards(ca);
  529. if (usage->d[BCH_DATA_need_gc_gens].buckets > avail)
  530. bch2_gc_gens_async(c);
  531. if (should_invalidate_buckets(ca, *usage))
  532. bch2_dev_do_invalidates(ca);
  533. if (!avail) {
  534. if (cl && !waiting) {
  535. closure_wait(&c->freelist_wait, cl);
  536. waiting = true;
  537. goto again;
  538. }
  539. track_event_change(&c->times[BCH_TIME_blocked_allocate], true);
  540. ob = ERR_PTR(-BCH_ERR_freelist_empty);
  541. goto err;
  542. }
  543. if (waiting)
  544. closure_wake_up(&c->freelist_wait);
  545. alloc:
  546. ob = likely(freespace)
  547. ? bch2_bucket_alloc_freelist(trans, ca, watermark, &s, cl)
  548. : bch2_bucket_alloc_early(trans, ca, watermark, &s, cl);
  549. if (s.skipped_need_journal_commit * 2 > avail)
  550. bch2_journal_flush_async(&c->journal, NULL);
  551. if (!ob && s.btree_bitmap != BTREE_BITMAP_ANY) {
  552. s.btree_bitmap = BTREE_BITMAP_ANY;
  553. goto alloc;
  554. }
  555. if (!ob && freespace && c->curr_recovery_pass <= BCH_RECOVERY_PASS_check_alloc_info) {
  556. freespace = false;
  557. goto alloc;
  558. }
  559. err:
  560. if (!ob)
  561. ob = ERR_PTR(-BCH_ERR_no_buckets_found);
  562. if (!IS_ERR(ob))
  563. ob->data_type = data_type;
  564. if (!IS_ERR(ob))
  565. count_event(c, bucket_alloc);
  566. else if (!bch2_err_matches(PTR_ERR(ob), BCH_ERR_transaction_restart))
  567. count_event(c, bucket_alloc_fail);
  568. if (!IS_ERR(ob)
  569. ? trace_bucket_alloc_enabled()
  570. : trace_bucket_alloc_fail_enabled())
  571. trace_bucket_alloc2(c, ca, watermark, data_type, cl, usage, &s, ob);
  572. return ob;
  573. }
  574. struct open_bucket *bch2_bucket_alloc(struct bch_fs *c, struct bch_dev *ca,
  575. enum bch_watermark watermark,
  576. enum bch_data_type data_type,
  577. struct closure *cl)
  578. {
  579. struct bch_dev_usage usage;
  580. struct open_bucket *ob;
  581. bch2_trans_do(c,
  582. PTR_ERR_OR_ZERO(ob = bch2_bucket_alloc_trans(trans, ca, watermark,
  583. data_type, cl, false, &usage)));
  584. return ob;
  585. }
  586. static int __dev_stripe_cmp(struct dev_stripe_state *stripe,
  587. unsigned l, unsigned r)
  588. {
  589. return ((stripe->next_alloc[l] > stripe->next_alloc[r]) -
  590. (stripe->next_alloc[l] < stripe->next_alloc[r]));
  591. }
  592. #define dev_stripe_cmp(l, r) __dev_stripe_cmp(stripe, l, r)
  593. struct dev_alloc_list bch2_dev_alloc_list(struct bch_fs *c,
  594. struct dev_stripe_state *stripe,
  595. struct bch_devs_mask *devs)
  596. {
  597. struct dev_alloc_list ret = { .nr = 0 };
  598. unsigned i;
  599. for_each_set_bit(i, devs->d, BCH_SB_MEMBERS_MAX)
  600. ret.devs[ret.nr++] = i;
  601. bubble_sort(ret.devs, ret.nr, dev_stripe_cmp);
  602. return ret;
  603. }
  604. static inline void bch2_dev_stripe_increment_inlined(struct bch_dev *ca,
  605. struct dev_stripe_state *stripe,
  606. struct bch_dev_usage *usage)
  607. {
  608. u64 *v = stripe->next_alloc + ca->dev_idx;
  609. u64 free_space = dev_buckets_available(ca, BCH_WATERMARK_normal);
  610. u64 free_space_inv = free_space
  611. ? div64_u64(1ULL << 48, free_space)
  612. : 1ULL << 48;
  613. u64 scale = *v / 4;
  614. if (*v + free_space_inv >= *v)
  615. *v += free_space_inv;
  616. else
  617. *v = U64_MAX;
  618. for (v = stripe->next_alloc;
  619. v < stripe->next_alloc + ARRAY_SIZE(stripe->next_alloc); v++)
  620. *v = *v < scale ? 0 : *v - scale;
  621. }
  622. void bch2_dev_stripe_increment(struct bch_dev *ca,
  623. struct dev_stripe_state *stripe)
  624. {
  625. struct bch_dev_usage usage;
  626. bch2_dev_usage_read_fast(ca, &usage);
  627. bch2_dev_stripe_increment_inlined(ca, stripe, &usage);
  628. }
  629. static int add_new_bucket(struct bch_fs *c,
  630. struct open_buckets *ptrs,
  631. struct bch_devs_mask *devs_may_alloc,
  632. unsigned nr_replicas,
  633. unsigned *nr_effective,
  634. bool *have_cache,
  635. struct open_bucket *ob)
  636. {
  637. unsigned durability = ob_dev(c, ob)->mi.durability;
  638. BUG_ON(*nr_effective >= nr_replicas);
  639. __clear_bit(ob->dev, devs_may_alloc->d);
  640. *nr_effective += durability;
  641. *have_cache |= !durability;
  642. ob_push(c, ptrs, ob);
  643. if (*nr_effective >= nr_replicas)
  644. return 1;
  645. if (ob->ec)
  646. return 1;
  647. return 0;
  648. }
  649. int bch2_bucket_alloc_set_trans(struct btree_trans *trans,
  650. struct open_buckets *ptrs,
  651. struct dev_stripe_state *stripe,
  652. struct bch_devs_mask *devs_may_alloc,
  653. unsigned nr_replicas,
  654. unsigned *nr_effective,
  655. bool *have_cache,
  656. enum bch_write_flags flags,
  657. enum bch_data_type data_type,
  658. enum bch_watermark watermark,
  659. struct closure *cl)
  660. {
  661. struct bch_fs *c = trans->c;
  662. struct dev_alloc_list devs_sorted =
  663. bch2_dev_alloc_list(c, stripe, devs_may_alloc);
  664. int ret = -BCH_ERR_insufficient_devices;
  665. BUG_ON(*nr_effective >= nr_replicas);
  666. for (unsigned i = 0; i < devs_sorted.nr; i++) {
  667. struct bch_dev_usage usage;
  668. struct open_bucket *ob;
  669. unsigned dev = devs_sorted.devs[i];
  670. struct bch_dev *ca = bch2_dev_tryget_noerror(c, dev);
  671. if (!ca)
  672. continue;
  673. if (!ca->mi.durability && *have_cache) {
  674. bch2_dev_put(ca);
  675. continue;
  676. }
  677. ob = bch2_bucket_alloc_trans(trans, ca, watermark, data_type,
  678. cl, flags & BCH_WRITE_ALLOC_NOWAIT, &usage);
  679. if (!IS_ERR(ob))
  680. bch2_dev_stripe_increment_inlined(ca, stripe, &usage);
  681. bch2_dev_put(ca);
  682. if (IS_ERR(ob)) {
  683. ret = PTR_ERR(ob);
  684. if (bch2_err_matches(ret, BCH_ERR_transaction_restart) || cl)
  685. break;
  686. continue;
  687. }
  688. if (add_new_bucket(c, ptrs, devs_may_alloc,
  689. nr_replicas, nr_effective,
  690. have_cache, ob)) {
  691. ret = 0;
  692. break;
  693. }
  694. }
  695. return ret;
  696. }
  697. /* Allocate from stripes: */
  698. /*
  699. * if we can't allocate a new stripe because there are already too many
  700. * partially filled stripes, force allocating from an existing stripe even when
  701. * it's to a device we don't want:
  702. */
  703. static int bucket_alloc_from_stripe(struct btree_trans *trans,
  704. struct open_buckets *ptrs,
  705. struct write_point *wp,
  706. struct bch_devs_mask *devs_may_alloc,
  707. u16 target,
  708. unsigned nr_replicas,
  709. unsigned *nr_effective,
  710. bool *have_cache,
  711. enum bch_watermark watermark,
  712. enum bch_write_flags flags,
  713. struct closure *cl)
  714. {
  715. struct bch_fs *c = trans->c;
  716. struct dev_alloc_list devs_sorted;
  717. struct ec_stripe_head *h;
  718. struct open_bucket *ob;
  719. unsigned i, ec_idx;
  720. int ret = 0;
  721. if (nr_replicas < 2)
  722. return 0;
  723. if (ec_open_bucket(c, ptrs))
  724. return 0;
  725. h = bch2_ec_stripe_head_get(trans, target, 0, nr_replicas - 1, watermark, cl);
  726. if (IS_ERR(h))
  727. return PTR_ERR(h);
  728. if (!h)
  729. return 0;
  730. devs_sorted = bch2_dev_alloc_list(c, &wp->stripe, devs_may_alloc);
  731. for (i = 0; i < devs_sorted.nr; i++)
  732. for (ec_idx = 0; ec_idx < h->s->nr_data; ec_idx++) {
  733. if (!h->s->blocks[ec_idx])
  734. continue;
  735. ob = c->open_buckets + h->s->blocks[ec_idx];
  736. if (ob->dev == devs_sorted.devs[i] &&
  737. !test_and_set_bit(ec_idx, h->s->blocks_allocated))
  738. goto got_bucket;
  739. }
  740. goto out_put_head;
  741. got_bucket:
  742. ob->ec_idx = ec_idx;
  743. ob->ec = h->s;
  744. ec_stripe_new_get(h->s, STRIPE_REF_io);
  745. ret = add_new_bucket(c, ptrs, devs_may_alloc,
  746. nr_replicas, nr_effective,
  747. have_cache, ob);
  748. out_put_head:
  749. bch2_ec_stripe_head_put(c, h);
  750. return ret;
  751. }
  752. /* Sector allocator */
  753. static bool want_bucket(struct bch_fs *c,
  754. struct write_point *wp,
  755. struct bch_devs_mask *devs_may_alloc,
  756. bool *have_cache, bool ec,
  757. struct open_bucket *ob)
  758. {
  759. struct bch_dev *ca = ob_dev(c, ob);
  760. if (!test_bit(ob->dev, devs_may_alloc->d))
  761. return false;
  762. if (ob->data_type != wp->data_type)
  763. return false;
  764. if (!ca->mi.durability &&
  765. (wp->data_type == BCH_DATA_btree || ec || *have_cache))
  766. return false;
  767. if (ec != (ob->ec != NULL))
  768. return false;
  769. return true;
  770. }
  771. static int bucket_alloc_set_writepoint(struct bch_fs *c,
  772. struct open_buckets *ptrs,
  773. struct write_point *wp,
  774. struct bch_devs_mask *devs_may_alloc,
  775. unsigned nr_replicas,
  776. unsigned *nr_effective,
  777. bool *have_cache,
  778. bool ec)
  779. {
  780. struct open_buckets ptrs_skip = { .nr = 0 };
  781. struct open_bucket *ob;
  782. unsigned i;
  783. int ret = 0;
  784. open_bucket_for_each(c, &wp->ptrs, ob, i) {
  785. if (!ret && want_bucket(c, wp, devs_may_alloc,
  786. have_cache, ec, ob))
  787. ret = add_new_bucket(c, ptrs, devs_may_alloc,
  788. nr_replicas, nr_effective,
  789. have_cache, ob);
  790. else
  791. ob_push(c, &ptrs_skip, ob);
  792. }
  793. wp->ptrs = ptrs_skip;
  794. return ret;
  795. }
  796. static int bucket_alloc_set_partial(struct bch_fs *c,
  797. struct open_buckets *ptrs,
  798. struct write_point *wp,
  799. struct bch_devs_mask *devs_may_alloc,
  800. unsigned nr_replicas,
  801. unsigned *nr_effective,
  802. bool *have_cache, bool ec,
  803. enum bch_watermark watermark)
  804. {
  805. int i, ret = 0;
  806. if (!c->open_buckets_partial_nr)
  807. return 0;
  808. spin_lock(&c->freelist_lock);
  809. if (!c->open_buckets_partial_nr)
  810. goto unlock;
  811. for (i = c->open_buckets_partial_nr - 1; i >= 0; --i) {
  812. struct open_bucket *ob = c->open_buckets + c->open_buckets_partial[i];
  813. if (want_bucket(c, wp, devs_may_alloc, have_cache, ec, ob)) {
  814. struct bch_dev *ca = ob_dev(c, ob);
  815. struct bch_dev_usage usage;
  816. u64 avail;
  817. bch2_dev_usage_read_fast(ca, &usage);
  818. avail = dev_buckets_free(ca, usage, watermark) + ca->nr_partial_buckets;
  819. if (!avail)
  820. continue;
  821. array_remove_item(c->open_buckets_partial,
  822. c->open_buckets_partial_nr,
  823. i);
  824. ob->on_partial_list = false;
  825. rcu_read_lock();
  826. bch2_dev_rcu(c, ob->dev)->nr_partial_buckets--;
  827. rcu_read_unlock();
  828. ret = add_new_bucket(c, ptrs, devs_may_alloc,
  829. nr_replicas, nr_effective,
  830. have_cache, ob);
  831. if (ret)
  832. break;
  833. }
  834. }
  835. unlock:
  836. spin_unlock(&c->freelist_lock);
  837. return ret;
  838. }
  839. static int __open_bucket_add_buckets(struct btree_trans *trans,
  840. struct open_buckets *ptrs,
  841. struct write_point *wp,
  842. struct bch_devs_list *devs_have,
  843. u16 target,
  844. bool erasure_code,
  845. unsigned nr_replicas,
  846. unsigned *nr_effective,
  847. bool *have_cache,
  848. enum bch_watermark watermark,
  849. enum bch_write_flags flags,
  850. struct closure *_cl)
  851. {
  852. struct bch_fs *c = trans->c;
  853. struct bch_devs_mask devs;
  854. struct open_bucket *ob;
  855. struct closure *cl = NULL;
  856. unsigned i;
  857. int ret;
  858. devs = target_rw_devs(c, wp->data_type, target);
  859. /* Don't allocate from devices we already have pointers to: */
  860. darray_for_each(*devs_have, i)
  861. __clear_bit(*i, devs.d);
  862. open_bucket_for_each(c, ptrs, ob, i)
  863. __clear_bit(ob->dev, devs.d);
  864. ret = bucket_alloc_set_writepoint(c, ptrs, wp, &devs,
  865. nr_replicas, nr_effective,
  866. have_cache, erasure_code);
  867. if (ret)
  868. return ret;
  869. ret = bucket_alloc_set_partial(c, ptrs, wp, &devs,
  870. nr_replicas, nr_effective,
  871. have_cache, erasure_code, watermark);
  872. if (ret)
  873. return ret;
  874. if (erasure_code) {
  875. ret = bucket_alloc_from_stripe(trans, ptrs, wp, &devs,
  876. target,
  877. nr_replicas, nr_effective,
  878. have_cache,
  879. watermark, flags, _cl);
  880. } else {
  881. retry_blocking:
  882. /*
  883. * Try nonblocking first, so that if one device is full we'll try from
  884. * other devices:
  885. */
  886. ret = bch2_bucket_alloc_set_trans(trans, ptrs, &wp->stripe, &devs,
  887. nr_replicas, nr_effective, have_cache,
  888. flags, wp->data_type, watermark, cl);
  889. if (ret &&
  890. !bch2_err_matches(ret, BCH_ERR_transaction_restart) &&
  891. !bch2_err_matches(ret, BCH_ERR_insufficient_devices) &&
  892. !cl && _cl) {
  893. cl = _cl;
  894. goto retry_blocking;
  895. }
  896. }
  897. return ret;
  898. }
  899. static int open_bucket_add_buckets(struct btree_trans *trans,
  900. struct open_buckets *ptrs,
  901. struct write_point *wp,
  902. struct bch_devs_list *devs_have,
  903. u16 target,
  904. unsigned erasure_code,
  905. unsigned nr_replicas,
  906. unsigned *nr_effective,
  907. bool *have_cache,
  908. enum bch_watermark watermark,
  909. enum bch_write_flags flags,
  910. struct closure *cl)
  911. {
  912. int ret;
  913. if (erasure_code && !ec_open_bucket(trans->c, ptrs)) {
  914. ret = __open_bucket_add_buckets(trans, ptrs, wp,
  915. devs_have, target, erasure_code,
  916. nr_replicas, nr_effective, have_cache,
  917. watermark, flags, cl);
  918. if (bch2_err_matches(ret, BCH_ERR_transaction_restart) ||
  919. bch2_err_matches(ret, BCH_ERR_operation_blocked) ||
  920. bch2_err_matches(ret, BCH_ERR_freelist_empty) ||
  921. bch2_err_matches(ret, BCH_ERR_open_buckets_empty))
  922. return ret;
  923. if (*nr_effective >= nr_replicas)
  924. return 0;
  925. }
  926. ret = __open_bucket_add_buckets(trans, ptrs, wp,
  927. devs_have, target, false,
  928. nr_replicas, nr_effective, have_cache,
  929. watermark, flags, cl);
  930. return ret < 0 ? ret : 0;
  931. }
  932. /**
  933. * should_drop_bucket - check if this is open_bucket should go away
  934. * @ob: open_bucket to predicate on
  935. * @c: filesystem handle
  936. * @ca: if set, we're killing buckets for a particular device
  937. * @ec: if true, we're shutting down erasure coding and killing all ec
  938. * open_buckets
  939. * otherwise, return true
  940. * Returns: true if we should kill this open_bucket
  941. *
  942. * We're killing open_buckets because we're shutting down a device, erasure
  943. * coding, or the entire filesystem - check if this open_bucket matches:
  944. */
  945. static bool should_drop_bucket(struct open_bucket *ob, struct bch_fs *c,
  946. struct bch_dev *ca, bool ec)
  947. {
  948. if (ec) {
  949. return ob->ec != NULL;
  950. } else if (ca) {
  951. bool drop = ob->dev == ca->dev_idx;
  952. struct open_bucket *ob2;
  953. unsigned i;
  954. if (!drop && ob->ec) {
  955. unsigned nr_blocks;
  956. mutex_lock(&ob->ec->lock);
  957. nr_blocks = bkey_i_to_stripe(&ob->ec->new_stripe.key)->v.nr_blocks;
  958. for (i = 0; i < nr_blocks; i++) {
  959. if (!ob->ec->blocks[i])
  960. continue;
  961. ob2 = c->open_buckets + ob->ec->blocks[i];
  962. drop |= ob2->dev == ca->dev_idx;
  963. }
  964. mutex_unlock(&ob->ec->lock);
  965. }
  966. return drop;
  967. } else {
  968. return true;
  969. }
  970. }
  971. static void bch2_writepoint_stop(struct bch_fs *c, struct bch_dev *ca,
  972. bool ec, struct write_point *wp)
  973. {
  974. struct open_buckets ptrs = { .nr = 0 };
  975. struct open_bucket *ob;
  976. unsigned i;
  977. mutex_lock(&wp->lock);
  978. open_bucket_for_each(c, &wp->ptrs, ob, i)
  979. if (should_drop_bucket(ob, c, ca, ec))
  980. bch2_open_bucket_put(c, ob);
  981. else
  982. ob_push(c, &ptrs, ob);
  983. wp->ptrs = ptrs;
  984. mutex_unlock(&wp->lock);
  985. }
  986. void bch2_open_buckets_stop(struct bch_fs *c, struct bch_dev *ca,
  987. bool ec)
  988. {
  989. unsigned i;
  990. /* Next, close write points that point to this device... */
  991. for (i = 0; i < ARRAY_SIZE(c->write_points); i++)
  992. bch2_writepoint_stop(c, ca, ec, &c->write_points[i]);
  993. bch2_writepoint_stop(c, ca, ec, &c->copygc_write_point);
  994. bch2_writepoint_stop(c, ca, ec, &c->rebalance_write_point);
  995. bch2_writepoint_stop(c, ca, ec, &c->btree_write_point);
  996. mutex_lock(&c->btree_reserve_cache_lock);
  997. while (c->btree_reserve_cache_nr) {
  998. struct btree_alloc *a =
  999. &c->btree_reserve_cache[--c->btree_reserve_cache_nr];
  1000. bch2_open_buckets_put(c, &a->ob);
  1001. }
  1002. mutex_unlock(&c->btree_reserve_cache_lock);
  1003. spin_lock(&c->freelist_lock);
  1004. i = 0;
  1005. while (i < c->open_buckets_partial_nr) {
  1006. struct open_bucket *ob =
  1007. c->open_buckets + c->open_buckets_partial[i];
  1008. if (should_drop_bucket(ob, c, ca, ec)) {
  1009. --c->open_buckets_partial_nr;
  1010. swap(c->open_buckets_partial[i],
  1011. c->open_buckets_partial[c->open_buckets_partial_nr]);
  1012. ob->on_partial_list = false;
  1013. rcu_read_lock();
  1014. bch2_dev_rcu(c, ob->dev)->nr_partial_buckets--;
  1015. rcu_read_unlock();
  1016. spin_unlock(&c->freelist_lock);
  1017. bch2_open_bucket_put(c, ob);
  1018. spin_lock(&c->freelist_lock);
  1019. } else {
  1020. i++;
  1021. }
  1022. }
  1023. spin_unlock(&c->freelist_lock);
  1024. bch2_ec_stop_dev(c, ca);
  1025. }
  1026. static inline struct hlist_head *writepoint_hash(struct bch_fs *c,
  1027. unsigned long write_point)
  1028. {
  1029. unsigned hash =
  1030. hash_long(write_point, ilog2(ARRAY_SIZE(c->write_points_hash)));
  1031. return &c->write_points_hash[hash];
  1032. }
  1033. static struct write_point *__writepoint_find(struct hlist_head *head,
  1034. unsigned long write_point)
  1035. {
  1036. struct write_point *wp;
  1037. rcu_read_lock();
  1038. hlist_for_each_entry_rcu(wp, head, node)
  1039. if (wp->write_point == write_point)
  1040. goto out;
  1041. wp = NULL;
  1042. out:
  1043. rcu_read_unlock();
  1044. return wp;
  1045. }
  1046. static inline bool too_many_writepoints(struct bch_fs *c, unsigned factor)
  1047. {
  1048. u64 stranded = c->write_points_nr * c->bucket_size_max;
  1049. u64 free = bch2_fs_usage_read_short(c).free;
  1050. return stranded * factor > free;
  1051. }
  1052. static bool try_increase_writepoints(struct bch_fs *c)
  1053. {
  1054. struct write_point *wp;
  1055. if (c->write_points_nr == ARRAY_SIZE(c->write_points) ||
  1056. too_many_writepoints(c, 32))
  1057. return false;
  1058. wp = c->write_points + c->write_points_nr++;
  1059. hlist_add_head_rcu(&wp->node, writepoint_hash(c, wp->write_point));
  1060. return true;
  1061. }
  1062. static bool try_decrease_writepoints(struct btree_trans *trans, unsigned old_nr)
  1063. {
  1064. struct bch_fs *c = trans->c;
  1065. struct write_point *wp;
  1066. struct open_bucket *ob;
  1067. unsigned i;
  1068. mutex_lock(&c->write_points_hash_lock);
  1069. if (c->write_points_nr < old_nr) {
  1070. mutex_unlock(&c->write_points_hash_lock);
  1071. return true;
  1072. }
  1073. if (c->write_points_nr == 1 ||
  1074. !too_many_writepoints(c, 8)) {
  1075. mutex_unlock(&c->write_points_hash_lock);
  1076. return false;
  1077. }
  1078. wp = c->write_points + --c->write_points_nr;
  1079. hlist_del_rcu(&wp->node);
  1080. mutex_unlock(&c->write_points_hash_lock);
  1081. bch2_trans_mutex_lock_norelock(trans, &wp->lock);
  1082. open_bucket_for_each(c, &wp->ptrs, ob, i)
  1083. open_bucket_free_unused(c, ob);
  1084. wp->ptrs.nr = 0;
  1085. mutex_unlock(&wp->lock);
  1086. return true;
  1087. }
  1088. static struct write_point *writepoint_find(struct btree_trans *trans,
  1089. unsigned long write_point)
  1090. {
  1091. struct bch_fs *c = trans->c;
  1092. struct write_point *wp, *oldest;
  1093. struct hlist_head *head;
  1094. if (!(write_point & 1UL)) {
  1095. wp = (struct write_point *) write_point;
  1096. bch2_trans_mutex_lock_norelock(trans, &wp->lock);
  1097. return wp;
  1098. }
  1099. head = writepoint_hash(c, write_point);
  1100. restart_find:
  1101. wp = __writepoint_find(head, write_point);
  1102. if (wp) {
  1103. lock_wp:
  1104. bch2_trans_mutex_lock_norelock(trans, &wp->lock);
  1105. if (wp->write_point == write_point)
  1106. goto out;
  1107. mutex_unlock(&wp->lock);
  1108. goto restart_find;
  1109. }
  1110. restart_find_oldest:
  1111. oldest = NULL;
  1112. for (wp = c->write_points;
  1113. wp < c->write_points + c->write_points_nr; wp++)
  1114. if (!oldest || time_before64(wp->last_used, oldest->last_used))
  1115. oldest = wp;
  1116. bch2_trans_mutex_lock_norelock(trans, &oldest->lock);
  1117. bch2_trans_mutex_lock_norelock(trans, &c->write_points_hash_lock);
  1118. if (oldest >= c->write_points + c->write_points_nr ||
  1119. try_increase_writepoints(c)) {
  1120. mutex_unlock(&c->write_points_hash_lock);
  1121. mutex_unlock(&oldest->lock);
  1122. goto restart_find_oldest;
  1123. }
  1124. wp = __writepoint_find(head, write_point);
  1125. if (wp && wp != oldest) {
  1126. mutex_unlock(&c->write_points_hash_lock);
  1127. mutex_unlock(&oldest->lock);
  1128. goto lock_wp;
  1129. }
  1130. wp = oldest;
  1131. hlist_del_rcu(&wp->node);
  1132. wp->write_point = write_point;
  1133. hlist_add_head_rcu(&wp->node, head);
  1134. mutex_unlock(&c->write_points_hash_lock);
  1135. out:
  1136. wp->last_used = local_clock();
  1137. return wp;
  1138. }
  1139. static noinline void
  1140. deallocate_extra_replicas(struct bch_fs *c,
  1141. struct open_buckets *ptrs,
  1142. struct open_buckets *ptrs_no_use,
  1143. unsigned extra_replicas)
  1144. {
  1145. struct open_buckets ptrs2 = { 0 };
  1146. struct open_bucket *ob;
  1147. unsigned i;
  1148. open_bucket_for_each(c, ptrs, ob, i) {
  1149. unsigned d = ob_dev(c, ob)->mi.durability;
  1150. if (d && d <= extra_replicas) {
  1151. extra_replicas -= d;
  1152. ob_push(c, ptrs_no_use, ob);
  1153. } else {
  1154. ob_push(c, &ptrs2, ob);
  1155. }
  1156. }
  1157. *ptrs = ptrs2;
  1158. }
  1159. /*
  1160. * Get us an open_bucket we can allocate from, return with it locked:
  1161. */
  1162. int bch2_alloc_sectors_start_trans(struct btree_trans *trans,
  1163. unsigned target,
  1164. unsigned erasure_code,
  1165. struct write_point_specifier write_point,
  1166. struct bch_devs_list *devs_have,
  1167. unsigned nr_replicas,
  1168. unsigned nr_replicas_required,
  1169. enum bch_watermark watermark,
  1170. enum bch_write_flags flags,
  1171. struct closure *cl,
  1172. struct write_point **wp_ret)
  1173. {
  1174. struct bch_fs *c = trans->c;
  1175. struct write_point *wp;
  1176. struct open_bucket *ob;
  1177. struct open_buckets ptrs;
  1178. unsigned nr_effective, write_points_nr;
  1179. bool have_cache;
  1180. int ret;
  1181. int i;
  1182. if (!IS_ENABLED(CONFIG_BCACHEFS_ERASURE_CODING))
  1183. erasure_code = false;
  1184. BUG_ON(!nr_replicas || !nr_replicas_required);
  1185. retry:
  1186. ptrs.nr = 0;
  1187. nr_effective = 0;
  1188. write_points_nr = c->write_points_nr;
  1189. have_cache = false;
  1190. *wp_ret = wp = writepoint_find(trans, write_point.v);
  1191. ret = bch2_trans_relock(trans);
  1192. if (ret)
  1193. goto err;
  1194. /* metadata may not allocate on cache devices: */
  1195. if (wp->data_type != BCH_DATA_user)
  1196. have_cache = true;
  1197. if (target && !(flags & BCH_WRITE_ONLY_SPECIFIED_DEVS)) {
  1198. ret = open_bucket_add_buckets(trans, &ptrs, wp, devs_have,
  1199. target, erasure_code,
  1200. nr_replicas, &nr_effective,
  1201. &have_cache, watermark,
  1202. flags, NULL);
  1203. if (!ret ||
  1204. bch2_err_matches(ret, BCH_ERR_transaction_restart))
  1205. goto alloc_done;
  1206. /* Don't retry from all devices if we're out of open buckets: */
  1207. if (bch2_err_matches(ret, BCH_ERR_open_buckets_empty)) {
  1208. int ret2 = open_bucket_add_buckets(trans, &ptrs, wp, devs_have,
  1209. target, erasure_code,
  1210. nr_replicas, &nr_effective,
  1211. &have_cache, watermark,
  1212. flags, cl);
  1213. if (!ret2 ||
  1214. bch2_err_matches(ret2, BCH_ERR_transaction_restart) ||
  1215. bch2_err_matches(ret2, BCH_ERR_open_buckets_empty)) {
  1216. ret = ret2;
  1217. goto alloc_done;
  1218. }
  1219. }
  1220. /*
  1221. * Only try to allocate cache (durability = 0 devices) from the
  1222. * specified target:
  1223. */
  1224. have_cache = true;
  1225. ret = open_bucket_add_buckets(trans, &ptrs, wp, devs_have,
  1226. 0, erasure_code,
  1227. nr_replicas, &nr_effective,
  1228. &have_cache, watermark,
  1229. flags, cl);
  1230. } else {
  1231. ret = open_bucket_add_buckets(trans, &ptrs, wp, devs_have,
  1232. target, erasure_code,
  1233. nr_replicas, &nr_effective,
  1234. &have_cache, watermark,
  1235. flags, cl);
  1236. }
  1237. alloc_done:
  1238. BUG_ON(!ret && nr_effective < nr_replicas);
  1239. if (erasure_code && !ec_open_bucket(c, &ptrs))
  1240. pr_debug("failed to get ec bucket: ret %u", ret);
  1241. if (ret == -BCH_ERR_insufficient_devices &&
  1242. nr_effective >= nr_replicas_required)
  1243. ret = 0;
  1244. if (ret)
  1245. goto err;
  1246. if (nr_effective > nr_replicas)
  1247. deallocate_extra_replicas(c, &ptrs, &wp->ptrs, nr_effective - nr_replicas);
  1248. /* Free buckets we didn't use: */
  1249. open_bucket_for_each(c, &wp->ptrs, ob, i)
  1250. open_bucket_free_unused(c, ob);
  1251. wp->ptrs = ptrs;
  1252. wp->sectors_free = UINT_MAX;
  1253. open_bucket_for_each(c, &wp->ptrs, ob, i)
  1254. wp->sectors_free = min(wp->sectors_free, ob->sectors_free);
  1255. BUG_ON(!wp->sectors_free || wp->sectors_free == UINT_MAX);
  1256. return 0;
  1257. err:
  1258. open_bucket_for_each(c, &wp->ptrs, ob, i)
  1259. if (ptrs.nr < ARRAY_SIZE(ptrs.v))
  1260. ob_push(c, &ptrs, ob);
  1261. else
  1262. open_bucket_free_unused(c, ob);
  1263. wp->ptrs = ptrs;
  1264. mutex_unlock(&wp->lock);
  1265. if (bch2_err_matches(ret, BCH_ERR_freelist_empty) &&
  1266. try_decrease_writepoints(trans, write_points_nr))
  1267. goto retry;
  1268. if (cl && bch2_err_matches(ret, BCH_ERR_open_buckets_empty))
  1269. ret = -BCH_ERR_bucket_alloc_blocked;
  1270. if (cl && !(flags & BCH_WRITE_ALLOC_NOWAIT) &&
  1271. bch2_err_matches(ret, BCH_ERR_freelist_empty))
  1272. ret = -BCH_ERR_bucket_alloc_blocked;
  1273. return ret;
  1274. }
  1275. struct bch_extent_ptr bch2_ob_ptr(struct bch_fs *c, struct open_bucket *ob)
  1276. {
  1277. struct bch_dev *ca = ob_dev(c, ob);
  1278. return (struct bch_extent_ptr) {
  1279. .type = 1 << BCH_EXTENT_ENTRY_ptr,
  1280. .gen = ob->gen,
  1281. .dev = ob->dev,
  1282. .offset = bucket_to_sector(ca, ob->bucket) +
  1283. ca->mi.bucket_size -
  1284. ob->sectors_free,
  1285. };
  1286. }
  1287. void bch2_alloc_sectors_append_ptrs(struct bch_fs *c, struct write_point *wp,
  1288. struct bkey_i *k, unsigned sectors,
  1289. bool cached)
  1290. {
  1291. bch2_alloc_sectors_append_ptrs_inlined(c, wp, k, sectors, cached);
  1292. }
  1293. /*
  1294. * Append pointers to the space we just allocated to @k, and mark @sectors space
  1295. * as allocated out of @ob
  1296. */
  1297. void bch2_alloc_sectors_done(struct bch_fs *c, struct write_point *wp)
  1298. {
  1299. bch2_alloc_sectors_done_inlined(c, wp);
  1300. }
  1301. static inline void writepoint_init(struct write_point *wp,
  1302. enum bch_data_type type)
  1303. {
  1304. mutex_init(&wp->lock);
  1305. wp->data_type = type;
  1306. INIT_WORK(&wp->index_update_work, bch2_write_point_do_index_updates);
  1307. INIT_LIST_HEAD(&wp->writes);
  1308. spin_lock_init(&wp->writes_lock);
  1309. }
  1310. void bch2_fs_allocator_foreground_init(struct bch_fs *c)
  1311. {
  1312. struct open_bucket *ob;
  1313. struct write_point *wp;
  1314. mutex_init(&c->write_points_hash_lock);
  1315. c->write_points_nr = ARRAY_SIZE(c->write_points);
  1316. /* open bucket 0 is a sentinal NULL: */
  1317. spin_lock_init(&c->open_buckets[0].lock);
  1318. for (ob = c->open_buckets + 1;
  1319. ob < c->open_buckets + ARRAY_SIZE(c->open_buckets); ob++) {
  1320. spin_lock_init(&ob->lock);
  1321. c->open_buckets_nr_free++;
  1322. ob->freelist = c->open_buckets_freelist;
  1323. c->open_buckets_freelist = ob - c->open_buckets;
  1324. }
  1325. writepoint_init(&c->btree_write_point, BCH_DATA_btree);
  1326. writepoint_init(&c->rebalance_write_point, BCH_DATA_user);
  1327. writepoint_init(&c->copygc_write_point, BCH_DATA_user);
  1328. for (wp = c->write_points;
  1329. wp < c->write_points + c->write_points_nr; wp++) {
  1330. writepoint_init(wp, BCH_DATA_user);
  1331. wp->last_used = local_clock();
  1332. wp->write_point = (unsigned long) wp;
  1333. hlist_add_head_rcu(&wp->node,
  1334. writepoint_hash(c, wp->write_point));
  1335. }
  1336. }
  1337. void bch2_open_bucket_to_text(struct printbuf *out, struct bch_fs *c, struct open_bucket *ob)
  1338. {
  1339. struct bch_dev *ca = ob_dev(c, ob);
  1340. unsigned data_type = ob->data_type;
  1341. barrier(); /* READ_ONCE() doesn't work on bitfields */
  1342. prt_printf(out, "%zu ref %u ",
  1343. ob - c->open_buckets,
  1344. atomic_read(&ob->pin));
  1345. bch2_prt_data_type(out, data_type);
  1346. prt_printf(out, " %u:%llu gen %u allocated %u/%u",
  1347. ob->dev, ob->bucket, ob->gen,
  1348. ca->mi.bucket_size - ob->sectors_free, ca->mi.bucket_size);
  1349. if (ob->ec)
  1350. prt_printf(out, " ec idx %llu", ob->ec->idx);
  1351. if (ob->on_partial_list)
  1352. prt_str(out, " partial");
  1353. prt_newline(out);
  1354. }
  1355. void bch2_open_buckets_to_text(struct printbuf *out, struct bch_fs *c,
  1356. struct bch_dev *ca)
  1357. {
  1358. struct open_bucket *ob;
  1359. out->atomic++;
  1360. for (ob = c->open_buckets;
  1361. ob < c->open_buckets + ARRAY_SIZE(c->open_buckets);
  1362. ob++) {
  1363. spin_lock(&ob->lock);
  1364. if (ob->valid && (!ca || ob->dev == ca->dev_idx))
  1365. bch2_open_bucket_to_text(out, c, ob);
  1366. spin_unlock(&ob->lock);
  1367. }
  1368. --out->atomic;
  1369. }
  1370. void bch2_open_buckets_partial_to_text(struct printbuf *out, struct bch_fs *c)
  1371. {
  1372. unsigned i;
  1373. out->atomic++;
  1374. spin_lock(&c->freelist_lock);
  1375. for (i = 0; i < c->open_buckets_partial_nr; i++)
  1376. bch2_open_bucket_to_text(out, c,
  1377. c->open_buckets + c->open_buckets_partial[i]);
  1378. spin_unlock(&c->freelist_lock);
  1379. --out->atomic;
  1380. }
  1381. static const char * const bch2_write_point_states[] = {
  1382. #define x(n) #n,
  1383. WRITE_POINT_STATES()
  1384. #undef x
  1385. NULL
  1386. };
  1387. static void bch2_write_point_to_text(struct printbuf *out, struct bch_fs *c,
  1388. struct write_point *wp)
  1389. {
  1390. struct open_bucket *ob;
  1391. unsigned i;
  1392. prt_printf(out, "%lu: ", wp->write_point);
  1393. prt_human_readable_u64(out, wp->sectors_allocated);
  1394. prt_printf(out, " last wrote: ");
  1395. bch2_pr_time_units(out, sched_clock() - wp->last_used);
  1396. for (i = 0; i < WRITE_POINT_STATE_NR; i++) {
  1397. prt_printf(out, " %s: ", bch2_write_point_states[i]);
  1398. bch2_pr_time_units(out, wp->time[i]);
  1399. }
  1400. prt_newline(out);
  1401. printbuf_indent_add(out, 2);
  1402. open_bucket_for_each(c, &wp->ptrs, ob, i)
  1403. bch2_open_bucket_to_text(out, c, ob);
  1404. printbuf_indent_sub(out, 2);
  1405. }
  1406. void bch2_write_points_to_text(struct printbuf *out, struct bch_fs *c)
  1407. {
  1408. struct write_point *wp;
  1409. prt_str(out, "Foreground write points\n");
  1410. for (wp = c->write_points;
  1411. wp < c->write_points + ARRAY_SIZE(c->write_points);
  1412. wp++)
  1413. bch2_write_point_to_text(out, c, wp);
  1414. prt_str(out, "Copygc write point\n");
  1415. bch2_write_point_to_text(out, c, &c->copygc_write_point);
  1416. prt_str(out, "Rebalance write point\n");
  1417. bch2_write_point_to_text(out, c, &c->rebalance_write_point);
  1418. prt_str(out, "Btree write point\n");
  1419. bch2_write_point_to_text(out, c, &c->btree_write_point);
  1420. }
  1421. void bch2_fs_alloc_debug_to_text(struct printbuf *out, struct bch_fs *c)
  1422. {
  1423. unsigned nr[BCH_DATA_NR];
  1424. memset(nr, 0, sizeof(nr));
  1425. for (unsigned i = 0; i < ARRAY_SIZE(c->open_buckets); i++)
  1426. nr[c->open_buckets[i].data_type]++;
  1427. printbuf_tabstops_reset(out);
  1428. printbuf_tabstop_push(out, 24);
  1429. prt_printf(out, "capacity\t%llu\n", c->capacity);
  1430. prt_printf(out, "reserved\t%llu\n", c->reserved);
  1431. prt_printf(out, "hidden\t%llu\n", percpu_u64_get(&c->usage->hidden));
  1432. prt_printf(out, "btree\t%llu\n", percpu_u64_get(&c->usage->btree));
  1433. prt_printf(out, "data\t%llu\n", percpu_u64_get(&c->usage->data));
  1434. prt_printf(out, "cached\t%llu\n", percpu_u64_get(&c->usage->cached));
  1435. prt_printf(out, "reserved\t%llu\n", percpu_u64_get(&c->usage->reserved));
  1436. prt_printf(out, "online_reserved\t%llu\n", percpu_u64_get(c->online_reserved));
  1437. prt_printf(out, "nr_inodes\t%llu\n", percpu_u64_get(&c->usage->nr_inodes));
  1438. prt_newline(out);
  1439. prt_printf(out, "freelist_wait\t%s\n", c->freelist_wait.list.first ? "waiting" : "empty");
  1440. prt_printf(out, "open buckets allocated\t%i\n", OPEN_BUCKETS_COUNT - c->open_buckets_nr_free);
  1441. prt_printf(out, "open buckets total\t%u\n", OPEN_BUCKETS_COUNT);
  1442. prt_printf(out, "open_buckets_wait\t%s\n", c->open_buckets_wait.list.first ? "waiting" : "empty");
  1443. prt_printf(out, "open_buckets_btree\t%u\n", nr[BCH_DATA_btree]);
  1444. prt_printf(out, "open_buckets_user\t%u\n", nr[BCH_DATA_user]);
  1445. prt_printf(out, "btree reserve cache\t%u\n", c->btree_reserve_cache_nr);
  1446. }
  1447. void bch2_dev_alloc_debug_to_text(struct printbuf *out, struct bch_dev *ca)
  1448. {
  1449. struct bch_fs *c = ca->fs;
  1450. struct bch_dev_usage stats = bch2_dev_usage_read(ca);
  1451. unsigned nr[BCH_DATA_NR];
  1452. memset(nr, 0, sizeof(nr));
  1453. for (unsigned i = 0; i < ARRAY_SIZE(c->open_buckets); i++)
  1454. nr[c->open_buckets[i].data_type]++;
  1455. bch2_dev_usage_to_text(out, ca, &stats);
  1456. prt_newline(out);
  1457. prt_printf(out, "reserves:\n");
  1458. for (unsigned i = 0; i < BCH_WATERMARK_NR; i++)
  1459. prt_printf(out, "%s\t%llu\r\n", bch2_watermarks[i], bch2_dev_buckets_reserved(ca, i));
  1460. prt_newline(out);
  1461. printbuf_tabstops_reset(out);
  1462. printbuf_tabstop_push(out, 12);
  1463. printbuf_tabstop_push(out, 16);
  1464. prt_printf(out, "open buckets\t%i\r\n", ca->nr_open_buckets);
  1465. prt_printf(out, "buckets to invalidate\t%llu\r\n", should_invalidate_buckets(ca, stats));
  1466. }
  1467. static noinline void bch2_print_allocator_stuck(struct bch_fs *c)
  1468. {
  1469. struct printbuf buf = PRINTBUF;
  1470. prt_printf(&buf, "Allocator stuck? Waited for %u seconds\n",
  1471. c->opts.allocator_stuck_timeout);
  1472. prt_printf(&buf, "Allocator debug:\n");
  1473. printbuf_indent_add(&buf, 2);
  1474. bch2_fs_alloc_debug_to_text(&buf, c);
  1475. printbuf_indent_sub(&buf, 2);
  1476. prt_newline(&buf);
  1477. for_each_online_member(c, ca) {
  1478. prt_printf(&buf, "Dev %u:\n", ca->dev_idx);
  1479. printbuf_indent_add(&buf, 2);
  1480. bch2_dev_alloc_debug_to_text(&buf, ca);
  1481. printbuf_indent_sub(&buf, 2);
  1482. prt_newline(&buf);
  1483. }
  1484. prt_printf(&buf, "Copygc debug:\n");
  1485. printbuf_indent_add(&buf, 2);
  1486. bch2_copygc_wait_to_text(&buf, c);
  1487. printbuf_indent_sub(&buf, 2);
  1488. prt_newline(&buf);
  1489. prt_printf(&buf, "Journal debug:\n");
  1490. printbuf_indent_add(&buf, 2);
  1491. bch2_journal_debug_to_text(&buf, &c->journal);
  1492. printbuf_indent_sub(&buf, 2);
  1493. bch2_print_string_as_lines(KERN_ERR, buf.buf);
  1494. printbuf_exit(&buf);
  1495. }
  1496. static inline unsigned allocator_wait_timeout(struct bch_fs *c)
  1497. {
  1498. if (c->allocator_last_stuck &&
  1499. time_after(c->allocator_last_stuck + HZ * 60 * 2, jiffies))
  1500. return 0;
  1501. return c->opts.allocator_stuck_timeout * HZ;
  1502. }
  1503. void __bch2_wait_on_allocator(struct bch_fs *c, struct closure *cl)
  1504. {
  1505. unsigned t = allocator_wait_timeout(c);
  1506. if (t && closure_sync_timeout(cl, t)) {
  1507. c->allocator_last_stuck = jiffies;
  1508. bch2_print_allocator_stuck(c);
  1509. }
  1510. closure_sync(cl);
  1511. }