ec.c 62 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020102110221023102410251026102710281029103010311032103310341035103610371038103910401041104210431044104510461047104810491050105110521053105410551056105710581059106010611062106310641065106610671068106910701071107210731074107510761077107810791080108110821083108410851086108710881089109010911092109310941095109610971098109911001101110211031104110511061107110811091110111111121113111411151116111711181119112011211122112311241125112611271128112911301131113211331134113511361137113811391140114111421143114411451146114711481149115011511152115311541155115611571158115911601161116211631164116511661167116811691170117111721173117411751176117711781179118011811182118311841185118611871188118911901191119211931194119511961197119811991200120112021203120412051206120712081209121012111212121312141215121612171218121912201221122212231224122512261227122812291230123112321233123412351236123712381239124012411242124312441245124612471248124912501251125212531254125512561257125812591260126112621263126412651266126712681269127012711272127312741275127612771278127912801281128212831284128512861287128812891290129112921293129412951296129712981299130013011302130313041305130613071308130913101311131213131314131513161317131813191320132113221323132413251326132713281329133013311332133313341335133613371338133913401341134213431344134513461347134813491350135113521353135413551356135713581359136013611362136313641365136613671368136913701371137213731374137513761377137813791380138113821383138413851386138713881389139013911392139313941395139613971398139914001401140214031404140514061407140814091410141114121413141414151416141714181419142014211422142314241425142614271428142914301431143214331434143514361437143814391440144114421443144414451446144714481449145014511452145314541455145614571458145914601461146214631464146514661467146814691470147114721473147414751476147714781479148014811482148314841485148614871488148914901491149214931494149514961497149814991500150115021503150415051506150715081509151015111512151315141515151615171518151915201521152215231524152515261527152815291530153115321533153415351536153715381539154015411542154315441545154615471548154915501551155215531554155515561557155815591560156115621563156415651566156715681569157015711572157315741575157615771578157915801581158215831584158515861587158815891590159115921593159415951596159715981599160016011602160316041605160616071608160916101611161216131614161516161617161816191620162116221623162416251626162716281629163016311632163316341635163616371638163916401641164216431644164516461647164816491650165116521653165416551656165716581659166016611662166316641665166616671668166916701671167216731674167516761677167816791680168116821683168416851686168716881689169016911692169316941695169616971698169917001701170217031704170517061707170817091710171117121713171417151716171717181719172017211722172317241725172617271728172917301731173217331734173517361737173817391740174117421743174417451746174717481749175017511752175317541755175617571758175917601761176217631764176517661767176817691770177117721773177417751776177717781779178017811782178317841785178617871788178917901791179217931794179517961797179817991800180118021803180418051806180718081809181018111812181318141815181618171818181918201821182218231824182518261827182818291830183118321833183418351836183718381839184018411842184318441845184618471848184918501851185218531854185518561857185818591860186118621863186418651866186718681869187018711872187318741875187618771878187918801881188218831884188518861887188818891890189118921893189418951896189718981899190019011902190319041905190619071908190919101911191219131914191519161917191819191920192119221923192419251926192719281929193019311932193319341935193619371938193919401941194219431944194519461947194819491950195119521953195419551956195719581959196019611962196319641965196619671968196919701971197219731974197519761977197819791980198119821983198419851986198719881989199019911992199319941995199619971998199920002001200220032004200520062007200820092010201120122013201420152016201720182019202020212022202320242025202620272028202920302031203220332034203520362037203820392040204120422043204420452046204720482049205020512052205320542055205620572058205920602061206220632064206520662067206820692070207120722073207420752076207720782079208020812082208320842085208620872088208920902091209220932094209520962097209820992100210121022103210421052106210721082109211021112112211321142115211621172118211921202121212221232124212521262127212821292130213121322133213421352136213721382139214021412142214321442145214621472148214921502151215221532154215521562157215821592160216121622163216421652166216721682169217021712172217321742175217621772178217921802181218221832184218521862187218821892190219121922193219421952196219721982199220022012202220322042205220622072208220922102211221222132214221522162217221822192220222122222223222422252226222722282229223022312232223322342235223622372238223922402241224222432244224522462247224822492250225122522253225422552256225722582259226022612262226322642265226622672268226922702271227222732274227522762277227822792280228122822283228422852286228722882289229022912292229322942295229622972298229923002301230223032304230523062307230823092310231123122313231423152316231723182319232023212322232323242325232623272328232923302331233223332334233523362337233823392340234123422343234423452346234723482349235023512352235323542355235623572358235923602361236223632364236523662367236823692370237123722373237423752376237723782379238023812382238323842385238623872388238923902391239223932394239523962397239823992400240124022403240424052406240724082409241024112412241324142415241624172418241924202421242224232424242524262427242824292430243124322433243424352436243724382439244024412442244324442445244624472448244924502451245224532454245524562457245824592460246124622463246424652466246724682469247024712472247324742475247624772478247924802481248224832484248524862487248824892490249124922493249424952496249724982499250025012502250325042505
  1. // SPDX-License-Identifier: GPL-2.0
  2. /* erasure coding */
  3. #include "bcachefs.h"
  4. #include "alloc_background.h"
  5. #include "alloc_foreground.h"
  6. #include "backpointers.h"
  7. #include "bkey_buf.h"
  8. #include "bset.h"
  9. #include "btree_gc.h"
  10. #include "btree_update.h"
  11. #include "btree_write_buffer.h"
  12. #include "buckets.h"
  13. #include "checksum.h"
  14. #include "disk_accounting.h"
  15. #include "disk_groups.h"
  16. #include "ec.h"
  17. #include "error.h"
  18. #include "io_read.h"
  19. #include "io_write.h"
  20. #include "keylist.h"
  21. #include "recovery.h"
  22. #include "replicas.h"
  23. #include "super-io.h"
  24. #include "util.h"
  25. #include <linux/sort.h>
  26. #ifdef __KERNEL__
  27. #include <linux/raid/pq.h>
  28. #include <linux/raid/xor.h>
  29. static void raid5_recov(unsigned disks, unsigned failed_idx,
  30. size_t size, void **data)
  31. {
  32. unsigned i = 2, nr;
  33. BUG_ON(failed_idx >= disks);
  34. swap(data[0], data[failed_idx]);
  35. memcpy(data[0], data[1], size);
  36. while (i < disks) {
  37. nr = min_t(unsigned, disks - i, MAX_XOR_BLOCKS);
  38. xor_blocks(nr, size, data[0], data + i);
  39. i += nr;
  40. }
  41. swap(data[0], data[failed_idx]);
  42. }
  43. static void raid_gen(int nd, int np, size_t size, void **v)
  44. {
  45. if (np >= 1)
  46. raid5_recov(nd + np, nd, size, v);
  47. if (np >= 2)
  48. raid6_call.gen_syndrome(nd + np, size, v);
  49. BUG_ON(np > 2);
  50. }
  51. static void raid_rec(int nr, int *ir, int nd, int np, size_t size, void **v)
  52. {
  53. switch (nr) {
  54. case 0:
  55. break;
  56. case 1:
  57. if (ir[0] < nd + 1)
  58. raid5_recov(nd + 1, ir[0], size, v);
  59. else
  60. raid6_call.gen_syndrome(nd + np, size, v);
  61. break;
  62. case 2:
  63. if (ir[1] < nd) {
  64. /* data+data failure. */
  65. raid6_2data_recov(nd + np, size, ir[0], ir[1], v);
  66. } else if (ir[0] < nd) {
  67. /* data + p/q failure */
  68. if (ir[1] == nd) /* data + p failure */
  69. raid6_datap_recov(nd + np, size, ir[0], v);
  70. else { /* data + q failure */
  71. raid5_recov(nd + 1, ir[0], size, v);
  72. raid6_call.gen_syndrome(nd + np, size, v);
  73. }
  74. } else {
  75. raid_gen(nd, np, size, v);
  76. }
  77. break;
  78. default:
  79. BUG();
  80. }
  81. }
  82. #else
  83. #include <raid/raid.h>
  84. #endif
  85. struct ec_bio {
  86. struct bch_dev *ca;
  87. struct ec_stripe_buf *buf;
  88. size_t idx;
  89. struct bio bio;
  90. };
  91. /* Stripes btree keys: */
  92. int bch2_stripe_validate(struct bch_fs *c, struct bkey_s_c k,
  93. enum bch_validate_flags flags)
  94. {
  95. const struct bch_stripe *s = bkey_s_c_to_stripe(k).v;
  96. int ret = 0;
  97. bkey_fsck_err_on(bkey_eq(k.k->p, POS_MIN) ||
  98. bpos_gt(k.k->p, POS(0, U32_MAX)),
  99. c, stripe_pos_bad,
  100. "stripe at bad pos");
  101. bkey_fsck_err_on(bkey_val_u64s(k.k) < stripe_val_u64s(s),
  102. c, stripe_val_size_bad,
  103. "incorrect value size (%zu < %u)",
  104. bkey_val_u64s(k.k), stripe_val_u64s(s));
  105. bkey_fsck_err_on(s->csum_granularity_bits >= 64,
  106. c, stripe_csum_granularity_bad,
  107. "invalid csum granularity (%u >= 64)",
  108. s->csum_granularity_bits);
  109. ret = bch2_bkey_ptrs_validate(c, k, flags);
  110. fsck_err:
  111. return ret;
  112. }
  113. void bch2_stripe_to_text(struct printbuf *out, struct bch_fs *c,
  114. struct bkey_s_c k)
  115. {
  116. const struct bch_stripe *sp = bkey_s_c_to_stripe(k).v;
  117. struct bch_stripe s = {};
  118. memcpy(&s, sp, min(sizeof(s), bkey_val_bytes(k.k)));
  119. unsigned nr_data = s.nr_blocks - s.nr_redundant;
  120. prt_printf(out, "algo %u sectors %u blocks %u:%u csum ",
  121. s.algorithm,
  122. le16_to_cpu(s.sectors),
  123. nr_data,
  124. s.nr_redundant);
  125. bch2_prt_csum_type(out, s.csum_type);
  126. prt_str(out, " gran ");
  127. if (s.csum_granularity_bits < 64)
  128. prt_printf(out, "%llu", 1ULL << s.csum_granularity_bits);
  129. else
  130. prt_printf(out, "(invalid shift %u)", s.csum_granularity_bits);
  131. if (s.disk_label) {
  132. prt_str(out, " label");
  133. bch2_disk_path_to_text(out, c, s.disk_label - 1);
  134. }
  135. for (unsigned i = 0; i < s.nr_blocks; i++) {
  136. const struct bch_extent_ptr *ptr = sp->ptrs + i;
  137. if ((void *) ptr >= bkey_val_end(k))
  138. break;
  139. prt_char(out, ' ');
  140. bch2_extent_ptr_to_text(out, c, ptr);
  141. if (s.csum_type < BCH_CSUM_NR &&
  142. i < nr_data &&
  143. stripe_blockcount_offset(&s, i) < bkey_val_bytes(k.k))
  144. prt_printf(out, "#%u", stripe_blockcount_get(sp, i));
  145. }
  146. }
  147. /* Triggers: */
  148. static int __mark_stripe_bucket(struct btree_trans *trans,
  149. struct bch_dev *ca,
  150. struct bkey_s_c_stripe s,
  151. unsigned ptr_idx, bool deleting,
  152. struct bpos bucket,
  153. struct bch_alloc_v4 *a,
  154. enum btree_iter_update_trigger_flags flags)
  155. {
  156. const struct bch_extent_ptr *ptr = s.v->ptrs + ptr_idx;
  157. unsigned nr_data = s.v->nr_blocks - s.v->nr_redundant;
  158. bool parity = ptr_idx >= nr_data;
  159. enum bch_data_type data_type = parity ? BCH_DATA_parity : BCH_DATA_stripe;
  160. s64 sectors = parity ? le16_to_cpu(s.v->sectors) : 0;
  161. struct printbuf buf = PRINTBUF;
  162. int ret = 0;
  163. struct bch_fs *c = trans->c;
  164. if (deleting)
  165. sectors = -sectors;
  166. if (!deleting) {
  167. if (bch2_trans_inconsistent_on(a->stripe ||
  168. a->stripe_redundancy, trans,
  169. "bucket %llu:%llu gen %u data type %s dirty_sectors %u: multiple stripes using same bucket (%u, %llu)\n%s",
  170. bucket.inode, bucket.offset, a->gen,
  171. bch2_data_type_str(a->data_type),
  172. a->dirty_sectors,
  173. a->stripe, s.k->p.offset,
  174. (bch2_bkey_val_to_text(&buf, c, s.s_c), buf.buf))) {
  175. ret = -BCH_ERR_mark_stripe;
  176. goto err;
  177. }
  178. if (bch2_trans_inconsistent_on(parity && bch2_bucket_sectors_total(*a), trans,
  179. "bucket %llu:%llu gen %u data type %s dirty_sectors %u cached_sectors %u: data already in parity bucket\n%s",
  180. bucket.inode, bucket.offset, a->gen,
  181. bch2_data_type_str(a->data_type),
  182. a->dirty_sectors,
  183. a->cached_sectors,
  184. (bch2_bkey_val_to_text(&buf, c, s.s_c), buf.buf))) {
  185. ret = -BCH_ERR_mark_stripe;
  186. goto err;
  187. }
  188. } else {
  189. if (bch2_trans_inconsistent_on(a->stripe != s.k->p.offset ||
  190. a->stripe_redundancy != s.v->nr_redundant, trans,
  191. "bucket %llu:%llu gen %u: not marked as stripe when deleting stripe (got %u)\n%s",
  192. bucket.inode, bucket.offset, a->gen,
  193. a->stripe,
  194. (bch2_bkey_val_to_text(&buf, c, s.s_c), buf.buf))) {
  195. ret = -BCH_ERR_mark_stripe;
  196. goto err;
  197. }
  198. if (bch2_trans_inconsistent_on(a->data_type != data_type, trans,
  199. "bucket %llu:%llu gen %u data type %s: wrong data type when stripe, should be %s\n%s",
  200. bucket.inode, bucket.offset, a->gen,
  201. bch2_data_type_str(a->data_type),
  202. bch2_data_type_str(data_type),
  203. (bch2_bkey_val_to_text(&buf, c, s.s_c), buf.buf))) {
  204. ret = -BCH_ERR_mark_stripe;
  205. goto err;
  206. }
  207. if (bch2_trans_inconsistent_on(parity &&
  208. (a->dirty_sectors != -sectors ||
  209. a->cached_sectors), trans,
  210. "bucket %llu:%llu gen %u dirty_sectors %u cached_sectors %u: wrong sectors when deleting parity block of stripe\n%s",
  211. bucket.inode, bucket.offset, a->gen,
  212. a->dirty_sectors,
  213. a->cached_sectors,
  214. (bch2_bkey_val_to_text(&buf, c, s.s_c), buf.buf))) {
  215. ret = -BCH_ERR_mark_stripe;
  216. goto err;
  217. }
  218. }
  219. if (sectors) {
  220. ret = bch2_bucket_ref_update(trans, ca, s.s_c, ptr, sectors, data_type,
  221. a->gen, a->data_type, &a->dirty_sectors);
  222. if (ret)
  223. goto err;
  224. }
  225. if (!deleting) {
  226. a->stripe = s.k->p.offset;
  227. a->stripe_redundancy = s.v->nr_redundant;
  228. alloc_data_type_set(a, data_type);
  229. } else {
  230. a->stripe = 0;
  231. a->stripe_redundancy = 0;
  232. alloc_data_type_set(a, BCH_DATA_user);
  233. }
  234. err:
  235. printbuf_exit(&buf);
  236. return ret;
  237. }
  238. static int mark_stripe_bucket(struct btree_trans *trans,
  239. struct bkey_s_c_stripe s,
  240. unsigned ptr_idx, bool deleting,
  241. enum btree_iter_update_trigger_flags flags)
  242. {
  243. struct bch_fs *c = trans->c;
  244. const struct bch_extent_ptr *ptr = s.v->ptrs + ptr_idx;
  245. struct printbuf buf = PRINTBUF;
  246. int ret = 0;
  247. struct bch_dev *ca = bch2_dev_tryget(c, ptr->dev);
  248. if (unlikely(!ca)) {
  249. if (ptr->dev != BCH_SB_MEMBER_INVALID && !(flags & BTREE_TRIGGER_overwrite))
  250. ret = -BCH_ERR_mark_stripe;
  251. goto err;
  252. }
  253. struct bpos bucket = PTR_BUCKET_POS(ca, ptr);
  254. if (flags & BTREE_TRIGGER_transactional) {
  255. struct bkey_i_alloc_v4 *a =
  256. bch2_trans_start_alloc_update(trans, bucket, 0);
  257. ret = PTR_ERR_OR_ZERO(a) ?:
  258. __mark_stripe_bucket(trans, ca, s, ptr_idx, deleting, bucket, &a->v, flags);
  259. }
  260. if (flags & BTREE_TRIGGER_gc) {
  261. percpu_down_read(&c->mark_lock);
  262. struct bucket *g = gc_bucket(ca, bucket.offset);
  263. if (bch2_fs_inconsistent_on(!g, c, "reference to invalid bucket on device %u\n %s",
  264. ptr->dev,
  265. (bch2_bkey_val_to_text(&buf, c, s.s_c), buf.buf))) {
  266. ret = -BCH_ERR_mark_stripe;
  267. goto err_unlock;
  268. }
  269. bucket_lock(g);
  270. struct bch_alloc_v4 old = bucket_m_to_alloc(*g), new = old;
  271. ret = __mark_stripe_bucket(trans, ca, s, ptr_idx, deleting, bucket, &new, flags);
  272. alloc_to_bucket(g, new);
  273. bucket_unlock(g);
  274. err_unlock:
  275. percpu_up_read(&c->mark_lock);
  276. if (!ret)
  277. ret = bch2_alloc_key_to_dev_counters(trans, ca, &old, &new, flags);
  278. }
  279. err:
  280. bch2_dev_put(ca);
  281. printbuf_exit(&buf);
  282. return ret;
  283. }
  284. static int mark_stripe_buckets(struct btree_trans *trans,
  285. struct bkey_s_c old, struct bkey_s_c new,
  286. enum btree_iter_update_trigger_flags flags)
  287. {
  288. const struct bch_stripe *old_s = old.k->type == KEY_TYPE_stripe
  289. ? bkey_s_c_to_stripe(old).v : NULL;
  290. const struct bch_stripe *new_s = new.k->type == KEY_TYPE_stripe
  291. ? bkey_s_c_to_stripe(new).v : NULL;
  292. BUG_ON(old_s && new_s && old_s->nr_blocks != new_s->nr_blocks);
  293. unsigned nr_blocks = new_s ? new_s->nr_blocks : old_s->nr_blocks;
  294. for (unsigned i = 0; i < nr_blocks; i++) {
  295. if (new_s && old_s &&
  296. !memcmp(&new_s->ptrs[i],
  297. &old_s->ptrs[i],
  298. sizeof(new_s->ptrs[i])))
  299. continue;
  300. if (new_s) {
  301. int ret = mark_stripe_bucket(trans,
  302. bkey_s_c_to_stripe(new), i, false, flags);
  303. if (ret)
  304. return ret;
  305. }
  306. if (old_s) {
  307. int ret = mark_stripe_bucket(trans,
  308. bkey_s_c_to_stripe(old), i, true, flags);
  309. if (ret)
  310. return ret;
  311. }
  312. }
  313. return 0;
  314. }
  315. static inline void stripe_to_mem(struct stripe *m, const struct bch_stripe *s)
  316. {
  317. m->sectors = le16_to_cpu(s->sectors);
  318. m->algorithm = s->algorithm;
  319. m->nr_blocks = s->nr_blocks;
  320. m->nr_redundant = s->nr_redundant;
  321. m->disk_label = s->disk_label;
  322. m->blocks_nonempty = 0;
  323. for (unsigned i = 0; i < s->nr_blocks; i++)
  324. m->blocks_nonempty += !!stripe_blockcount_get(s, i);
  325. }
  326. int bch2_trigger_stripe(struct btree_trans *trans,
  327. enum btree_id btree, unsigned level,
  328. struct bkey_s_c old, struct bkey_s _new,
  329. enum btree_iter_update_trigger_flags flags)
  330. {
  331. struct bkey_s_c new = _new.s_c;
  332. struct bch_fs *c = trans->c;
  333. u64 idx = new.k->p.offset;
  334. const struct bch_stripe *old_s = old.k->type == KEY_TYPE_stripe
  335. ? bkey_s_c_to_stripe(old).v : NULL;
  336. const struct bch_stripe *new_s = new.k->type == KEY_TYPE_stripe
  337. ? bkey_s_c_to_stripe(new).v : NULL;
  338. if (unlikely(flags & BTREE_TRIGGER_check_repair))
  339. return bch2_check_fix_ptrs(trans, btree, level, _new.s_c, flags);
  340. BUG_ON(new_s && old_s &&
  341. (new_s->nr_blocks != old_s->nr_blocks ||
  342. new_s->nr_redundant != old_s->nr_redundant));
  343. if (flags & (BTREE_TRIGGER_transactional|BTREE_TRIGGER_gc)) {
  344. /*
  345. * If the pointers aren't changing, we don't need to do anything:
  346. */
  347. if (new_s && old_s &&
  348. new_s->nr_blocks == old_s->nr_blocks &&
  349. new_s->nr_redundant == old_s->nr_redundant &&
  350. !memcmp(old_s->ptrs, new_s->ptrs,
  351. new_s->nr_blocks * sizeof(struct bch_extent_ptr)))
  352. return 0;
  353. struct gc_stripe *gc = NULL;
  354. if (flags & BTREE_TRIGGER_gc) {
  355. gc = genradix_ptr_alloc(&c->gc_stripes, idx, GFP_KERNEL);
  356. if (!gc) {
  357. bch_err(c, "error allocating memory for gc_stripes, idx %llu", idx);
  358. return -BCH_ERR_ENOMEM_mark_stripe;
  359. }
  360. /*
  361. * This will be wrong when we bring back runtime gc: we should
  362. * be unmarking the old key and then marking the new key
  363. *
  364. * Also: when we bring back runtime gc, locking
  365. */
  366. gc->alive = true;
  367. gc->sectors = le16_to_cpu(new_s->sectors);
  368. gc->nr_blocks = new_s->nr_blocks;
  369. gc->nr_redundant = new_s->nr_redundant;
  370. for (unsigned i = 0; i < new_s->nr_blocks; i++)
  371. gc->ptrs[i] = new_s->ptrs[i];
  372. /*
  373. * gc recalculates this field from stripe ptr
  374. * references:
  375. */
  376. memset(gc->block_sectors, 0, sizeof(gc->block_sectors));
  377. }
  378. if (new_s) {
  379. s64 sectors = (u64) le16_to_cpu(new_s->sectors) * new_s->nr_redundant;
  380. struct disk_accounting_pos acc = {
  381. .type = BCH_DISK_ACCOUNTING_replicas,
  382. };
  383. bch2_bkey_to_replicas(&acc.replicas, new);
  384. int ret = bch2_disk_accounting_mod(trans, &acc, &sectors, 1, gc);
  385. if (ret)
  386. return ret;
  387. if (gc)
  388. memcpy(&gc->r.e, &acc.replicas, replicas_entry_bytes(&acc.replicas));
  389. }
  390. if (old_s) {
  391. s64 sectors = -((s64) le16_to_cpu(old_s->sectors)) * old_s->nr_redundant;
  392. struct disk_accounting_pos acc = {
  393. .type = BCH_DISK_ACCOUNTING_replicas,
  394. };
  395. bch2_bkey_to_replicas(&acc.replicas, old);
  396. int ret = bch2_disk_accounting_mod(trans, &acc, &sectors, 1, gc);
  397. if (ret)
  398. return ret;
  399. }
  400. int ret = mark_stripe_buckets(trans, old, new, flags);
  401. if (ret)
  402. return ret;
  403. }
  404. if (flags & BTREE_TRIGGER_atomic) {
  405. struct stripe *m = genradix_ptr(&c->stripes, idx);
  406. if (!m) {
  407. struct printbuf buf1 = PRINTBUF;
  408. struct printbuf buf2 = PRINTBUF;
  409. bch2_bkey_val_to_text(&buf1, c, old);
  410. bch2_bkey_val_to_text(&buf2, c, new);
  411. bch_err_ratelimited(c, "error marking nonexistent stripe %llu while marking\n"
  412. "old %s\n"
  413. "new %s", idx, buf1.buf, buf2.buf);
  414. printbuf_exit(&buf2);
  415. printbuf_exit(&buf1);
  416. bch2_inconsistent_error(c);
  417. return -1;
  418. }
  419. if (!new_s) {
  420. bch2_stripes_heap_del(c, m, idx);
  421. memset(m, 0, sizeof(*m));
  422. } else {
  423. stripe_to_mem(m, new_s);
  424. if (!old_s)
  425. bch2_stripes_heap_insert(c, m, idx);
  426. else
  427. bch2_stripes_heap_update(c, m, idx);
  428. }
  429. }
  430. return 0;
  431. }
  432. /* returns blocknr in stripe that we matched: */
  433. static const struct bch_extent_ptr *bkey_matches_stripe(struct bch_stripe *s,
  434. struct bkey_s_c k, unsigned *block)
  435. {
  436. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  437. unsigned i, nr_data = s->nr_blocks - s->nr_redundant;
  438. bkey_for_each_ptr(ptrs, ptr)
  439. for (i = 0; i < nr_data; i++)
  440. if (__bch2_ptr_matches_stripe(&s->ptrs[i], ptr,
  441. le16_to_cpu(s->sectors))) {
  442. *block = i;
  443. return ptr;
  444. }
  445. return NULL;
  446. }
  447. static bool extent_has_stripe_ptr(struct bkey_s_c k, u64 idx)
  448. {
  449. switch (k.k->type) {
  450. case KEY_TYPE_extent: {
  451. struct bkey_s_c_extent e = bkey_s_c_to_extent(k);
  452. const union bch_extent_entry *entry;
  453. extent_for_each_entry(e, entry)
  454. if (extent_entry_type(entry) ==
  455. BCH_EXTENT_ENTRY_stripe_ptr &&
  456. entry->stripe_ptr.idx == idx)
  457. return true;
  458. break;
  459. }
  460. }
  461. return false;
  462. }
  463. /* Stripe bufs: */
  464. static void ec_stripe_buf_exit(struct ec_stripe_buf *buf)
  465. {
  466. if (buf->key.k.type == KEY_TYPE_stripe) {
  467. struct bkey_i_stripe *s = bkey_i_to_stripe(&buf->key);
  468. unsigned i;
  469. for (i = 0; i < s->v.nr_blocks; i++) {
  470. kvfree(buf->data[i]);
  471. buf->data[i] = NULL;
  472. }
  473. }
  474. }
  475. /* XXX: this is a non-mempoolified memory allocation: */
  476. static int ec_stripe_buf_init(struct ec_stripe_buf *buf,
  477. unsigned offset, unsigned size)
  478. {
  479. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  480. unsigned csum_granularity = 1U << v->csum_granularity_bits;
  481. unsigned end = offset + size;
  482. unsigned i;
  483. BUG_ON(end > le16_to_cpu(v->sectors));
  484. offset = round_down(offset, csum_granularity);
  485. end = min_t(unsigned, le16_to_cpu(v->sectors),
  486. round_up(end, csum_granularity));
  487. buf->offset = offset;
  488. buf->size = end - offset;
  489. memset(buf->valid, 0xFF, sizeof(buf->valid));
  490. for (i = 0; i < v->nr_blocks; i++) {
  491. buf->data[i] = kvmalloc(buf->size << 9, GFP_KERNEL);
  492. if (!buf->data[i])
  493. goto err;
  494. }
  495. return 0;
  496. err:
  497. ec_stripe_buf_exit(buf);
  498. return -BCH_ERR_ENOMEM_stripe_buf;
  499. }
  500. /* Checksumming: */
  501. static struct bch_csum ec_block_checksum(struct ec_stripe_buf *buf,
  502. unsigned block, unsigned offset)
  503. {
  504. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  505. unsigned csum_granularity = 1 << v->csum_granularity_bits;
  506. unsigned end = buf->offset + buf->size;
  507. unsigned len = min(csum_granularity, end - offset);
  508. BUG_ON(offset >= end);
  509. BUG_ON(offset < buf->offset);
  510. BUG_ON(offset & (csum_granularity - 1));
  511. BUG_ON(offset + len != le16_to_cpu(v->sectors) &&
  512. (len & (csum_granularity - 1)));
  513. return bch2_checksum(NULL, v->csum_type,
  514. null_nonce(),
  515. buf->data[block] + ((offset - buf->offset) << 9),
  516. len << 9);
  517. }
  518. static void ec_generate_checksums(struct ec_stripe_buf *buf)
  519. {
  520. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  521. unsigned i, j, csums_per_device = stripe_csums_per_device(v);
  522. if (!v->csum_type)
  523. return;
  524. BUG_ON(buf->offset);
  525. BUG_ON(buf->size != le16_to_cpu(v->sectors));
  526. for (i = 0; i < v->nr_blocks; i++)
  527. for (j = 0; j < csums_per_device; j++)
  528. stripe_csum_set(v, i, j,
  529. ec_block_checksum(buf, i, j << v->csum_granularity_bits));
  530. }
  531. static void ec_validate_checksums(struct bch_fs *c, struct ec_stripe_buf *buf)
  532. {
  533. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  534. unsigned csum_granularity = 1 << v->csum_granularity_bits;
  535. unsigned i;
  536. if (!v->csum_type)
  537. return;
  538. for (i = 0; i < v->nr_blocks; i++) {
  539. unsigned offset = buf->offset;
  540. unsigned end = buf->offset + buf->size;
  541. if (!test_bit(i, buf->valid))
  542. continue;
  543. while (offset < end) {
  544. unsigned j = offset >> v->csum_granularity_bits;
  545. unsigned len = min(csum_granularity, end - offset);
  546. struct bch_csum want = stripe_csum_get(v, i, j);
  547. struct bch_csum got = ec_block_checksum(buf, i, offset);
  548. if (bch2_crc_cmp(want, got)) {
  549. struct bch_dev *ca = bch2_dev_tryget(c, v->ptrs[i].dev);
  550. if (ca) {
  551. struct printbuf err = PRINTBUF;
  552. prt_str(&err, "stripe ");
  553. bch2_csum_err_msg(&err, v->csum_type, want, got);
  554. prt_printf(&err, " for %ps at %u of\n ", (void *) _RET_IP_, i);
  555. bch2_bkey_val_to_text(&err, c, bkey_i_to_s_c(&buf->key));
  556. bch_err_ratelimited(ca, "%s", err.buf);
  557. printbuf_exit(&err);
  558. bch2_io_error(ca, BCH_MEMBER_ERROR_checksum);
  559. }
  560. clear_bit(i, buf->valid);
  561. break;
  562. }
  563. offset += len;
  564. }
  565. }
  566. }
  567. /* Erasure coding: */
  568. static void ec_generate_ec(struct ec_stripe_buf *buf)
  569. {
  570. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  571. unsigned nr_data = v->nr_blocks - v->nr_redundant;
  572. unsigned bytes = le16_to_cpu(v->sectors) << 9;
  573. raid_gen(nr_data, v->nr_redundant, bytes, buf->data);
  574. }
  575. static unsigned ec_nr_failed(struct ec_stripe_buf *buf)
  576. {
  577. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  578. return v->nr_blocks - bitmap_weight(buf->valid, v->nr_blocks);
  579. }
  580. static int ec_do_recov(struct bch_fs *c, struct ec_stripe_buf *buf)
  581. {
  582. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  583. unsigned i, failed[BCH_BKEY_PTRS_MAX], nr_failed = 0;
  584. unsigned nr_data = v->nr_blocks - v->nr_redundant;
  585. unsigned bytes = buf->size << 9;
  586. if (ec_nr_failed(buf) > v->nr_redundant) {
  587. bch_err_ratelimited(c,
  588. "error doing reconstruct read: unable to read enough blocks");
  589. return -1;
  590. }
  591. for (i = 0; i < nr_data; i++)
  592. if (!test_bit(i, buf->valid))
  593. failed[nr_failed++] = i;
  594. raid_rec(nr_failed, failed, nr_data, v->nr_redundant, bytes, buf->data);
  595. return 0;
  596. }
  597. /* IO: */
  598. static void ec_block_endio(struct bio *bio)
  599. {
  600. struct ec_bio *ec_bio = container_of(bio, struct ec_bio, bio);
  601. struct bch_stripe *v = &bkey_i_to_stripe(&ec_bio->buf->key)->v;
  602. struct bch_extent_ptr *ptr = &v->ptrs[ec_bio->idx];
  603. struct bch_dev *ca = ec_bio->ca;
  604. struct closure *cl = bio->bi_private;
  605. if (bch2_dev_io_err_on(bio->bi_status, ca,
  606. bio_data_dir(bio)
  607. ? BCH_MEMBER_ERROR_write
  608. : BCH_MEMBER_ERROR_read,
  609. "erasure coding %s error: %s",
  610. bio_data_dir(bio) ? "write" : "read",
  611. bch2_blk_status_to_str(bio->bi_status)))
  612. clear_bit(ec_bio->idx, ec_bio->buf->valid);
  613. int stale = dev_ptr_stale(ca, ptr);
  614. if (stale) {
  615. bch_err_ratelimited(ca->fs,
  616. "error %s stripe: stale/invalid pointer (%i) after io",
  617. bio_data_dir(bio) == READ ? "reading from" : "writing to",
  618. stale);
  619. clear_bit(ec_bio->idx, ec_bio->buf->valid);
  620. }
  621. bio_put(&ec_bio->bio);
  622. percpu_ref_put(&ca->io_ref);
  623. closure_put(cl);
  624. }
  625. static void ec_block_io(struct bch_fs *c, struct ec_stripe_buf *buf,
  626. blk_opf_t opf, unsigned idx, struct closure *cl)
  627. {
  628. struct bch_stripe *v = &bkey_i_to_stripe(&buf->key)->v;
  629. unsigned offset = 0, bytes = buf->size << 9;
  630. struct bch_extent_ptr *ptr = &v->ptrs[idx];
  631. enum bch_data_type data_type = idx < v->nr_blocks - v->nr_redundant
  632. ? BCH_DATA_user
  633. : BCH_DATA_parity;
  634. int rw = op_is_write(opf);
  635. struct bch_dev *ca = bch2_dev_get_ioref(c, ptr->dev, rw);
  636. if (!ca) {
  637. clear_bit(idx, buf->valid);
  638. return;
  639. }
  640. int stale = dev_ptr_stale(ca, ptr);
  641. if (stale) {
  642. bch_err_ratelimited(c,
  643. "error %s stripe: stale pointer (%i)",
  644. rw == READ ? "reading from" : "writing to",
  645. stale);
  646. clear_bit(idx, buf->valid);
  647. return;
  648. }
  649. this_cpu_add(ca->io_done->sectors[rw][data_type], buf->size);
  650. while (offset < bytes) {
  651. unsigned nr_iovecs = min_t(size_t, BIO_MAX_VECS,
  652. DIV_ROUND_UP(bytes, PAGE_SIZE));
  653. unsigned b = min_t(size_t, bytes - offset,
  654. nr_iovecs << PAGE_SHIFT);
  655. struct ec_bio *ec_bio;
  656. ec_bio = container_of(bio_alloc_bioset(ca->disk_sb.bdev,
  657. nr_iovecs,
  658. opf,
  659. GFP_KERNEL,
  660. &c->ec_bioset),
  661. struct ec_bio, bio);
  662. ec_bio->ca = ca;
  663. ec_bio->buf = buf;
  664. ec_bio->idx = idx;
  665. ec_bio->bio.bi_iter.bi_sector = ptr->offset + buf->offset + (offset >> 9);
  666. ec_bio->bio.bi_end_io = ec_block_endio;
  667. ec_bio->bio.bi_private = cl;
  668. bch2_bio_map(&ec_bio->bio, buf->data[idx] + offset, b);
  669. closure_get(cl);
  670. percpu_ref_get(&ca->io_ref);
  671. submit_bio(&ec_bio->bio);
  672. offset += b;
  673. }
  674. percpu_ref_put(&ca->io_ref);
  675. }
  676. static int get_stripe_key_trans(struct btree_trans *trans, u64 idx,
  677. struct ec_stripe_buf *stripe)
  678. {
  679. struct btree_iter iter;
  680. struct bkey_s_c k;
  681. int ret;
  682. k = bch2_bkey_get_iter(trans, &iter, BTREE_ID_stripes,
  683. POS(0, idx), BTREE_ITER_slots);
  684. ret = bkey_err(k);
  685. if (ret)
  686. goto err;
  687. if (k.k->type != KEY_TYPE_stripe) {
  688. ret = -ENOENT;
  689. goto err;
  690. }
  691. bkey_reassemble(&stripe->key, k);
  692. err:
  693. bch2_trans_iter_exit(trans, &iter);
  694. return ret;
  695. }
  696. /* recovery read path: */
  697. int bch2_ec_read_extent(struct btree_trans *trans, struct bch_read_bio *rbio,
  698. struct bkey_s_c orig_k)
  699. {
  700. struct bch_fs *c = trans->c;
  701. struct ec_stripe_buf *buf = NULL;
  702. struct closure cl;
  703. struct bch_stripe *v;
  704. unsigned i, offset;
  705. const char *msg = NULL;
  706. struct printbuf msgbuf = PRINTBUF;
  707. int ret = 0;
  708. closure_init_stack(&cl);
  709. BUG_ON(!rbio->pick.has_ec);
  710. buf = kzalloc(sizeof(*buf), GFP_NOFS);
  711. if (!buf)
  712. return -BCH_ERR_ENOMEM_ec_read_extent;
  713. ret = lockrestart_do(trans, get_stripe_key_trans(trans, rbio->pick.ec.idx, buf));
  714. if (ret) {
  715. msg = "stripe not found";
  716. goto err;
  717. }
  718. v = &bkey_i_to_stripe(&buf->key)->v;
  719. if (!bch2_ptr_matches_stripe(v, rbio->pick)) {
  720. msg = "pointer doesn't match stripe";
  721. goto err;
  722. }
  723. offset = rbio->bio.bi_iter.bi_sector - v->ptrs[rbio->pick.ec.block].offset;
  724. if (offset + bio_sectors(&rbio->bio) > le16_to_cpu(v->sectors)) {
  725. msg = "read is bigger than stripe";
  726. goto err;
  727. }
  728. ret = ec_stripe_buf_init(buf, offset, bio_sectors(&rbio->bio));
  729. if (ret) {
  730. msg = "-ENOMEM";
  731. goto err;
  732. }
  733. for (i = 0; i < v->nr_blocks; i++)
  734. ec_block_io(c, buf, REQ_OP_READ, i, &cl);
  735. closure_sync(&cl);
  736. if (ec_nr_failed(buf) > v->nr_redundant) {
  737. msg = "unable to read enough blocks";
  738. goto err;
  739. }
  740. ec_validate_checksums(c, buf);
  741. ret = ec_do_recov(c, buf);
  742. if (ret)
  743. goto err;
  744. memcpy_to_bio(&rbio->bio, rbio->bio.bi_iter,
  745. buf->data[rbio->pick.ec.block] + ((offset - buf->offset) << 9));
  746. out:
  747. ec_stripe_buf_exit(buf);
  748. kfree(buf);
  749. return ret;
  750. err:
  751. bch2_bkey_val_to_text(&msgbuf, c, orig_k);
  752. bch_err_ratelimited(c,
  753. "error doing reconstruct read: %s\n %s", msg, msgbuf.buf);
  754. printbuf_exit(&msgbuf);;
  755. ret = -BCH_ERR_stripe_reconstruct;
  756. goto out;
  757. }
  758. /* stripe bucket accounting: */
  759. static int __ec_stripe_mem_alloc(struct bch_fs *c, size_t idx, gfp_t gfp)
  760. {
  761. ec_stripes_heap n, *h = &c->ec_stripes_heap;
  762. if (idx >= h->size) {
  763. if (!init_heap(&n, max(1024UL, roundup_pow_of_two(idx + 1)), gfp))
  764. return -BCH_ERR_ENOMEM_ec_stripe_mem_alloc;
  765. mutex_lock(&c->ec_stripes_heap_lock);
  766. if (n.size > h->size) {
  767. memcpy(n.data, h->data, h->nr * sizeof(h->data[0]));
  768. n.nr = h->nr;
  769. swap(*h, n);
  770. }
  771. mutex_unlock(&c->ec_stripes_heap_lock);
  772. free_heap(&n);
  773. }
  774. if (!genradix_ptr_alloc(&c->stripes, idx, gfp))
  775. return -BCH_ERR_ENOMEM_ec_stripe_mem_alloc;
  776. if (c->gc_pos.phase != GC_PHASE_not_running &&
  777. !genradix_ptr_alloc(&c->gc_stripes, idx, gfp))
  778. return -BCH_ERR_ENOMEM_ec_stripe_mem_alloc;
  779. return 0;
  780. }
  781. static int ec_stripe_mem_alloc(struct btree_trans *trans,
  782. struct btree_iter *iter)
  783. {
  784. return allocate_dropping_locks_errcode(trans,
  785. __ec_stripe_mem_alloc(trans->c, iter->pos.offset, _gfp));
  786. }
  787. /*
  788. * Hash table of open stripes:
  789. * Stripes that are being created or modified are kept in a hash table, so that
  790. * stripe deletion can skip them.
  791. */
  792. static bool __bch2_stripe_is_open(struct bch_fs *c, u64 idx)
  793. {
  794. unsigned hash = hash_64(idx, ilog2(ARRAY_SIZE(c->ec_stripes_new)));
  795. struct ec_stripe_new *s;
  796. hlist_for_each_entry(s, &c->ec_stripes_new[hash], hash)
  797. if (s->idx == idx)
  798. return true;
  799. return false;
  800. }
  801. static bool bch2_stripe_is_open(struct bch_fs *c, u64 idx)
  802. {
  803. bool ret = false;
  804. spin_lock(&c->ec_stripes_new_lock);
  805. ret = __bch2_stripe_is_open(c, idx);
  806. spin_unlock(&c->ec_stripes_new_lock);
  807. return ret;
  808. }
  809. static bool bch2_try_open_stripe(struct bch_fs *c,
  810. struct ec_stripe_new *s,
  811. u64 idx)
  812. {
  813. bool ret;
  814. spin_lock(&c->ec_stripes_new_lock);
  815. ret = !__bch2_stripe_is_open(c, idx);
  816. if (ret) {
  817. unsigned hash = hash_64(idx, ilog2(ARRAY_SIZE(c->ec_stripes_new)));
  818. s->idx = idx;
  819. hlist_add_head(&s->hash, &c->ec_stripes_new[hash]);
  820. }
  821. spin_unlock(&c->ec_stripes_new_lock);
  822. return ret;
  823. }
  824. static void bch2_stripe_close(struct bch_fs *c, struct ec_stripe_new *s)
  825. {
  826. BUG_ON(!s->idx);
  827. spin_lock(&c->ec_stripes_new_lock);
  828. hlist_del_init(&s->hash);
  829. spin_unlock(&c->ec_stripes_new_lock);
  830. s->idx = 0;
  831. }
  832. /* Heap of all existing stripes, ordered by blocks_nonempty */
  833. static u64 stripe_idx_to_delete(struct bch_fs *c)
  834. {
  835. ec_stripes_heap *h = &c->ec_stripes_heap;
  836. lockdep_assert_held(&c->ec_stripes_heap_lock);
  837. if (h->nr &&
  838. h->data[0].blocks_nonempty == 0 &&
  839. !bch2_stripe_is_open(c, h->data[0].idx))
  840. return h->data[0].idx;
  841. return 0;
  842. }
  843. static inline void ec_stripes_heap_set_backpointer(ec_stripes_heap *h,
  844. size_t i)
  845. {
  846. struct bch_fs *c = container_of(h, struct bch_fs, ec_stripes_heap);
  847. genradix_ptr(&c->stripes, h->data[i].idx)->heap_idx = i;
  848. }
  849. static inline bool ec_stripes_heap_cmp(const void *l, const void *r, void __always_unused *args)
  850. {
  851. struct ec_stripe_heap_entry *_l = (struct ec_stripe_heap_entry *)l;
  852. struct ec_stripe_heap_entry *_r = (struct ec_stripe_heap_entry *)r;
  853. return ((_l->blocks_nonempty > _r->blocks_nonempty) <
  854. (_l->blocks_nonempty < _r->blocks_nonempty));
  855. }
  856. static inline void ec_stripes_heap_swap(void *l, void *r, void *h)
  857. {
  858. struct ec_stripe_heap_entry *_l = (struct ec_stripe_heap_entry *)l;
  859. struct ec_stripe_heap_entry *_r = (struct ec_stripe_heap_entry *)r;
  860. ec_stripes_heap *_h = (ec_stripes_heap *)h;
  861. size_t i = _l - _h->data;
  862. size_t j = _r - _h->data;
  863. swap(*_l, *_r);
  864. ec_stripes_heap_set_backpointer(_h, i);
  865. ec_stripes_heap_set_backpointer(_h, j);
  866. }
  867. static void heap_verify_backpointer(struct bch_fs *c, size_t idx)
  868. {
  869. ec_stripes_heap *h = &c->ec_stripes_heap;
  870. struct stripe *m = genradix_ptr(&c->stripes, idx);
  871. BUG_ON(m->heap_idx >= h->nr);
  872. BUG_ON(h->data[m->heap_idx].idx != idx);
  873. }
  874. void bch2_stripes_heap_del(struct bch_fs *c,
  875. struct stripe *m, size_t idx)
  876. {
  877. const struct min_heap_callbacks callbacks = {
  878. .less = ec_stripes_heap_cmp,
  879. .swp = ec_stripes_heap_swap,
  880. };
  881. mutex_lock(&c->ec_stripes_heap_lock);
  882. heap_verify_backpointer(c, idx);
  883. min_heap_del(&c->ec_stripes_heap, m->heap_idx, &callbacks, &c->ec_stripes_heap);
  884. mutex_unlock(&c->ec_stripes_heap_lock);
  885. }
  886. void bch2_stripes_heap_insert(struct bch_fs *c,
  887. struct stripe *m, size_t idx)
  888. {
  889. const struct min_heap_callbacks callbacks = {
  890. .less = ec_stripes_heap_cmp,
  891. .swp = ec_stripes_heap_swap,
  892. };
  893. mutex_lock(&c->ec_stripes_heap_lock);
  894. BUG_ON(min_heap_full(&c->ec_stripes_heap));
  895. genradix_ptr(&c->stripes, idx)->heap_idx = c->ec_stripes_heap.nr;
  896. min_heap_push(&c->ec_stripes_heap, &((struct ec_stripe_heap_entry) {
  897. .idx = idx,
  898. .blocks_nonempty = m->blocks_nonempty,
  899. }),
  900. &callbacks,
  901. &c->ec_stripes_heap);
  902. heap_verify_backpointer(c, idx);
  903. mutex_unlock(&c->ec_stripes_heap_lock);
  904. }
  905. void bch2_stripes_heap_update(struct bch_fs *c,
  906. struct stripe *m, size_t idx)
  907. {
  908. const struct min_heap_callbacks callbacks = {
  909. .less = ec_stripes_heap_cmp,
  910. .swp = ec_stripes_heap_swap,
  911. };
  912. ec_stripes_heap *h = &c->ec_stripes_heap;
  913. bool do_deletes;
  914. size_t i;
  915. mutex_lock(&c->ec_stripes_heap_lock);
  916. heap_verify_backpointer(c, idx);
  917. h->data[m->heap_idx].blocks_nonempty = m->blocks_nonempty;
  918. i = m->heap_idx;
  919. min_heap_sift_up(h, i, &callbacks, &c->ec_stripes_heap);
  920. min_heap_sift_down(h, i, &callbacks, &c->ec_stripes_heap);
  921. heap_verify_backpointer(c, idx);
  922. do_deletes = stripe_idx_to_delete(c) != 0;
  923. mutex_unlock(&c->ec_stripes_heap_lock);
  924. if (do_deletes)
  925. bch2_do_stripe_deletes(c);
  926. }
  927. /* stripe deletion */
  928. static int ec_stripe_delete(struct btree_trans *trans, u64 idx)
  929. {
  930. struct bch_fs *c = trans->c;
  931. struct btree_iter iter;
  932. struct bkey_s_c k;
  933. struct bkey_s_c_stripe s;
  934. int ret;
  935. k = bch2_bkey_get_iter(trans, &iter, BTREE_ID_stripes, POS(0, idx),
  936. BTREE_ITER_intent);
  937. ret = bkey_err(k);
  938. if (ret)
  939. goto err;
  940. if (k.k->type != KEY_TYPE_stripe) {
  941. bch2_fs_inconsistent(c, "attempting to delete nonexistent stripe %llu", idx);
  942. ret = -EINVAL;
  943. goto err;
  944. }
  945. s = bkey_s_c_to_stripe(k);
  946. for (unsigned i = 0; i < s.v->nr_blocks; i++)
  947. if (stripe_blockcount_get(s.v, i)) {
  948. struct printbuf buf = PRINTBUF;
  949. bch2_bkey_val_to_text(&buf, c, k);
  950. bch2_fs_inconsistent(c, "attempting to delete nonempty stripe %s", buf.buf);
  951. printbuf_exit(&buf);
  952. ret = -EINVAL;
  953. goto err;
  954. }
  955. ret = bch2_btree_delete_at(trans, &iter, 0);
  956. err:
  957. bch2_trans_iter_exit(trans, &iter);
  958. return ret;
  959. }
  960. static void ec_stripe_delete_work(struct work_struct *work)
  961. {
  962. struct bch_fs *c =
  963. container_of(work, struct bch_fs, ec_stripe_delete_work);
  964. while (1) {
  965. mutex_lock(&c->ec_stripes_heap_lock);
  966. u64 idx = stripe_idx_to_delete(c);
  967. mutex_unlock(&c->ec_stripes_heap_lock);
  968. if (!idx)
  969. break;
  970. int ret = bch2_trans_commit_do(c, NULL, NULL, BCH_TRANS_COMMIT_no_enospc,
  971. ec_stripe_delete(trans, idx));
  972. bch_err_fn(c, ret);
  973. if (ret)
  974. break;
  975. }
  976. bch2_write_ref_put(c, BCH_WRITE_REF_stripe_delete);
  977. }
  978. void bch2_do_stripe_deletes(struct bch_fs *c)
  979. {
  980. if (bch2_write_ref_tryget(c, BCH_WRITE_REF_stripe_delete) &&
  981. !queue_work(c->write_ref_wq, &c->ec_stripe_delete_work))
  982. bch2_write_ref_put(c, BCH_WRITE_REF_stripe_delete);
  983. }
  984. /* stripe creation: */
  985. static int ec_stripe_key_update(struct btree_trans *trans,
  986. struct bkey_i_stripe *old,
  987. struct bkey_i_stripe *new)
  988. {
  989. struct bch_fs *c = trans->c;
  990. bool create = !old;
  991. struct btree_iter iter;
  992. struct bkey_s_c k = bch2_bkey_get_iter(trans, &iter, BTREE_ID_stripes,
  993. new->k.p, BTREE_ITER_intent);
  994. int ret = bkey_err(k);
  995. if (ret)
  996. goto err;
  997. if (bch2_fs_inconsistent_on(k.k->type != (create ? KEY_TYPE_deleted : KEY_TYPE_stripe),
  998. c, "error %s stripe: got existing key type %s",
  999. create ? "creating" : "updating",
  1000. bch2_bkey_types[k.k->type])) {
  1001. ret = -EINVAL;
  1002. goto err;
  1003. }
  1004. if (k.k->type == KEY_TYPE_stripe) {
  1005. const struct bch_stripe *v = bkey_s_c_to_stripe(k).v;
  1006. BUG_ON(old->v.nr_blocks != new->v.nr_blocks);
  1007. BUG_ON(old->v.nr_blocks != v->nr_blocks);
  1008. for (unsigned i = 0; i < new->v.nr_blocks; i++) {
  1009. unsigned sectors = stripe_blockcount_get(v, i);
  1010. if (!bch2_extent_ptr_eq(old->v.ptrs[i], new->v.ptrs[i]) && sectors) {
  1011. struct printbuf buf = PRINTBUF;
  1012. prt_printf(&buf, "stripe changed nonempty block %u", i);
  1013. prt_str(&buf, "\nold: ");
  1014. bch2_bkey_val_to_text(&buf, c, k);
  1015. prt_str(&buf, "\nnew: ");
  1016. bch2_bkey_val_to_text(&buf, c, bkey_i_to_s_c(&new->k_i));
  1017. bch2_fs_inconsistent(c, "%s", buf.buf);
  1018. printbuf_exit(&buf);
  1019. ret = -EINVAL;
  1020. goto err;
  1021. }
  1022. /*
  1023. * If the stripe ptr changed underneath us, it must have
  1024. * been dev_remove_stripes() -> * invalidate_stripe_to_dev()
  1025. */
  1026. if (!bch2_extent_ptr_eq(old->v.ptrs[i], v->ptrs[i])) {
  1027. BUG_ON(v->ptrs[i].dev != BCH_SB_MEMBER_INVALID);
  1028. if (bch2_extent_ptr_eq(old->v.ptrs[i], new->v.ptrs[i]))
  1029. new->v.ptrs[i].dev = BCH_SB_MEMBER_INVALID;
  1030. }
  1031. stripe_blockcount_set(&new->v, i, sectors);
  1032. }
  1033. }
  1034. ret = bch2_trans_update(trans, &iter, &new->k_i, 0);
  1035. err:
  1036. bch2_trans_iter_exit(trans, &iter);
  1037. return ret;
  1038. }
  1039. static int ec_stripe_update_extent(struct btree_trans *trans,
  1040. struct bch_dev *ca,
  1041. struct bpos bucket, u8 gen,
  1042. struct ec_stripe_buf *s,
  1043. struct bpos *bp_pos)
  1044. {
  1045. struct bch_stripe *v = &bkey_i_to_stripe(&s->key)->v;
  1046. struct bch_fs *c = trans->c;
  1047. struct bch_backpointer bp;
  1048. struct btree_iter iter;
  1049. struct bkey_s_c k;
  1050. const struct bch_extent_ptr *ptr_c;
  1051. struct bch_extent_ptr *ec_ptr = NULL;
  1052. struct bch_extent_stripe_ptr stripe_ptr;
  1053. struct bkey_i *n;
  1054. int ret, dev, block;
  1055. ret = bch2_get_next_backpointer(trans, ca, bucket, gen,
  1056. bp_pos, &bp, BTREE_ITER_cached);
  1057. if (ret)
  1058. return ret;
  1059. if (bpos_eq(*bp_pos, SPOS_MAX))
  1060. return 0;
  1061. if (bp.level) {
  1062. struct printbuf buf = PRINTBUF;
  1063. struct btree_iter node_iter;
  1064. struct btree *b;
  1065. b = bch2_backpointer_get_node(trans, &node_iter, *bp_pos, bp);
  1066. bch2_trans_iter_exit(trans, &node_iter);
  1067. if (!b)
  1068. return 0;
  1069. prt_printf(&buf, "found btree node in erasure coded bucket: b=%px\n", b);
  1070. bch2_backpointer_to_text(&buf, &bp);
  1071. bch2_fs_inconsistent(c, "%s", buf.buf);
  1072. printbuf_exit(&buf);
  1073. return -EIO;
  1074. }
  1075. k = bch2_backpointer_get_key(trans, &iter, *bp_pos, bp, BTREE_ITER_intent);
  1076. ret = bkey_err(k);
  1077. if (ret)
  1078. return ret;
  1079. if (!k.k) {
  1080. /*
  1081. * extent no longer exists - we could flush the btree
  1082. * write buffer and retry to verify, but no need:
  1083. */
  1084. return 0;
  1085. }
  1086. if (extent_has_stripe_ptr(k, s->key.k.p.offset))
  1087. goto out;
  1088. ptr_c = bkey_matches_stripe(v, k, &block);
  1089. /*
  1090. * It doesn't generally make sense to erasure code cached ptrs:
  1091. * XXX: should we be incrementing a counter?
  1092. */
  1093. if (!ptr_c || ptr_c->cached)
  1094. goto out;
  1095. dev = v->ptrs[block].dev;
  1096. n = bch2_trans_kmalloc(trans, bkey_bytes(k.k) + sizeof(stripe_ptr));
  1097. ret = PTR_ERR_OR_ZERO(n);
  1098. if (ret)
  1099. goto out;
  1100. bkey_reassemble(n, k);
  1101. bch2_bkey_drop_ptrs_noerror(bkey_i_to_s(n), ptr, ptr->dev != dev);
  1102. ec_ptr = bch2_bkey_has_device(bkey_i_to_s(n), dev);
  1103. BUG_ON(!ec_ptr);
  1104. stripe_ptr = (struct bch_extent_stripe_ptr) {
  1105. .type = 1 << BCH_EXTENT_ENTRY_stripe_ptr,
  1106. .block = block,
  1107. .redundancy = v->nr_redundant,
  1108. .idx = s->key.k.p.offset,
  1109. };
  1110. __extent_entry_insert(n,
  1111. (union bch_extent_entry *) ec_ptr,
  1112. (union bch_extent_entry *) &stripe_ptr);
  1113. ret = bch2_trans_update(trans, &iter, n, 0);
  1114. out:
  1115. bch2_trans_iter_exit(trans, &iter);
  1116. return ret;
  1117. }
  1118. static int ec_stripe_update_bucket(struct btree_trans *trans, struct ec_stripe_buf *s,
  1119. unsigned block)
  1120. {
  1121. struct bch_fs *c = trans->c;
  1122. struct bch_stripe *v = &bkey_i_to_stripe(&s->key)->v;
  1123. struct bch_extent_ptr ptr = v->ptrs[block];
  1124. struct bpos bp_pos = POS_MIN;
  1125. int ret = 0;
  1126. struct bch_dev *ca = bch2_dev_tryget(c, ptr.dev);
  1127. if (!ca)
  1128. return -EIO;
  1129. struct bpos bucket_pos = PTR_BUCKET_POS(ca, &ptr);
  1130. while (1) {
  1131. ret = commit_do(trans, NULL, NULL,
  1132. BCH_TRANS_COMMIT_no_check_rw|
  1133. BCH_TRANS_COMMIT_no_enospc,
  1134. ec_stripe_update_extent(trans, ca, bucket_pos, ptr.gen, s, &bp_pos));
  1135. if (ret)
  1136. break;
  1137. if (bkey_eq(bp_pos, POS_MAX))
  1138. break;
  1139. bp_pos = bpos_nosnap_successor(bp_pos);
  1140. }
  1141. bch2_dev_put(ca);
  1142. return ret;
  1143. }
  1144. static int ec_stripe_update_extents(struct bch_fs *c, struct ec_stripe_buf *s)
  1145. {
  1146. struct btree_trans *trans = bch2_trans_get(c);
  1147. struct bch_stripe *v = &bkey_i_to_stripe(&s->key)->v;
  1148. unsigned i, nr_data = v->nr_blocks - v->nr_redundant;
  1149. int ret = 0;
  1150. ret = bch2_btree_write_buffer_flush_sync(trans);
  1151. if (ret)
  1152. goto err;
  1153. for (i = 0; i < nr_data; i++) {
  1154. ret = ec_stripe_update_bucket(trans, s, i);
  1155. if (ret)
  1156. break;
  1157. }
  1158. err:
  1159. bch2_trans_put(trans);
  1160. return ret;
  1161. }
  1162. static void zero_out_rest_of_ec_bucket(struct bch_fs *c,
  1163. struct ec_stripe_new *s,
  1164. unsigned block,
  1165. struct open_bucket *ob)
  1166. {
  1167. struct bch_dev *ca = bch2_dev_get_ioref(c, ob->dev, WRITE);
  1168. if (!ca) {
  1169. s->err = -BCH_ERR_erofs_no_writes;
  1170. return;
  1171. }
  1172. unsigned offset = ca->mi.bucket_size - ob->sectors_free;
  1173. memset(s->new_stripe.data[block] + (offset << 9),
  1174. 0,
  1175. ob->sectors_free << 9);
  1176. int ret = blkdev_issue_zeroout(ca->disk_sb.bdev,
  1177. ob->bucket * ca->mi.bucket_size + offset,
  1178. ob->sectors_free,
  1179. GFP_KERNEL, 0);
  1180. percpu_ref_put(&ca->io_ref);
  1181. if (ret)
  1182. s->err = ret;
  1183. }
  1184. void bch2_ec_stripe_new_free(struct bch_fs *c, struct ec_stripe_new *s)
  1185. {
  1186. if (s->idx)
  1187. bch2_stripe_close(c, s);
  1188. kfree(s);
  1189. }
  1190. /*
  1191. * data buckets of new stripe all written: create the stripe
  1192. */
  1193. static void ec_stripe_create(struct ec_stripe_new *s)
  1194. {
  1195. struct bch_fs *c = s->c;
  1196. struct open_bucket *ob;
  1197. struct bch_stripe *v = &bkey_i_to_stripe(&s->new_stripe.key)->v;
  1198. unsigned i, nr_data = v->nr_blocks - v->nr_redundant;
  1199. int ret;
  1200. BUG_ON(s->h->s == s);
  1201. closure_sync(&s->iodone);
  1202. if (!s->err) {
  1203. for (i = 0; i < nr_data; i++)
  1204. if (s->blocks[i]) {
  1205. ob = c->open_buckets + s->blocks[i];
  1206. if (ob->sectors_free)
  1207. zero_out_rest_of_ec_bucket(c, s, i, ob);
  1208. }
  1209. }
  1210. if (s->err) {
  1211. if (!bch2_err_matches(s->err, EROFS))
  1212. bch_err(c, "error creating stripe: error writing data buckets");
  1213. goto err;
  1214. }
  1215. if (s->have_existing_stripe) {
  1216. ec_validate_checksums(c, &s->existing_stripe);
  1217. if (ec_do_recov(c, &s->existing_stripe)) {
  1218. bch_err(c, "error creating stripe: error reading existing stripe");
  1219. goto err;
  1220. }
  1221. for (i = 0; i < nr_data; i++)
  1222. if (stripe_blockcount_get(&bkey_i_to_stripe(&s->existing_stripe.key)->v, i))
  1223. swap(s->new_stripe.data[i],
  1224. s->existing_stripe.data[i]);
  1225. ec_stripe_buf_exit(&s->existing_stripe);
  1226. }
  1227. BUG_ON(!s->allocated);
  1228. BUG_ON(!s->idx);
  1229. ec_generate_ec(&s->new_stripe);
  1230. ec_generate_checksums(&s->new_stripe);
  1231. /* write p/q: */
  1232. for (i = nr_data; i < v->nr_blocks; i++)
  1233. ec_block_io(c, &s->new_stripe, REQ_OP_WRITE, i, &s->iodone);
  1234. closure_sync(&s->iodone);
  1235. if (ec_nr_failed(&s->new_stripe)) {
  1236. bch_err(c, "error creating stripe: error writing redundancy buckets");
  1237. goto err;
  1238. }
  1239. ret = bch2_trans_commit_do(c, &s->res, NULL,
  1240. BCH_TRANS_COMMIT_no_check_rw|
  1241. BCH_TRANS_COMMIT_no_enospc,
  1242. ec_stripe_key_update(trans,
  1243. s->have_existing_stripe
  1244. ? bkey_i_to_stripe(&s->existing_stripe.key)
  1245. : NULL,
  1246. bkey_i_to_stripe(&s->new_stripe.key)));
  1247. bch_err_msg(c, ret, "creating stripe key");
  1248. if (ret) {
  1249. goto err;
  1250. }
  1251. ret = ec_stripe_update_extents(c, &s->new_stripe);
  1252. bch_err_msg(c, ret, "error updating extents");
  1253. if (ret)
  1254. goto err;
  1255. err:
  1256. bch2_disk_reservation_put(c, &s->res);
  1257. for (i = 0; i < v->nr_blocks; i++)
  1258. if (s->blocks[i]) {
  1259. ob = c->open_buckets + s->blocks[i];
  1260. if (i < nr_data) {
  1261. ob->ec = NULL;
  1262. __bch2_open_bucket_put(c, ob);
  1263. } else {
  1264. bch2_open_bucket_put(c, ob);
  1265. }
  1266. }
  1267. mutex_lock(&c->ec_stripe_new_lock);
  1268. list_del(&s->list);
  1269. mutex_unlock(&c->ec_stripe_new_lock);
  1270. wake_up(&c->ec_stripe_new_wait);
  1271. ec_stripe_buf_exit(&s->existing_stripe);
  1272. ec_stripe_buf_exit(&s->new_stripe);
  1273. closure_debug_destroy(&s->iodone);
  1274. ec_stripe_new_put(c, s, STRIPE_REF_stripe);
  1275. }
  1276. static struct ec_stripe_new *get_pending_stripe(struct bch_fs *c)
  1277. {
  1278. struct ec_stripe_new *s;
  1279. mutex_lock(&c->ec_stripe_new_lock);
  1280. list_for_each_entry(s, &c->ec_stripe_new_list, list)
  1281. if (!atomic_read(&s->ref[STRIPE_REF_io]))
  1282. goto out;
  1283. s = NULL;
  1284. out:
  1285. mutex_unlock(&c->ec_stripe_new_lock);
  1286. return s;
  1287. }
  1288. static void ec_stripe_create_work(struct work_struct *work)
  1289. {
  1290. struct bch_fs *c = container_of(work,
  1291. struct bch_fs, ec_stripe_create_work);
  1292. struct ec_stripe_new *s;
  1293. while ((s = get_pending_stripe(c)))
  1294. ec_stripe_create(s);
  1295. bch2_write_ref_put(c, BCH_WRITE_REF_stripe_create);
  1296. }
  1297. void bch2_ec_do_stripe_creates(struct bch_fs *c)
  1298. {
  1299. bch2_write_ref_get(c, BCH_WRITE_REF_stripe_create);
  1300. if (!queue_work(system_long_wq, &c->ec_stripe_create_work))
  1301. bch2_write_ref_put(c, BCH_WRITE_REF_stripe_create);
  1302. }
  1303. static void ec_stripe_new_set_pending(struct bch_fs *c, struct ec_stripe_head *h)
  1304. {
  1305. struct ec_stripe_new *s = h->s;
  1306. lockdep_assert_held(&h->lock);
  1307. BUG_ON(!s->allocated && !s->err);
  1308. h->s = NULL;
  1309. s->pending = true;
  1310. mutex_lock(&c->ec_stripe_new_lock);
  1311. list_add(&s->list, &c->ec_stripe_new_list);
  1312. mutex_unlock(&c->ec_stripe_new_lock);
  1313. ec_stripe_new_put(c, s, STRIPE_REF_io);
  1314. }
  1315. static void ec_stripe_new_cancel(struct bch_fs *c, struct ec_stripe_head *h, int err)
  1316. {
  1317. h->s->err = err;
  1318. ec_stripe_new_set_pending(c, h);
  1319. }
  1320. void bch2_ec_bucket_cancel(struct bch_fs *c, struct open_bucket *ob)
  1321. {
  1322. struct ec_stripe_new *s = ob->ec;
  1323. s->err = -EIO;
  1324. }
  1325. void *bch2_writepoint_ec_buf(struct bch_fs *c, struct write_point *wp)
  1326. {
  1327. struct open_bucket *ob = ec_open_bucket(c, &wp->ptrs);
  1328. if (!ob)
  1329. return NULL;
  1330. BUG_ON(!ob->ec->new_stripe.data[ob->ec_idx]);
  1331. struct bch_dev *ca = ob_dev(c, ob);
  1332. unsigned offset = ca->mi.bucket_size - ob->sectors_free;
  1333. return ob->ec->new_stripe.data[ob->ec_idx] + (offset << 9);
  1334. }
  1335. static int unsigned_cmp(const void *_l, const void *_r)
  1336. {
  1337. unsigned l = *((const unsigned *) _l);
  1338. unsigned r = *((const unsigned *) _r);
  1339. return cmp_int(l, r);
  1340. }
  1341. /* pick most common bucket size: */
  1342. static unsigned pick_blocksize(struct bch_fs *c,
  1343. struct bch_devs_mask *devs)
  1344. {
  1345. unsigned nr = 0, sizes[BCH_SB_MEMBERS_MAX];
  1346. struct {
  1347. unsigned nr, size;
  1348. } cur = { 0, 0 }, best = { 0, 0 };
  1349. for_each_member_device_rcu(c, ca, devs)
  1350. sizes[nr++] = ca->mi.bucket_size;
  1351. sort(sizes, nr, sizeof(unsigned), unsigned_cmp, NULL);
  1352. for (unsigned i = 0; i < nr; i++) {
  1353. if (sizes[i] != cur.size) {
  1354. if (cur.nr > best.nr)
  1355. best = cur;
  1356. cur.nr = 0;
  1357. cur.size = sizes[i];
  1358. }
  1359. cur.nr++;
  1360. }
  1361. if (cur.nr > best.nr)
  1362. best = cur;
  1363. return best.size;
  1364. }
  1365. static bool may_create_new_stripe(struct bch_fs *c)
  1366. {
  1367. return false;
  1368. }
  1369. static void ec_stripe_key_init(struct bch_fs *c,
  1370. struct bkey_i *k,
  1371. unsigned nr_data,
  1372. unsigned nr_parity,
  1373. unsigned stripe_size,
  1374. unsigned disk_label)
  1375. {
  1376. struct bkey_i_stripe *s = bkey_stripe_init(k);
  1377. unsigned u64s;
  1378. s->v.sectors = cpu_to_le16(stripe_size);
  1379. s->v.algorithm = 0;
  1380. s->v.nr_blocks = nr_data + nr_parity;
  1381. s->v.nr_redundant = nr_parity;
  1382. s->v.csum_granularity_bits = ilog2(c->opts.encoded_extent_max >> 9);
  1383. s->v.csum_type = BCH_CSUM_crc32c;
  1384. s->v.disk_label = disk_label;
  1385. while ((u64s = stripe_val_u64s(&s->v)) > BKEY_VAL_U64s_MAX) {
  1386. BUG_ON(1 << s->v.csum_granularity_bits >=
  1387. le16_to_cpu(s->v.sectors) ||
  1388. s->v.csum_granularity_bits == U8_MAX);
  1389. s->v.csum_granularity_bits++;
  1390. }
  1391. set_bkey_val_u64s(&s->k, u64s);
  1392. }
  1393. static int ec_new_stripe_alloc(struct bch_fs *c, struct ec_stripe_head *h)
  1394. {
  1395. struct ec_stripe_new *s;
  1396. lockdep_assert_held(&h->lock);
  1397. s = kzalloc(sizeof(*s), GFP_KERNEL);
  1398. if (!s)
  1399. return -BCH_ERR_ENOMEM_ec_new_stripe_alloc;
  1400. mutex_init(&s->lock);
  1401. closure_init(&s->iodone, NULL);
  1402. atomic_set(&s->ref[STRIPE_REF_stripe], 1);
  1403. atomic_set(&s->ref[STRIPE_REF_io], 1);
  1404. s->c = c;
  1405. s->h = h;
  1406. s->nr_data = min_t(unsigned, h->nr_active_devs,
  1407. BCH_BKEY_PTRS_MAX) - h->redundancy;
  1408. s->nr_parity = h->redundancy;
  1409. ec_stripe_key_init(c, &s->new_stripe.key,
  1410. s->nr_data, s->nr_parity,
  1411. h->blocksize, h->disk_label);
  1412. h->s = s;
  1413. h->nr_created++;
  1414. return 0;
  1415. }
  1416. static void ec_stripe_head_devs_update(struct bch_fs *c, struct ec_stripe_head *h)
  1417. {
  1418. struct bch_devs_mask devs = h->devs;
  1419. rcu_read_lock();
  1420. h->devs = target_rw_devs(c, BCH_DATA_user, h->disk_label
  1421. ? group_to_target(h->disk_label - 1)
  1422. : 0);
  1423. unsigned nr_devs = dev_mask_nr(&h->devs);
  1424. for_each_member_device_rcu(c, ca, &h->devs)
  1425. if (!ca->mi.durability)
  1426. __clear_bit(ca->dev_idx, h->devs.d);
  1427. unsigned nr_devs_with_durability = dev_mask_nr(&h->devs);
  1428. h->blocksize = pick_blocksize(c, &h->devs);
  1429. h->nr_active_devs = 0;
  1430. for_each_member_device_rcu(c, ca, &h->devs)
  1431. if (ca->mi.bucket_size == h->blocksize)
  1432. h->nr_active_devs++;
  1433. rcu_read_unlock();
  1434. /*
  1435. * If we only have redundancy + 1 devices, we're better off with just
  1436. * replication:
  1437. */
  1438. h->insufficient_devs = h->nr_active_devs < h->redundancy + 2;
  1439. if (h->insufficient_devs) {
  1440. const char *err;
  1441. if (nr_devs < h->redundancy + 2)
  1442. err = NULL;
  1443. else if (nr_devs_with_durability < h->redundancy + 2)
  1444. err = "cannot use durability=0 devices";
  1445. else
  1446. err = "mismatched bucket sizes";
  1447. if (err)
  1448. bch_err(c, "insufficient devices available to create stripe (have %u, need %u): %s",
  1449. h->nr_active_devs, h->redundancy + 2, err);
  1450. }
  1451. struct bch_devs_mask devs_leaving;
  1452. bitmap_andnot(devs_leaving.d, devs.d, h->devs.d, BCH_SB_MEMBERS_MAX);
  1453. if (h->s && !h->s->allocated && dev_mask_nr(&devs_leaving))
  1454. ec_stripe_new_cancel(c, h, -EINTR);
  1455. h->rw_devs_change_count = c->rw_devs_change_count;
  1456. }
  1457. static struct ec_stripe_head *
  1458. ec_new_stripe_head_alloc(struct bch_fs *c, unsigned disk_label,
  1459. unsigned algo, unsigned redundancy,
  1460. enum bch_watermark watermark)
  1461. {
  1462. struct ec_stripe_head *h;
  1463. h = kzalloc(sizeof(*h), GFP_KERNEL);
  1464. if (!h)
  1465. return NULL;
  1466. mutex_init(&h->lock);
  1467. BUG_ON(!mutex_trylock(&h->lock));
  1468. h->disk_label = disk_label;
  1469. h->algo = algo;
  1470. h->redundancy = redundancy;
  1471. h->watermark = watermark;
  1472. list_add(&h->list, &c->ec_stripe_head_list);
  1473. return h;
  1474. }
  1475. void bch2_ec_stripe_head_put(struct bch_fs *c, struct ec_stripe_head *h)
  1476. {
  1477. if (h->s &&
  1478. h->s->allocated &&
  1479. bitmap_weight(h->s->blocks_allocated,
  1480. h->s->nr_data) == h->s->nr_data)
  1481. ec_stripe_new_set_pending(c, h);
  1482. mutex_unlock(&h->lock);
  1483. }
  1484. static struct ec_stripe_head *
  1485. __bch2_ec_stripe_head_get(struct btree_trans *trans,
  1486. unsigned disk_label,
  1487. unsigned algo,
  1488. unsigned redundancy,
  1489. enum bch_watermark watermark)
  1490. {
  1491. struct bch_fs *c = trans->c;
  1492. struct ec_stripe_head *h;
  1493. int ret;
  1494. if (!redundancy)
  1495. return NULL;
  1496. ret = bch2_trans_mutex_lock(trans, &c->ec_stripe_head_lock);
  1497. if (ret)
  1498. return ERR_PTR(ret);
  1499. if (test_bit(BCH_FS_going_ro, &c->flags)) {
  1500. h = ERR_PTR(-BCH_ERR_erofs_no_writes);
  1501. goto err;
  1502. }
  1503. list_for_each_entry(h, &c->ec_stripe_head_list, list)
  1504. if (h->disk_label == disk_label &&
  1505. h->algo == algo &&
  1506. h->redundancy == redundancy &&
  1507. h->watermark == watermark) {
  1508. ret = bch2_trans_mutex_lock(trans, &h->lock);
  1509. if (ret) {
  1510. h = ERR_PTR(ret);
  1511. goto err;
  1512. }
  1513. goto found;
  1514. }
  1515. h = ec_new_stripe_head_alloc(c, disk_label, algo, redundancy, watermark);
  1516. if (!h) {
  1517. h = ERR_PTR(-BCH_ERR_ENOMEM_stripe_head_alloc);
  1518. goto err;
  1519. }
  1520. found:
  1521. if (h->rw_devs_change_count != c->rw_devs_change_count)
  1522. ec_stripe_head_devs_update(c, h);
  1523. if (h->insufficient_devs) {
  1524. mutex_unlock(&h->lock);
  1525. h = NULL;
  1526. }
  1527. err:
  1528. mutex_unlock(&c->ec_stripe_head_lock);
  1529. return h;
  1530. }
  1531. static int new_stripe_alloc_buckets(struct btree_trans *trans, struct ec_stripe_head *h,
  1532. enum bch_watermark watermark, struct closure *cl)
  1533. {
  1534. struct bch_fs *c = trans->c;
  1535. struct bch_devs_mask devs = h->devs;
  1536. struct open_bucket *ob;
  1537. struct open_buckets buckets;
  1538. struct bch_stripe *v = &bkey_i_to_stripe(&h->s->new_stripe.key)->v;
  1539. unsigned i, j, nr_have_parity = 0, nr_have_data = 0;
  1540. bool have_cache = true;
  1541. int ret = 0;
  1542. BUG_ON(v->nr_blocks != h->s->nr_data + h->s->nr_parity);
  1543. BUG_ON(v->nr_redundant != h->s->nr_parity);
  1544. /* * We bypass the sector allocator which normally does this: */
  1545. bitmap_and(devs.d, devs.d, c->rw_devs[BCH_DATA_user].d, BCH_SB_MEMBERS_MAX);
  1546. for_each_set_bit(i, h->s->blocks_gotten, v->nr_blocks) {
  1547. /*
  1548. * Note: we don't yet repair invalid blocks (failed/removed
  1549. * devices) when reusing stripes - we still need a codepath to
  1550. * walk backpointers and update all extents that point to that
  1551. * block when updating the stripe
  1552. */
  1553. if (v->ptrs[i].dev != BCH_SB_MEMBER_INVALID)
  1554. __clear_bit(v->ptrs[i].dev, devs.d);
  1555. if (i < h->s->nr_data)
  1556. nr_have_data++;
  1557. else
  1558. nr_have_parity++;
  1559. }
  1560. BUG_ON(nr_have_data > h->s->nr_data);
  1561. BUG_ON(nr_have_parity > h->s->nr_parity);
  1562. buckets.nr = 0;
  1563. if (nr_have_parity < h->s->nr_parity) {
  1564. ret = bch2_bucket_alloc_set_trans(trans, &buckets,
  1565. &h->parity_stripe,
  1566. &devs,
  1567. h->s->nr_parity,
  1568. &nr_have_parity,
  1569. &have_cache, 0,
  1570. BCH_DATA_parity,
  1571. watermark,
  1572. cl);
  1573. open_bucket_for_each(c, &buckets, ob, i) {
  1574. j = find_next_zero_bit(h->s->blocks_gotten,
  1575. h->s->nr_data + h->s->nr_parity,
  1576. h->s->nr_data);
  1577. BUG_ON(j >= h->s->nr_data + h->s->nr_parity);
  1578. h->s->blocks[j] = buckets.v[i];
  1579. v->ptrs[j] = bch2_ob_ptr(c, ob);
  1580. __set_bit(j, h->s->blocks_gotten);
  1581. }
  1582. if (ret)
  1583. return ret;
  1584. }
  1585. buckets.nr = 0;
  1586. if (nr_have_data < h->s->nr_data) {
  1587. ret = bch2_bucket_alloc_set_trans(trans, &buckets,
  1588. &h->block_stripe,
  1589. &devs,
  1590. h->s->nr_data,
  1591. &nr_have_data,
  1592. &have_cache, 0,
  1593. BCH_DATA_user,
  1594. watermark,
  1595. cl);
  1596. open_bucket_for_each(c, &buckets, ob, i) {
  1597. j = find_next_zero_bit(h->s->blocks_gotten,
  1598. h->s->nr_data, 0);
  1599. BUG_ON(j >= h->s->nr_data);
  1600. h->s->blocks[j] = buckets.v[i];
  1601. v->ptrs[j] = bch2_ob_ptr(c, ob);
  1602. __set_bit(j, h->s->blocks_gotten);
  1603. }
  1604. if (ret)
  1605. return ret;
  1606. }
  1607. return 0;
  1608. }
  1609. static s64 get_existing_stripe(struct bch_fs *c,
  1610. struct ec_stripe_head *head)
  1611. {
  1612. ec_stripes_heap *h = &c->ec_stripes_heap;
  1613. struct stripe *m;
  1614. size_t heap_idx;
  1615. u64 stripe_idx;
  1616. s64 ret = -1;
  1617. if (may_create_new_stripe(c))
  1618. return -1;
  1619. mutex_lock(&c->ec_stripes_heap_lock);
  1620. for (heap_idx = 0; heap_idx < h->nr; heap_idx++) {
  1621. /* No blocks worth reusing, stripe will just be deleted: */
  1622. if (!h->data[heap_idx].blocks_nonempty)
  1623. continue;
  1624. stripe_idx = h->data[heap_idx].idx;
  1625. m = genradix_ptr(&c->stripes, stripe_idx);
  1626. if (m->disk_label == head->disk_label &&
  1627. m->algorithm == head->algo &&
  1628. m->nr_redundant == head->redundancy &&
  1629. m->sectors == head->blocksize &&
  1630. m->blocks_nonempty < m->nr_blocks - m->nr_redundant &&
  1631. bch2_try_open_stripe(c, head->s, stripe_idx)) {
  1632. ret = stripe_idx;
  1633. break;
  1634. }
  1635. }
  1636. mutex_unlock(&c->ec_stripes_heap_lock);
  1637. return ret;
  1638. }
  1639. static int __bch2_ec_stripe_head_reuse(struct btree_trans *trans, struct ec_stripe_head *h)
  1640. {
  1641. struct bch_fs *c = trans->c;
  1642. struct bch_stripe *new_v = &bkey_i_to_stripe(&h->s->new_stripe.key)->v;
  1643. struct bch_stripe *existing_v;
  1644. unsigned i;
  1645. s64 idx;
  1646. int ret;
  1647. /*
  1648. * If we can't allocate a new stripe, and there's no stripes with empty
  1649. * blocks for us to reuse, that means we have to wait on copygc:
  1650. */
  1651. idx = get_existing_stripe(c, h);
  1652. if (idx < 0)
  1653. return -BCH_ERR_stripe_alloc_blocked;
  1654. ret = get_stripe_key_trans(trans, idx, &h->s->existing_stripe);
  1655. bch2_fs_fatal_err_on(ret && !bch2_err_matches(ret, BCH_ERR_transaction_restart), c,
  1656. "reading stripe key: %s", bch2_err_str(ret));
  1657. if (ret) {
  1658. bch2_stripe_close(c, h->s);
  1659. return ret;
  1660. }
  1661. existing_v = &bkey_i_to_stripe(&h->s->existing_stripe.key)->v;
  1662. BUG_ON(existing_v->nr_redundant != h->s->nr_parity);
  1663. h->s->nr_data = existing_v->nr_blocks -
  1664. existing_v->nr_redundant;
  1665. ret = ec_stripe_buf_init(&h->s->existing_stripe, 0, h->blocksize);
  1666. if (ret) {
  1667. bch2_stripe_close(c, h->s);
  1668. return ret;
  1669. }
  1670. BUG_ON(h->s->existing_stripe.size != h->blocksize);
  1671. BUG_ON(h->s->existing_stripe.size != le16_to_cpu(existing_v->sectors));
  1672. /*
  1673. * Free buckets we initially allocated - they might conflict with
  1674. * blocks from the stripe we're reusing:
  1675. */
  1676. for_each_set_bit(i, h->s->blocks_gotten, new_v->nr_blocks) {
  1677. bch2_open_bucket_put(c, c->open_buckets + h->s->blocks[i]);
  1678. h->s->blocks[i] = 0;
  1679. }
  1680. memset(h->s->blocks_gotten, 0, sizeof(h->s->blocks_gotten));
  1681. memset(h->s->blocks_allocated, 0, sizeof(h->s->blocks_allocated));
  1682. for (i = 0; i < existing_v->nr_blocks; i++) {
  1683. if (stripe_blockcount_get(existing_v, i)) {
  1684. __set_bit(i, h->s->blocks_gotten);
  1685. __set_bit(i, h->s->blocks_allocated);
  1686. }
  1687. ec_block_io(c, &h->s->existing_stripe, READ, i, &h->s->iodone);
  1688. }
  1689. bkey_copy(&h->s->new_stripe.key, &h->s->existing_stripe.key);
  1690. h->s->have_existing_stripe = true;
  1691. return 0;
  1692. }
  1693. static int __bch2_ec_stripe_head_reserve(struct btree_trans *trans, struct ec_stripe_head *h)
  1694. {
  1695. struct bch_fs *c = trans->c;
  1696. struct btree_iter iter;
  1697. struct bkey_s_c k;
  1698. struct bpos min_pos = POS(0, 1);
  1699. struct bpos start_pos = bpos_max(min_pos, POS(0, c->ec_stripe_hint));
  1700. int ret;
  1701. if (!h->s->res.sectors) {
  1702. ret = bch2_disk_reservation_get(c, &h->s->res,
  1703. h->blocksize,
  1704. h->s->nr_parity,
  1705. BCH_DISK_RESERVATION_NOFAIL);
  1706. if (ret)
  1707. return ret;
  1708. }
  1709. for_each_btree_key_norestart(trans, iter, BTREE_ID_stripes, start_pos,
  1710. BTREE_ITER_slots|BTREE_ITER_intent, k, ret) {
  1711. if (bkey_gt(k.k->p, POS(0, U32_MAX))) {
  1712. if (start_pos.offset) {
  1713. start_pos = min_pos;
  1714. bch2_btree_iter_set_pos(&iter, start_pos);
  1715. continue;
  1716. }
  1717. ret = -BCH_ERR_ENOSPC_stripe_create;
  1718. break;
  1719. }
  1720. if (bkey_deleted(k.k) &&
  1721. bch2_try_open_stripe(c, h->s, k.k->p.offset))
  1722. break;
  1723. }
  1724. c->ec_stripe_hint = iter.pos.offset;
  1725. if (ret)
  1726. goto err;
  1727. ret = ec_stripe_mem_alloc(trans, &iter);
  1728. if (ret) {
  1729. bch2_stripe_close(c, h->s);
  1730. goto err;
  1731. }
  1732. h->s->new_stripe.key.k.p = iter.pos;
  1733. out:
  1734. bch2_trans_iter_exit(trans, &iter);
  1735. return ret;
  1736. err:
  1737. bch2_disk_reservation_put(c, &h->s->res);
  1738. goto out;
  1739. }
  1740. struct ec_stripe_head *bch2_ec_stripe_head_get(struct btree_trans *trans,
  1741. unsigned target,
  1742. unsigned algo,
  1743. unsigned redundancy,
  1744. enum bch_watermark watermark,
  1745. struct closure *cl)
  1746. {
  1747. struct bch_fs *c = trans->c;
  1748. struct ec_stripe_head *h;
  1749. bool waiting = false;
  1750. unsigned disk_label = 0;
  1751. struct target t = target_decode(target);
  1752. int ret;
  1753. if (t.type == TARGET_GROUP) {
  1754. if (t.group > U8_MAX) {
  1755. bch_err(c, "cannot create a stripe when disk_label > U8_MAX");
  1756. return NULL;
  1757. }
  1758. disk_label = t.group + 1; /* 0 == no label */
  1759. }
  1760. h = __bch2_ec_stripe_head_get(trans, disk_label, algo, redundancy, watermark);
  1761. if (IS_ERR_OR_NULL(h))
  1762. return h;
  1763. if (!h->s) {
  1764. ret = ec_new_stripe_alloc(c, h);
  1765. if (ret) {
  1766. bch_err(c, "failed to allocate new stripe");
  1767. goto err;
  1768. }
  1769. }
  1770. if (h->s->allocated)
  1771. goto allocated;
  1772. if (h->s->have_existing_stripe)
  1773. goto alloc_existing;
  1774. /* First, try to allocate a full stripe: */
  1775. ret = new_stripe_alloc_buckets(trans, h, BCH_WATERMARK_stripe, NULL) ?:
  1776. __bch2_ec_stripe_head_reserve(trans, h);
  1777. if (!ret)
  1778. goto allocate_buf;
  1779. if (bch2_err_matches(ret, BCH_ERR_transaction_restart) ||
  1780. bch2_err_matches(ret, ENOMEM))
  1781. goto err;
  1782. /*
  1783. * Not enough buckets available for a full stripe: we must reuse an
  1784. * existing stripe:
  1785. */
  1786. while (1) {
  1787. ret = __bch2_ec_stripe_head_reuse(trans, h);
  1788. if (!ret)
  1789. break;
  1790. if (waiting || !cl || ret != -BCH_ERR_stripe_alloc_blocked)
  1791. goto err;
  1792. if (watermark == BCH_WATERMARK_copygc) {
  1793. ret = new_stripe_alloc_buckets(trans, h, watermark, NULL) ?:
  1794. __bch2_ec_stripe_head_reserve(trans, h);
  1795. if (ret)
  1796. goto err;
  1797. goto allocate_buf;
  1798. }
  1799. /* XXX freelist_wait? */
  1800. closure_wait(&c->freelist_wait, cl);
  1801. waiting = true;
  1802. }
  1803. if (waiting)
  1804. closure_wake_up(&c->freelist_wait);
  1805. alloc_existing:
  1806. /*
  1807. * Retry allocating buckets, with the watermark for this
  1808. * particular write:
  1809. */
  1810. ret = new_stripe_alloc_buckets(trans, h, watermark, cl);
  1811. if (ret)
  1812. goto err;
  1813. allocate_buf:
  1814. ret = ec_stripe_buf_init(&h->s->new_stripe, 0, h->blocksize);
  1815. if (ret)
  1816. goto err;
  1817. h->s->allocated = true;
  1818. allocated:
  1819. BUG_ON(!h->s->idx);
  1820. BUG_ON(!h->s->new_stripe.data[0]);
  1821. BUG_ON(trans->restarted);
  1822. return h;
  1823. err:
  1824. bch2_ec_stripe_head_put(c, h);
  1825. return ERR_PTR(ret);
  1826. }
  1827. /* device removal */
  1828. static int bch2_invalidate_stripe_to_dev(struct btree_trans *trans, struct bkey_s_c k_a)
  1829. {
  1830. struct bch_alloc_v4 a_convert;
  1831. const struct bch_alloc_v4 *a = bch2_alloc_to_v4(k_a, &a_convert);
  1832. if (!a->stripe)
  1833. return 0;
  1834. if (a->stripe_sectors) {
  1835. bch_err(trans->c, "trying to invalidate device in stripe when bucket has stripe data");
  1836. return -BCH_ERR_invalidate_stripe_to_dev;
  1837. }
  1838. struct btree_iter iter;
  1839. struct bkey_i_stripe *s =
  1840. bch2_bkey_get_mut_typed(trans, &iter, BTREE_ID_stripes, POS(0, a->stripe),
  1841. BTREE_ITER_slots, stripe);
  1842. int ret = PTR_ERR_OR_ZERO(s);
  1843. if (ret)
  1844. return ret;
  1845. struct disk_accounting_pos acc = {
  1846. .type = BCH_DISK_ACCOUNTING_replicas,
  1847. };
  1848. s64 sectors = 0;
  1849. for (unsigned i = 0; i < s->v.nr_blocks; i++)
  1850. sectors -= stripe_blockcount_get(&s->v, i);
  1851. bch2_bkey_to_replicas(&acc.replicas, bkey_i_to_s_c(&s->k_i));
  1852. acc.replicas.data_type = BCH_DATA_user;
  1853. ret = bch2_disk_accounting_mod(trans, &acc, &sectors, 1, false);
  1854. if (ret)
  1855. goto err;
  1856. struct bkey_ptrs ptrs = bch2_bkey_ptrs(bkey_i_to_s(&s->k_i));
  1857. bkey_for_each_ptr(ptrs, ptr)
  1858. if (ptr->dev == k_a.k->p.inode)
  1859. ptr->dev = BCH_SB_MEMBER_INVALID;
  1860. sectors = -sectors;
  1861. bch2_bkey_to_replicas(&acc.replicas, bkey_i_to_s_c(&s->k_i));
  1862. acc.replicas.data_type = BCH_DATA_user;
  1863. ret = bch2_disk_accounting_mod(trans, &acc, &sectors, 1, false);
  1864. if (ret)
  1865. goto err;
  1866. err:
  1867. bch2_trans_iter_exit(trans, &iter);
  1868. return ret;
  1869. }
  1870. int bch2_dev_remove_stripes(struct bch_fs *c, unsigned dev_idx)
  1871. {
  1872. return bch2_trans_run(c,
  1873. for_each_btree_key_upto_commit(trans, iter,
  1874. BTREE_ID_alloc, POS(dev_idx, 0), POS(dev_idx, U64_MAX),
  1875. BTREE_ITER_intent, k,
  1876. NULL, NULL, 0, ({
  1877. bch2_invalidate_stripe_to_dev(trans, k);
  1878. })));
  1879. }
  1880. /* startup/shutdown */
  1881. static void __bch2_ec_stop(struct bch_fs *c, struct bch_dev *ca)
  1882. {
  1883. struct ec_stripe_head *h;
  1884. struct open_bucket *ob;
  1885. unsigned i;
  1886. mutex_lock(&c->ec_stripe_head_lock);
  1887. list_for_each_entry(h, &c->ec_stripe_head_list, list) {
  1888. mutex_lock(&h->lock);
  1889. if (!h->s)
  1890. goto unlock;
  1891. if (!ca)
  1892. goto found;
  1893. for (i = 0; i < bkey_i_to_stripe(&h->s->new_stripe.key)->v.nr_blocks; i++) {
  1894. if (!h->s->blocks[i])
  1895. continue;
  1896. ob = c->open_buckets + h->s->blocks[i];
  1897. if (ob->dev == ca->dev_idx)
  1898. goto found;
  1899. }
  1900. goto unlock;
  1901. found:
  1902. ec_stripe_new_cancel(c, h, -BCH_ERR_erofs_no_writes);
  1903. unlock:
  1904. mutex_unlock(&h->lock);
  1905. }
  1906. mutex_unlock(&c->ec_stripe_head_lock);
  1907. }
  1908. void bch2_ec_stop_dev(struct bch_fs *c, struct bch_dev *ca)
  1909. {
  1910. __bch2_ec_stop(c, ca);
  1911. }
  1912. void bch2_fs_ec_stop(struct bch_fs *c)
  1913. {
  1914. __bch2_ec_stop(c, NULL);
  1915. }
  1916. static bool bch2_fs_ec_flush_done(struct bch_fs *c)
  1917. {
  1918. bool ret;
  1919. mutex_lock(&c->ec_stripe_new_lock);
  1920. ret = list_empty(&c->ec_stripe_new_list);
  1921. mutex_unlock(&c->ec_stripe_new_lock);
  1922. return ret;
  1923. }
  1924. void bch2_fs_ec_flush(struct bch_fs *c)
  1925. {
  1926. wait_event(c->ec_stripe_new_wait, bch2_fs_ec_flush_done(c));
  1927. }
  1928. int bch2_stripes_read(struct bch_fs *c)
  1929. {
  1930. int ret = bch2_trans_run(c,
  1931. for_each_btree_key(trans, iter, BTREE_ID_stripes, POS_MIN,
  1932. BTREE_ITER_prefetch, k, ({
  1933. if (k.k->type != KEY_TYPE_stripe)
  1934. continue;
  1935. ret = __ec_stripe_mem_alloc(c, k.k->p.offset, GFP_KERNEL);
  1936. if (ret)
  1937. break;
  1938. struct stripe *m = genradix_ptr(&c->stripes, k.k->p.offset);
  1939. stripe_to_mem(m, bkey_s_c_to_stripe(k).v);
  1940. bch2_stripes_heap_insert(c, m, k.k->p.offset);
  1941. 0;
  1942. })));
  1943. bch_err_fn(c, ret);
  1944. return ret;
  1945. }
  1946. void bch2_stripes_heap_to_text(struct printbuf *out, struct bch_fs *c)
  1947. {
  1948. ec_stripes_heap *h = &c->ec_stripes_heap;
  1949. struct stripe *m;
  1950. size_t i;
  1951. mutex_lock(&c->ec_stripes_heap_lock);
  1952. for (i = 0; i < min_t(size_t, h->nr, 50); i++) {
  1953. m = genradix_ptr(&c->stripes, h->data[i].idx);
  1954. prt_printf(out, "%zu %u/%u+%u", h->data[i].idx,
  1955. h->data[i].blocks_nonempty,
  1956. m->nr_blocks - m->nr_redundant,
  1957. m->nr_redundant);
  1958. if (bch2_stripe_is_open(c, h->data[i].idx))
  1959. prt_str(out, " open");
  1960. prt_newline(out);
  1961. }
  1962. mutex_unlock(&c->ec_stripes_heap_lock);
  1963. }
  1964. static void bch2_new_stripe_to_text(struct printbuf *out, struct bch_fs *c,
  1965. struct ec_stripe_new *s)
  1966. {
  1967. prt_printf(out, "\tidx %llu blocks %u+%u allocated %u ref %u %u %s obs",
  1968. s->idx, s->nr_data, s->nr_parity,
  1969. bitmap_weight(s->blocks_allocated, s->nr_data),
  1970. atomic_read(&s->ref[STRIPE_REF_io]),
  1971. atomic_read(&s->ref[STRIPE_REF_stripe]),
  1972. bch2_watermarks[s->h->watermark]);
  1973. struct bch_stripe *v = &bkey_i_to_stripe(&s->new_stripe.key)->v;
  1974. unsigned i;
  1975. for_each_set_bit(i, s->blocks_gotten, v->nr_blocks)
  1976. prt_printf(out, " %u", s->blocks[i]);
  1977. prt_newline(out);
  1978. bch2_bkey_val_to_text(out, c, bkey_i_to_s_c(&s->new_stripe.key));
  1979. prt_newline(out);
  1980. }
  1981. void bch2_new_stripes_to_text(struct printbuf *out, struct bch_fs *c)
  1982. {
  1983. struct ec_stripe_head *h;
  1984. struct ec_stripe_new *s;
  1985. mutex_lock(&c->ec_stripe_head_lock);
  1986. list_for_each_entry(h, &c->ec_stripe_head_list, list) {
  1987. prt_printf(out, "disk label %u algo %u redundancy %u %s nr created %llu:\n",
  1988. h->disk_label, h->algo, h->redundancy,
  1989. bch2_watermarks[h->watermark],
  1990. h->nr_created);
  1991. if (h->s)
  1992. bch2_new_stripe_to_text(out, c, h->s);
  1993. }
  1994. mutex_unlock(&c->ec_stripe_head_lock);
  1995. prt_printf(out, "in flight:\n");
  1996. mutex_lock(&c->ec_stripe_new_lock);
  1997. list_for_each_entry(s, &c->ec_stripe_new_list, list)
  1998. bch2_new_stripe_to_text(out, c, s);
  1999. mutex_unlock(&c->ec_stripe_new_lock);
  2000. }
  2001. void bch2_fs_ec_exit(struct bch_fs *c)
  2002. {
  2003. struct ec_stripe_head *h;
  2004. unsigned i;
  2005. while (1) {
  2006. mutex_lock(&c->ec_stripe_head_lock);
  2007. h = list_first_entry_or_null(&c->ec_stripe_head_list,
  2008. struct ec_stripe_head, list);
  2009. if (h)
  2010. list_del(&h->list);
  2011. mutex_unlock(&c->ec_stripe_head_lock);
  2012. if (!h)
  2013. break;
  2014. if (h->s) {
  2015. for (i = 0; i < bkey_i_to_stripe(&h->s->new_stripe.key)->v.nr_blocks; i++)
  2016. BUG_ON(h->s->blocks[i]);
  2017. kfree(h->s);
  2018. }
  2019. kfree(h);
  2020. }
  2021. BUG_ON(!list_empty(&c->ec_stripe_new_list));
  2022. free_heap(&c->ec_stripes_heap);
  2023. genradix_free(&c->stripes);
  2024. bioset_exit(&c->ec_bioset);
  2025. }
  2026. void bch2_fs_ec_init_early(struct bch_fs *c)
  2027. {
  2028. spin_lock_init(&c->ec_stripes_new_lock);
  2029. mutex_init(&c->ec_stripes_heap_lock);
  2030. INIT_LIST_HEAD(&c->ec_stripe_head_list);
  2031. mutex_init(&c->ec_stripe_head_lock);
  2032. INIT_LIST_HEAD(&c->ec_stripe_new_list);
  2033. mutex_init(&c->ec_stripe_new_lock);
  2034. init_waitqueue_head(&c->ec_stripe_new_wait);
  2035. INIT_WORK(&c->ec_stripe_create_work, ec_stripe_create_work);
  2036. INIT_WORK(&c->ec_stripe_delete_work, ec_stripe_delete_work);
  2037. }
  2038. int bch2_fs_ec_init(struct bch_fs *c)
  2039. {
  2040. return bioset_init(&c->ec_bioset, 1, offsetof(struct ec_bio, bio),
  2041. BIOSET_NEED_BVECS);
  2042. }