data_update.c 21 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763
  1. // SPDX-License-Identifier: GPL-2.0
  2. #include "bcachefs.h"
  3. #include "alloc_foreground.h"
  4. #include "bkey_buf.h"
  5. #include "btree_update.h"
  6. #include "buckets.h"
  7. #include "compress.h"
  8. #include "data_update.h"
  9. #include "disk_groups.h"
  10. #include "ec.h"
  11. #include "error.h"
  12. #include "extents.h"
  13. #include "io_write.h"
  14. #include "keylist.h"
  15. #include "move.h"
  16. #include "nocow_locking.h"
  17. #include "rebalance.h"
  18. #include "snapshot.h"
  19. #include "subvolume.h"
  20. #include "trace.h"
  21. static void bkey_put_dev_refs(struct bch_fs *c, struct bkey_s_c k)
  22. {
  23. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  24. bkey_for_each_ptr(ptrs, ptr)
  25. bch2_dev_put(bch2_dev_have_ref(c, ptr->dev));
  26. }
  27. static bool bkey_get_dev_refs(struct bch_fs *c, struct bkey_s_c k)
  28. {
  29. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  30. bkey_for_each_ptr(ptrs, ptr) {
  31. if (!bch2_dev_tryget(c, ptr->dev)) {
  32. bkey_for_each_ptr(ptrs, ptr2) {
  33. if (ptr2 == ptr)
  34. break;
  35. bch2_dev_put(bch2_dev_have_ref(c, ptr2->dev));
  36. }
  37. return false;
  38. }
  39. }
  40. return true;
  41. }
  42. static void bkey_nocow_unlock(struct bch_fs *c, struct bkey_s_c k)
  43. {
  44. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  45. bkey_for_each_ptr(ptrs, ptr) {
  46. struct bch_dev *ca = bch2_dev_have_ref(c, ptr->dev);
  47. struct bpos bucket = PTR_BUCKET_POS(ca, ptr);
  48. bch2_bucket_nocow_unlock(&c->nocow_locks, bucket, 0);
  49. }
  50. }
  51. static bool bkey_nocow_lock(struct bch_fs *c, struct moving_context *ctxt, struct bkey_s_c k)
  52. {
  53. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  54. bkey_for_each_ptr(ptrs, ptr) {
  55. struct bch_dev *ca = bch2_dev_have_ref(c, ptr->dev);
  56. struct bpos bucket = PTR_BUCKET_POS(ca, ptr);
  57. if (ctxt) {
  58. bool locked;
  59. move_ctxt_wait_event(ctxt,
  60. (locked = bch2_bucket_nocow_trylock(&c->nocow_locks, bucket, 0)) ||
  61. list_empty(&ctxt->ios));
  62. if (!locked)
  63. bch2_bucket_nocow_lock(&c->nocow_locks, bucket, 0);
  64. } else {
  65. if (!bch2_bucket_nocow_trylock(&c->nocow_locks, bucket, 0)) {
  66. bkey_for_each_ptr(ptrs, ptr2) {
  67. if (ptr2 == ptr)
  68. break;
  69. ca = bch2_dev_have_ref(c, ptr2->dev);
  70. bucket = PTR_BUCKET_POS(ca, ptr2);
  71. bch2_bucket_nocow_unlock(&c->nocow_locks, bucket, 0);
  72. }
  73. return false;
  74. }
  75. }
  76. }
  77. return true;
  78. }
  79. static void trace_move_extent_finish2(struct bch_fs *c, struct bkey_s_c k)
  80. {
  81. if (trace_move_extent_finish_enabled()) {
  82. struct printbuf buf = PRINTBUF;
  83. bch2_bkey_val_to_text(&buf, c, k);
  84. trace_move_extent_finish(c, buf.buf);
  85. printbuf_exit(&buf);
  86. }
  87. }
  88. static void trace_move_extent_fail2(struct data_update *m,
  89. struct bkey_s_c new,
  90. struct bkey_s_c wrote,
  91. struct bkey_i *insert,
  92. const char *msg)
  93. {
  94. struct bch_fs *c = m->op.c;
  95. struct bkey_s_c old = bkey_i_to_s_c(m->k.k);
  96. const union bch_extent_entry *entry;
  97. struct bch_extent_ptr *ptr;
  98. struct extent_ptr_decoded p;
  99. struct printbuf buf = PRINTBUF;
  100. unsigned i, rewrites_found = 0;
  101. if (!trace_move_extent_fail_enabled())
  102. return;
  103. prt_str(&buf, msg);
  104. if (insert) {
  105. i = 0;
  106. bkey_for_each_ptr_decode(old.k, bch2_bkey_ptrs_c(old), p, entry) {
  107. if (((1U << i) & m->data_opts.rewrite_ptrs) &&
  108. (ptr = bch2_extent_has_ptr(old, p, bkey_i_to_s(insert))) &&
  109. !ptr->cached)
  110. rewrites_found |= 1U << i;
  111. i++;
  112. }
  113. }
  114. prt_printf(&buf, "\nrewrite ptrs: %u%u%u%u",
  115. (m->data_opts.rewrite_ptrs & (1 << 0)) != 0,
  116. (m->data_opts.rewrite_ptrs & (1 << 1)) != 0,
  117. (m->data_opts.rewrite_ptrs & (1 << 2)) != 0,
  118. (m->data_opts.rewrite_ptrs & (1 << 3)) != 0);
  119. prt_printf(&buf, "\nrewrites found: %u%u%u%u",
  120. (rewrites_found & (1 << 0)) != 0,
  121. (rewrites_found & (1 << 1)) != 0,
  122. (rewrites_found & (1 << 2)) != 0,
  123. (rewrites_found & (1 << 3)) != 0);
  124. prt_str(&buf, "\nold: ");
  125. bch2_bkey_val_to_text(&buf, c, old);
  126. prt_str(&buf, "\nnew: ");
  127. bch2_bkey_val_to_text(&buf, c, new);
  128. prt_str(&buf, "\nwrote: ");
  129. bch2_bkey_val_to_text(&buf, c, wrote);
  130. if (insert) {
  131. prt_str(&buf, "\ninsert: ");
  132. bch2_bkey_val_to_text(&buf, c, bkey_i_to_s_c(insert));
  133. }
  134. trace_move_extent_fail(c, buf.buf);
  135. printbuf_exit(&buf);
  136. }
  137. static int __bch2_data_update_index_update(struct btree_trans *trans,
  138. struct bch_write_op *op)
  139. {
  140. struct bch_fs *c = op->c;
  141. struct btree_iter iter;
  142. struct data_update *m =
  143. container_of(op, struct data_update, op);
  144. struct keylist *keys = &op->insert_keys;
  145. struct bkey_buf _new, _insert;
  146. int ret = 0;
  147. bch2_bkey_buf_init(&_new);
  148. bch2_bkey_buf_init(&_insert);
  149. bch2_bkey_buf_realloc(&_insert, c, U8_MAX);
  150. bch2_trans_iter_init(trans, &iter, m->btree_id,
  151. bkey_start_pos(&bch2_keylist_front(keys)->k),
  152. BTREE_ITER_slots|BTREE_ITER_intent);
  153. while (1) {
  154. struct bkey_s_c k;
  155. struct bkey_s_c old = bkey_i_to_s_c(m->k.k);
  156. struct bkey_i *insert = NULL;
  157. struct bkey_i_extent *new;
  158. const union bch_extent_entry *entry_c;
  159. union bch_extent_entry *entry;
  160. struct extent_ptr_decoded p;
  161. struct bch_extent_ptr *ptr;
  162. const struct bch_extent_ptr *ptr_c;
  163. struct bpos next_pos;
  164. bool should_check_enospc;
  165. s64 i_sectors_delta = 0, disk_sectors_delta = 0;
  166. unsigned rewrites_found = 0, durability, i;
  167. bch2_trans_begin(trans);
  168. k = bch2_btree_iter_peek_slot(&iter);
  169. ret = bkey_err(k);
  170. if (ret)
  171. goto err;
  172. new = bkey_i_to_extent(bch2_keylist_front(keys));
  173. if (!bch2_extents_match(k, old)) {
  174. trace_move_extent_fail2(m, k, bkey_i_to_s_c(&new->k_i),
  175. NULL, "no match:");
  176. goto nowork;
  177. }
  178. bkey_reassemble(_insert.k, k);
  179. insert = _insert.k;
  180. bch2_bkey_buf_copy(&_new, c, bch2_keylist_front(keys));
  181. new = bkey_i_to_extent(_new.k);
  182. bch2_cut_front(iter.pos, &new->k_i);
  183. bch2_cut_front(iter.pos, insert);
  184. bch2_cut_back(new->k.p, insert);
  185. bch2_cut_back(insert->k.p, &new->k_i);
  186. /*
  187. * @old: extent that we read from
  188. * @insert: key that we're going to update, initialized from
  189. * extent currently in btree - same as @old unless we raced with
  190. * other updates
  191. * @new: extent with new pointers that we'll be adding to @insert
  192. *
  193. * Fist, drop rewrite_ptrs from @new:
  194. */
  195. i = 0;
  196. bkey_for_each_ptr_decode(old.k, bch2_bkey_ptrs_c(old), p, entry_c) {
  197. if (((1U << i) & m->data_opts.rewrite_ptrs) &&
  198. (ptr = bch2_extent_has_ptr(old, p, bkey_i_to_s(insert))) &&
  199. !ptr->cached) {
  200. bch2_extent_ptr_set_cached(c, &m->op.opts,
  201. bkey_i_to_s(insert), ptr);
  202. rewrites_found |= 1U << i;
  203. }
  204. i++;
  205. }
  206. if (m->data_opts.rewrite_ptrs &&
  207. !rewrites_found &&
  208. bch2_bkey_durability(c, k) >= m->op.opts.data_replicas) {
  209. trace_move_extent_fail2(m, k, bkey_i_to_s_c(&new->k_i), insert, "no rewrites found:");
  210. goto nowork;
  211. }
  212. /*
  213. * A replica that we just wrote might conflict with a replica
  214. * that we want to keep, due to racing with another move:
  215. */
  216. restart_drop_conflicting_replicas:
  217. extent_for_each_ptr(extent_i_to_s(new), ptr)
  218. if ((ptr_c = bch2_bkey_has_device_c(bkey_i_to_s_c(insert), ptr->dev)) &&
  219. !ptr_c->cached) {
  220. bch2_bkey_drop_ptr_noerror(bkey_i_to_s(&new->k_i), ptr);
  221. goto restart_drop_conflicting_replicas;
  222. }
  223. if (!bkey_val_u64s(&new->k)) {
  224. trace_move_extent_fail2(m, k, bkey_i_to_s_c(&new->k_i), insert, "new replicas conflicted:");
  225. goto nowork;
  226. }
  227. /* Now, drop pointers that conflict with what we just wrote: */
  228. extent_for_each_ptr_decode(extent_i_to_s(new), p, entry)
  229. if ((ptr = bch2_bkey_has_device(bkey_i_to_s(insert), p.ptr.dev)))
  230. bch2_bkey_drop_ptr_noerror(bkey_i_to_s(insert), ptr);
  231. durability = bch2_bkey_durability(c, bkey_i_to_s_c(insert)) +
  232. bch2_bkey_durability(c, bkey_i_to_s_c(&new->k_i));
  233. /* Now, drop excess replicas: */
  234. rcu_read_lock();
  235. restart_drop_extra_replicas:
  236. bkey_for_each_ptr_decode(old.k, bch2_bkey_ptrs(bkey_i_to_s(insert)), p, entry) {
  237. unsigned ptr_durability = bch2_extent_ptr_durability(c, &p);
  238. if (!p.ptr.cached &&
  239. durability - ptr_durability >= m->op.opts.data_replicas) {
  240. durability -= ptr_durability;
  241. bch2_extent_ptr_set_cached(c, &m->op.opts,
  242. bkey_i_to_s(insert), &entry->ptr);
  243. goto restart_drop_extra_replicas;
  244. }
  245. }
  246. rcu_read_unlock();
  247. /* Finally, add the pointers we just wrote: */
  248. extent_for_each_ptr_decode(extent_i_to_s(new), p, entry)
  249. bch2_extent_ptr_decoded_append(insert, &p);
  250. bch2_bkey_narrow_crcs(insert, (struct bch_extent_crc_unpacked) { 0 });
  251. bch2_extent_normalize_by_opts(c, &m->op.opts, bkey_i_to_s(insert));
  252. ret = bch2_sum_sector_overwrites(trans, &iter, insert,
  253. &should_check_enospc,
  254. &i_sectors_delta,
  255. &disk_sectors_delta);
  256. if (ret)
  257. goto err;
  258. if (disk_sectors_delta > (s64) op->res.sectors) {
  259. ret = bch2_disk_reservation_add(c, &op->res,
  260. disk_sectors_delta - op->res.sectors,
  261. !should_check_enospc
  262. ? BCH_DISK_RESERVATION_NOFAIL : 0);
  263. if (ret)
  264. goto out;
  265. }
  266. next_pos = insert->k.p;
  267. /*
  268. * Check for nonce offset inconsistency:
  269. * This is debug code - we've been seeing this bug rarely, and
  270. * it's been hard to reproduce, so this should give us some more
  271. * information when it does occur:
  272. */
  273. int invalid = bch2_bkey_validate(c, bkey_i_to_s_c(insert), __btree_node_type(0, m->btree_id),
  274. BCH_VALIDATE_commit);
  275. if (invalid) {
  276. struct printbuf buf = PRINTBUF;
  277. prt_str(&buf, "about to insert invalid key in data update path");
  278. prt_str(&buf, "\nold: ");
  279. bch2_bkey_val_to_text(&buf, c, old);
  280. prt_str(&buf, "\nk: ");
  281. bch2_bkey_val_to_text(&buf, c, k);
  282. prt_str(&buf, "\nnew: ");
  283. bch2_bkey_val_to_text(&buf, c, bkey_i_to_s_c(insert));
  284. bch2_print_string_as_lines(KERN_ERR, buf.buf);
  285. printbuf_exit(&buf);
  286. bch2_fatal_error(c);
  287. ret = -EIO;
  288. goto out;
  289. }
  290. if (trace_data_update_enabled()) {
  291. struct printbuf buf = PRINTBUF;
  292. prt_str(&buf, "\nold: ");
  293. bch2_bkey_val_to_text(&buf, c, old);
  294. prt_str(&buf, "\nk: ");
  295. bch2_bkey_val_to_text(&buf, c, k);
  296. prt_str(&buf, "\nnew: ");
  297. bch2_bkey_val_to_text(&buf, c, bkey_i_to_s_c(insert));
  298. trace_data_update(c, buf.buf);
  299. printbuf_exit(&buf);
  300. }
  301. ret = bch2_insert_snapshot_whiteouts(trans, m->btree_id,
  302. k.k->p, bkey_start_pos(&insert->k)) ?:
  303. bch2_insert_snapshot_whiteouts(trans, m->btree_id,
  304. k.k->p, insert->k.p) ?:
  305. bch2_bkey_set_needs_rebalance(c, insert, &op->opts) ?:
  306. bch2_trans_update(trans, &iter, insert,
  307. BTREE_UPDATE_internal_snapshot_node) ?:
  308. bch2_trans_commit(trans, &op->res,
  309. NULL,
  310. BCH_TRANS_COMMIT_no_check_rw|
  311. BCH_TRANS_COMMIT_no_enospc|
  312. m->data_opts.btree_insert_flags);
  313. if (!ret) {
  314. bch2_btree_iter_set_pos(&iter, next_pos);
  315. this_cpu_add(c->counters[BCH_COUNTER_move_extent_finish], new->k.size);
  316. trace_move_extent_finish2(c, bkey_i_to_s_c(&new->k_i));
  317. }
  318. err:
  319. if (bch2_err_matches(ret, BCH_ERR_transaction_restart))
  320. ret = 0;
  321. if (ret)
  322. break;
  323. next:
  324. while (bkey_ge(iter.pos, bch2_keylist_front(keys)->k.p)) {
  325. bch2_keylist_pop_front(keys);
  326. if (bch2_keylist_empty(keys))
  327. goto out;
  328. }
  329. continue;
  330. nowork:
  331. if (m->stats) {
  332. BUG_ON(k.k->p.offset <= iter.pos.offset);
  333. atomic64_inc(&m->stats->keys_raced);
  334. atomic64_add(k.k->p.offset - iter.pos.offset,
  335. &m->stats->sectors_raced);
  336. }
  337. count_event(c, move_extent_fail);
  338. bch2_btree_iter_advance(&iter);
  339. goto next;
  340. }
  341. out:
  342. bch2_trans_iter_exit(trans, &iter);
  343. bch2_bkey_buf_exit(&_insert, c);
  344. bch2_bkey_buf_exit(&_new, c);
  345. BUG_ON(bch2_err_matches(ret, BCH_ERR_transaction_restart));
  346. return ret;
  347. }
  348. int bch2_data_update_index_update(struct bch_write_op *op)
  349. {
  350. return bch2_trans_run(op->c, __bch2_data_update_index_update(trans, op));
  351. }
  352. void bch2_data_update_read_done(struct data_update *m,
  353. struct bch_extent_crc_unpacked crc)
  354. {
  355. /* write bio must own pages: */
  356. BUG_ON(!m->op.wbio.bio.bi_vcnt);
  357. m->op.crc = crc;
  358. m->op.wbio.bio.bi_iter.bi_size = crc.compressed_size << 9;
  359. closure_call(&m->op.cl, bch2_write, NULL, NULL);
  360. }
  361. void bch2_data_update_exit(struct data_update *update)
  362. {
  363. struct bch_fs *c = update->op.c;
  364. struct bkey_s_c k = bkey_i_to_s_c(update->k.k);
  365. if (c->opts.nocow_enabled)
  366. bkey_nocow_unlock(c, k);
  367. bkey_put_dev_refs(c, k);
  368. bch2_bkey_buf_exit(&update->k, c);
  369. bch2_disk_reservation_put(c, &update->op.res);
  370. bch2_bio_free_pages_pool(c, &update->op.wbio.bio);
  371. }
  372. static void bch2_update_unwritten_extent(struct btree_trans *trans,
  373. struct data_update *update)
  374. {
  375. struct bch_fs *c = update->op.c;
  376. struct bio *bio = &update->op.wbio.bio;
  377. struct bkey_i_extent *e;
  378. struct write_point *wp;
  379. struct closure cl;
  380. struct btree_iter iter;
  381. struct bkey_s_c k;
  382. int ret;
  383. closure_init_stack(&cl);
  384. bch2_keylist_init(&update->op.insert_keys, update->op.inline_keys);
  385. while (bio_sectors(bio)) {
  386. unsigned sectors = bio_sectors(bio);
  387. bch2_trans_begin(trans);
  388. bch2_trans_iter_init(trans, &iter, update->btree_id, update->op.pos,
  389. BTREE_ITER_slots);
  390. ret = lockrestart_do(trans, ({
  391. k = bch2_btree_iter_peek_slot(&iter);
  392. bkey_err(k);
  393. }));
  394. bch2_trans_iter_exit(trans, &iter);
  395. if (ret || !bch2_extents_match(k, bkey_i_to_s_c(update->k.k)))
  396. break;
  397. e = bkey_extent_init(update->op.insert_keys.top);
  398. e->k.p = update->op.pos;
  399. ret = bch2_alloc_sectors_start_trans(trans,
  400. update->op.target,
  401. false,
  402. update->op.write_point,
  403. &update->op.devs_have,
  404. update->op.nr_replicas,
  405. update->op.nr_replicas,
  406. update->op.watermark,
  407. 0, &cl, &wp);
  408. if (bch2_err_matches(ret, BCH_ERR_operation_blocked)) {
  409. bch2_trans_unlock(trans);
  410. closure_sync(&cl);
  411. continue;
  412. }
  413. bch_err_fn_ratelimited(c, ret);
  414. if (ret)
  415. return;
  416. sectors = min(sectors, wp->sectors_free);
  417. bch2_key_resize(&e->k, sectors);
  418. bch2_open_bucket_get(c, wp, &update->op.open_buckets);
  419. bch2_alloc_sectors_append_ptrs(c, wp, &e->k_i, sectors, false);
  420. bch2_alloc_sectors_done(c, wp);
  421. bio_advance(bio, sectors << 9);
  422. update->op.pos.offset += sectors;
  423. extent_for_each_ptr(extent_i_to_s(e), ptr)
  424. ptr->unwritten = true;
  425. bch2_keylist_push(&update->op.insert_keys);
  426. ret = __bch2_data_update_index_update(trans, &update->op);
  427. bch2_open_buckets_put(c, &update->op.open_buckets);
  428. if (ret)
  429. break;
  430. }
  431. if (closure_nr_remaining(&cl) != 1) {
  432. bch2_trans_unlock(trans);
  433. closure_sync(&cl);
  434. }
  435. }
  436. void bch2_data_update_opts_to_text(struct printbuf *out, struct bch_fs *c,
  437. struct bch_io_opts *io_opts,
  438. struct data_update_opts *data_opts)
  439. {
  440. printbuf_tabstop_push(out, 20);
  441. prt_str(out, "rewrite ptrs:\t");
  442. bch2_prt_u64_base2(out, data_opts->rewrite_ptrs);
  443. prt_newline(out);
  444. prt_str(out, "kill ptrs:\t");
  445. bch2_prt_u64_base2(out, data_opts->kill_ptrs);
  446. prt_newline(out);
  447. prt_str(out, "target:\t");
  448. bch2_target_to_text(out, c, data_opts->target);
  449. prt_newline(out);
  450. prt_str(out, "compression:\t");
  451. bch2_compression_opt_to_text(out, background_compression(*io_opts));
  452. prt_newline(out);
  453. prt_str(out, "opts.replicas:\t");
  454. prt_u64(out, io_opts->data_replicas);
  455. prt_str(out, "extra replicas:\t");
  456. prt_u64(out, data_opts->extra_replicas);
  457. }
  458. void bch2_data_update_to_text(struct printbuf *out, struct data_update *m)
  459. {
  460. bch2_bkey_val_to_text(out, m->op.c, bkey_i_to_s_c(m->k.k));
  461. prt_newline(out);
  462. bch2_data_update_opts_to_text(out, m->op.c, &m->op.opts, &m->data_opts);
  463. }
  464. int bch2_extent_drop_ptrs(struct btree_trans *trans,
  465. struct btree_iter *iter,
  466. struct bkey_s_c k,
  467. struct bch_io_opts *io_opts,
  468. struct data_update_opts *data_opts)
  469. {
  470. struct bch_fs *c = trans->c;
  471. struct bkey_i *n;
  472. int ret;
  473. n = bch2_bkey_make_mut_noupdate(trans, k);
  474. ret = PTR_ERR_OR_ZERO(n);
  475. if (ret)
  476. return ret;
  477. while (data_opts->kill_ptrs) {
  478. unsigned i = 0, drop = __fls(data_opts->kill_ptrs);
  479. bch2_bkey_drop_ptrs_noerror(bkey_i_to_s(n), ptr, i++ == drop);
  480. data_opts->kill_ptrs ^= 1U << drop;
  481. }
  482. /*
  483. * If the new extent no longer has any pointers, bch2_extent_normalize()
  484. * will do the appropriate thing with it (turning it into a
  485. * KEY_TYPE_error key, or just a discard if it was a cached extent)
  486. */
  487. bch2_extent_normalize_by_opts(c, io_opts, bkey_i_to_s(n));
  488. /*
  489. * Since we're not inserting through an extent iterator
  490. * (BTREE_ITER_all_snapshots iterators aren't extent iterators),
  491. * we aren't using the extent overwrite path to delete, we're
  492. * just using the normal key deletion path:
  493. */
  494. if (bkey_deleted(&n->k) && !(iter->flags & BTREE_ITER_is_extents))
  495. n->k.size = 0;
  496. return bch2_trans_relock(trans) ?:
  497. bch2_trans_update(trans, iter, n, BTREE_UPDATE_internal_snapshot_node) ?:
  498. bch2_trans_commit(trans, NULL, NULL, BCH_TRANS_COMMIT_no_enospc);
  499. }
  500. int bch2_data_update_init(struct btree_trans *trans,
  501. struct btree_iter *iter,
  502. struct moving_context *ctxt,
  503. struct data_update *m,
  504. struct write_point_specifier wp,
  505. struct bch_io_opts io_opts,
  506. struct data_update_opts data_opts,
  507. enum btree_id btree_id,
  508. struct bkey_s_c k)
  509. {
  510. struct bch_fs *c = trans->c;
  511. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  512. const union bch_extent_entry *entry;
  513. struct extent_ptr_decoded p;
  514. unsigned i, reserve_sectors = k.k->size * data_opts.extra_replicas;
  515. int ret = 0;
  516. /*
  517. * fs is corrupt we have a key for a snapshot node that doesn't exist,
  518. * and we have to check for this because we go rw before repairing the
  519. * snapshots table - just skip it, we can move it later.
  520. */
  521. if (unlikely(k.k->p.snapshot && !bch2_snapshot_equiv(c, k.k->p.snapshot)))
  522. return -BCH_ERR_data_update_done;
  523. if (!bkey_get_dev_refs(c, k))
  524. return -BCH_ERR_data_update_done;
  525. if (c->opts.nocow_enabled &&
  526. !bkey_nocow_lock(c, ctxt, k)) {
  527. bkey_put_dev_refs(c, k);
  528. return -BCH_ERR_nocow_lock_blocked;
  529. }
  530. bch2_bkey_buf_init(&m->k);
  531. bch2_bkey_buf_reassemble(&m->k, c, k);
  532. m->btree_id = btree_id;
  533. m->data_opts = data_opts;
  534. m->ctxt = ctxt;
  535. m->stats = ctxt ? ctxt->stats : NULL;
  536. bch2_write_op_init(&m->op, c, io_opts);
  537. m->op.pos = bkey_start_pos(k.k);
  538. m->op.version = k.k->bversion;
  539. m->op.target = data_opts.target;
  540. m->op.write_point = wp;
  541. m->op.nr_replicas = 0;
  542. m->op.flags |= BCH_WRITE_PAGES_STABLE|
  543. BCH_WRITE_PAGES_OWNED|
  544. BCH_WRITE_DATA_ENCODED|
  545. BCH_WRITE_MOVE|
  546. m->data_opts.write_flags;
  547. m->op.compression_opt = background_compression(io_opts);
  548. m->op.watermark = m->data_opts.btree_insert_flags & BCH_WATERMARK_MASK;
  549. unsigned durability_have = 0, durability_removing = 0;
  550. i = 0;
  551. bkey_for_each_ptr_decode(k.k, ptrs, p, entry) {
  552. if (!p.ptr.cached) {
  553. rcu_read_lock();
  554. if (BIT(i) & m->data_opts.rewrite_ptrs) {
  555. if (crc_is_compressed(p.crc))
  556. reserve_sectors += k.k->size;
  557. m->op.nr_replicas += bch2_extent_ptr_desired_durability(c, &p);
  558. durability_removing += bch2_extent_ptr_desired_durability(c, &p);
  559. } else if (!(BIT(i) & m->data_opts.kill_ptrs)) {
  560. bch2_dev_list_add_dev(&m->op.devs_have, p.ptr.dev);
  561. durability_have += bch2_extent_ptr_durability(c, &p);
  562. }
  563. rcu_read_unlock();
  564. }
  565. /*
  566. * op->csum_type is normally initialized from the fs/file's
  567. * current options - but if an extent is encrypted, we require
  568. * that it stays encrypted:
  569. */
  570. if (bch2_csum_type_is_encryption(p.crc.csum_type)) {
  571. m->op.nonce = p.crc.nonce + p.crc.offset;
  572. m->op.csum_type = p.crc.csum_type;
  573. }
  574. if (p.crc.compression_type == BCH_COMPRESSION_TYPE_incompressible)
  575. m->op.incompressible = true;
  576. i++;
  577. }
  578. unsigned durability_required = max(0, (int) (io_opts.data_replicas - durability_have));
  579. /*
  580. * If current extent durability is less than io_opts.data_replicas,
  581. * we're not trying to rereplicate the extent up to data_replicas here -
  582. * unless extra_replicas was specified
  583. *
  584. * Increasing replication is an explicit operation triggered by
  585. * rereplicate, currently, so that users don't get an unexpected -ENOSPC
  586. */
  587. m->op.nr_replicas = min(durability_removing, durability_required) +
  588. m->data_opts.extra_replicas;
  589. /*
  590. * If device(s) were set to durability=0 after data was written to them
  591. * we can end up with a duribilty=0 extent, and the normal algorithm
  592. * that tries not to increase durability doesn't work:
  593. */
  594. if (!(durability_have + durability_removing))
  595. m->op.nr_replicas = max((unsigned) m->op.nr_replicas, 1);
  596. m->op.nr_replicas_required = m->op.nr_replicas;
  597. /*
  598. * It might turn out that we don't need any new replicas, if the
  599. * replicas or durability settings have been changed since the extent
  600. * was written:
  601. */
  602. if (!m->op.nr_replicas) {
  603. m->data_opts.kill_ptrs |= m->data_opts.rewrite_ptrs;
  604. m->data_opts.rewrite_ptrs = 0;
  605. /* if iter == NULL, it's just a promote */
  606. if (iter)
  607. ret = bch2_extent_drop_ptrs(trans, iter, k, &io_opts, &m->data_opts);
  608. goto out;
  609. }
  610. if (reserve_sectors) {
  611. ret = bch2_disk_reservation_add(c, &m->op.res, reserve_sectors,
  612. m->data_opts.extra_replicas
  613. ? 0
  614. : BCH_DISK_RESERVATION_NOFAIL);
  615. if (ret)
  616. goto out;
  617. }
  618. if (bkey_extent_is_unwritten(k)) {
  619. bch2_update_unwritten_extent(trans, m);
  620. goto out;
  621. }
  622. return 0;
  623. out:
  624. bch2_data_update_exit(m);
  625. return ret ?: -BCH_ERR_data_update_done;
  626. }
  627. void bch2_data_update_opts_normalize(struct bkey_s_c k, struct data_update_opts *opts)
  628. {
  629. struct bkey_ptrs_c ptrs = bch2_bkey_ptrs_c(k);
  630. unsigned i = 0;
  631. bkey_for_each_ptr(ptrs, ptr) {
  632. if ((opts->rewrite_ptrs & (1U << i)) && ptr->cached) {
  633. opts->kill_ptrs |= 1U << i;
  634. opts->rewrite_ptrs ^= 1U << i;
  635. }
  636. i++;
  637. }
  638. }