disk_accounting_format.h 4.2 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167
  1. /* SPDX-License-Identifier: GPL-2.0 */
  2. #ifndef _BCACHEFS_DISK_ACCOUNTING_FORMAT_H
  3. #define _BCACHEFS_DISK_ACCOUNTING_FORMAT_H
  4. #include "replicas_format.h"
  5. /*
  6. * Disk accounting - KEY_TYPE_accounting - on disk format:
  7. *
  8. * Here, the key has considerably more structure than a typical key (bpos); an
  9. * accounting key is 'struct disk_accounting_pos', which is a union of bpos.
  10. *
  11. * More specifically: a key is just a muliword integer (where word endianness
  12. * matches native byte order), so we're treating bpos as an opaque 20 byte
  13. * integer and mapping bch_accounting_key to that.
  14. *
  15. * This is a type-tagged union of all our various subtypes; a disk accounting
  16. * key can be device counters, replicas counters, et cetera - it's extensible.
  17. *
  18. * The value is a list of u64s or s64s; the number of counters is specific to a
  19. * given accounting type.
  20. *
  21. * Unlike with other key types, updates are _deltas_, and the deltas are not
  22. * resolved until the update to the underlying btree, done by btree write buffer
  23. * flush or journal replay.
  24. *
  25. * Journal replay in particular requires special handling. The journal tracks a
  26. * range of entries which may possibly have not yet been applied to the btree
  27. * yet - it does not know definitively whether individual entries are dirty and
  28. * still need to be applied.
  29. *
  30. * To handle this, we use the version field of struct bkey, and give every
  31. * accounting update a unique version number - a total ordering in time; the
  32. * version number is derived from the key's position in the journal. Then
  33. * journal replay can compare the version number of the key from the journal
  34. * with the version number of the key in the btree to determine if a key needs
  35. * to be replayed.
  36. *
  37. * For this to work, we must maintain this strict time ordering of updates as
  38. * they are flushed to the btree, both via write buffer flush and via journal
  39. * replay. This has complications for the write buffer code while journal replay
  40. * is still in progress; the write buffer cannot flush any accounting keys to
  41. * the btree until journal replay has finished replaying its accounting keys, or
  42. * the (newer) version number of the keys from the write buffer will cause
  43. * updates from journal replay to be lost.
  44. */
  45. struct bch_accounting {
  46. struct bch_val v;
  47. __u64 d[];
  48. };
  49. #define BCH_ACCOUNTING_MAX_COUNTERS 3
  50. #define BCH_DATA_TYPES() \
  51. x(free, 0) \
  52. x(sb, 1) \
  53. x(journal, 2) \
  54. x(btree, 3) \
  55. x(user, 4) \
  56. x(cached, 5) \
  57. x(parity, 6) \
  58. x(stripe, 7) \
  59. x(need_gc_gens, 8) \
  60. x(need_discard, 9) \
  61. x(unstriped, 10)
  62. enum bch_data_type {
  63. #define x(t, n) BCH_DATA_##t,
  64. BCH_DATA_TYPES()
  65. #undef x
  66. BCH_DATA_NR
  67. };
  68. static inline bool data_type_is_empty(enum bch_data_type type)
  69. {
  70. switch (type) {
  71. case BCH_DATA_free:
  72. case BCH_DATA_need_gc_gens:
  73. case BCH_DATA_need_discard:
  74. return true;
  75. default:
  76. return false;
  77. }
  78. }
  79. static inline bool data_type_is_hidden(enum bch_data_type type)
  80. {
  81. switch (type) {
  82. case BCH_DATA_sb:
  83. case BCH_DATA_journal:
  84. return true;
  85. default:
  86. return false;
  87. }
  88. }
  89. #define BCH_DISK_ACCOUNTING_TYPES() \
  90. x(nr_inodes, 0) \
  91. x(persistent_reserved, 1) \
  92. x(replicas, 2) \
  93. x(dev_data_type, 3) \
  94. x(compression, 4) \
  95. x(snapshot, 5) \
  96. x(btree, 6) \
  97. x(rebalance_work, 7) \
  98. x(inum, 8)
  99. enum disk_accounting_type {
  100. #define x(f, nr) BCH_DISK_ACCOUNTING_##f = nr,
  101. BCH_DISK_ACCOUNTING_TYPES()
  102. #undef x
  103. BCH_DISK_ACCOUNTING_TYPE_NR,
  104. };
  105. struct bch_nr_inodes {
  106. };
  107. struct bch_persistent_reserved {
  108. __u8 nr_replicas;
  109. };
  110. struct bch_dev_data_type {
  111. __u8 dev;
  112. __u8 data_type;
  113. };
  114. struct bch_acct_compression {
  115. __u8 type;
  116. };
  117. struct bch_acct_snapshot {
  118. __u32 id;
  119. } __packed;
  120. struct bch_acct_btree {
  121. __u32 id;
  122. } __packed;
  123. struct bch_acct_inum {
  124. __u64 inum;
  125. } __packed;
  126. struct bch_acct_rebalance_work {
  127. };
  128. struct disk_accounting_pos {
  129. union {
  130. struct {
  131. __u8 type;
  132. union {
  133. struct bch_nr_inodes nr_inodes;
  134. struct bch_persistent_reserved persistent_reserved;
  135. struct bch_replicas_entry_v1 replicas;
  136. struct bch_dev_data_type dev_data_type;
  137. struct bch_acct_compression compression;
  138. struct bch_acct_snapshot snapshot;
  139. struct bch_acct_btree btree;
  140. struct bch_acct_rebalance_work rebalance_work;
  141. struct bch_acct_inum inum;
  142. } __packed;
  143. } __packed;
  144. struct bpos _pad;
  145. };
  146. };
  147. #endif /* _BCACHEFS_DISK_ACCOUNTING_FORMAT_H */