buffer.c 83 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401240224032404240524062407240824092410241124122413241424152416241724182419242024212422242324242425242624272428242924302431243224332434243524362437243824392440244124422443244424452446244724482449245024512452245324542455245624572458245924602461246224632464246524662467246824692470247124722473247424752476247724782479248024812482248324842485248624872488248924902491249224932494249524962497249824992500250125022503250425052506250725082509251025112512251325142515251625172518251925202521252225232524252525262527252825292530253125322533253425352536253725382539254025412542254325442545254625472548254925502551255225532554255525562557255825592560256125622563256425652566256725682569257025712572257325742575257625772578257925802581258225832584258525862587258825892590259125922593259425952596259725982599260026012602260326042605260626072608260926102611261226132614261526162617261826192620262126222623262426252626262726282629263026312632263326342635263626372638263926402641264226432644264526462647264826492650265126522653265426552656265726582659266026612662266326642665266626672668266926702671267226732674267526762677267826792680268126822683268426852686268726882689269026912692269326942695269626972698269927002701270227032704270527062707270827092710271127122713271427152716271727182719272027212722272327242725272627272728272927302731273227332734273527362737273827392740274127422743274427452746274727482749275027512752275327542755275627572758275927602761276227632764276527662767276827692770277127722773277427752776277727782779278027812782278327842785278627872788278927902791279227932794279527962797279827992800280128022803280428052806280728082809281028112812281328142815281628172818281928202821282228232824282528262827282828292830283128322833283428352836283728382839284028412842284328442845284628472848284928502851285228532854285528562857285828592860286128622863286428652866286728682869287028712872287328742875287628772878287928802881288228832884288528862887288828892890289128922893289428952896289728982899290029012902290329042905290629072908290929102911291229132914291529162917291829192920292129222923292429252926292729282929293029312932293329342935293629372938293929402941294229432944294529462947294829492950295129522953295429552956295729582959296029612962296329642965296629672968296929702971297229732974297529762977297829792980298129822983298429852986298729882989299029912992299329942995299629972998299930003001300230033004300530063007300830093010301130123013301430153016301730183019302030213022302330243025302630273028302930303031303230333034303530363037303830393040304130423043304430453046304730483049305030513052305330543055305630573058305930603061306230633064306530663067306830693070307130723073307430753076307730783079308030813082308330843085308630873088308930903091309230933094309530963097309830993100310131023103310431053106310731083109311031113112311331143115311631173118311931203121312231233124312531263127312831293130313131323133313431353136313731383139314031413142314331443145314631473148314931503151315231533154315531563157
  1. // SPDX-License-Identifier: GPL-2.0-only
  2. /*
  3. * linux/fs/buffer.c
  4. *
  5. * Copyright (C) 1991, 1992, 2002 Linus Torvalds
  6. */
  7. /*
  8. * Start bdflush() with kernel_thread not syscall - Paul Gortmaker, 12/95
  9. *
  10. * Removed a lot of unnecessary code and simplified things now that
  11. * the buffer cache isn't our primary cache - Andrew Tridgell 12/96
  12. *
  13. * Speed up hash, lru, and free list operations. Use gfp() for allocating
  14. * hash table, use SLAB cache for buffer heads. SMP threading. -DaveM
  15. *
  16. * Added 32k buffer block sizes - these are required older ARM systems. - RMK
  17. *
  18. * async buffer flushing, 1999 Andrea Arcangeli <andrea@suse.de>
  19. */
  20. #include <linux/kernel.h>
  21. #include <linux/sched/signal.h>
  22. #include <linux/syscalls.h>
  23. #include <linux/fs.h>
  24. #include <linux/iomap.h>
  25. #include <linux/mm.h>
  26. #include <linux/percpu.h>
  27. #include <linux/slab.h>
  28. #include <linux/capability.h>
  29. #include <linux/blkdev.h>
  30. #include <linux/file.h>
  31. #include <linux/quotaops.h>
  32. #include <linux/highmem.h>
  33. #include <linux/export.h>
  34. #include <linux/backing-dev.h>
  35. #include <linux/writeback.h>
  36. #include <linux/hash.h>
  37. #include <linux/suspend.h>
  38. #include <linux/buffer_head.h>
  39. #include <linux/task_io_accounting_ops.h>
  40. #include <linux/bio.h>
  41. #include <linux/cpu.h>
  42. #include <linux/bitops.h>
  43. #include <linux/mpage.h>
  44. #include <linux/bit_spinlock.h>
  45. #include <linux/pagevec.h>
  46. #include <linux/sched/mm.h>
  47. #include <trace/events/block.h>
  48. #include <linux/fscrypt.h>
  49. #include <linux/fsverity.h>
  50. #include <linux/sched/isolation.h>
  51. #include "internal.h"
  52. static int fsync_buffers_list(spinlock_t *lock, struct list_head *list);
  53. static void submit_bh_wbc(blk_opf_t opf, struct buffer_head *bh,
  54. enum rw_hint hint, struct writeback_control *wbc);
  55. #define BH_ENTRY(list) list_entry((list), struct buffer_head, b_assoc_buffers)
  56. inline void touch_buffer(struct buffer_head *bh)
  57. {
  58. trace_block_touch_buffer(bh);
  59. folio_mark_accessed(bh->b_folio);
  60. }
  61. EXPORT_SYMBOL(touch_buffer);
  62. void __lock_buffer(struct buffer_head *bh)
  63. {
  64. wait_on_bit_lock_io(&bh->b_state, BH_Lock, TASK_UNINTERRUPTIBLE);
  65. }
  66. EXPORT_SYMBOL(__lock_buffer);
  67. void unlock_buffer(struct buffer_head *bh)
  68. {
  69. clear_bit_unlock(BH_Lock, &bh->b_state);
  70. smp_mb__after_atomic();
  71. wake_up_bit(&bh->b_state, BH_Lock);
  72. }
  73. EXPORT_SYMBOL(unlock_buffer);
  74. /*
  75. * Returns if the folio has dirty or writeback buffers. If all the buffers
  76. * are unlocked and clean then the folio_test_dirty information is stale. If
  77. * any of the buffers are locked, it is assumed they are locked for IO.
  78. */
  79. void buffer_check_dirty_writeback(struct folio *folio,
  80. bool *dirty, bool *writeback)
  81. {
  82. struct buffer_head *head, *bh;
  83. *dirty = false;
  84. *writeback = false;
  85. BUG_ON(!folio_test_locked(folio));
  86. head = folio_buffers(folio);
  87. if (!head)
  88. return;
  89. if (folio_test_writeback(folio))
  90. *writeback = true;
  91. bh = head;
  92. do {
  93. if (buffer_locked(bh))
  94. *writeback = true;
  95. if (buffer_dirty(bh))
  96. *dirty = true;
  97. bh = bh->b_this_page;
  98. } while (bh != head);
  99. }
  100. /*
  101. * Block until a buffer comes unlocked. This doesn't stop it
  102. * from becoming locked again - you have to lock it yourself
  103. * if you want to preserve its state.
  104. */
  105. void __wait_on_buffer(struct buffer_head * bh)
  106. {
  107. wait_on_bit_io(&bh->b_state, BH_Lock, TASK_UNINTERRUPTIBLE);
  108. }
  109. EXPORT_SYMBOL(__wait_on_buffer);
  110. static void buffer_io_error(struct buffer_head *bh, char *msg)
  111. {
  112. if (!test_bit(BH_Quiet, &bh->b_state))
  113. printk_ratelimited(KERN_ERR
  114. "Buffer I/O error on dev %pg, logical block %llu%s\n",
  115. bh->b_bdev, (unsigned long long)bh->b_blocknr, msg);
  116. }
  117. /*
  118. * End-of-IO handler helper function which does not touch the bh after
  119. * unlocking it.
  120. * Note: unlock_buffer() sort-of does touch the bh after unlocking it, but
  121. * a race there is benign: unlock_buffer() only use the bh's address for
  122. * hashing after unlocking the buffer, so it doesn't actually touch the bh
  123. * itself.
  124. */
  125. static void __end_buffer_read_notouch(struct buffer_head *bh, int uptodate)
  126. {
  127. if (uptodate) {
  128. set_buffer_uptodate(bh);
  129. } else {
  130. /* This happens, due to failed read-ahead attempts. */
  131. clear_buffer_uptodate(bh);
  132. }
  133. unlock_buffer(bh);
  134. }
  135. /*
  136. * Default synchronous end-of-IO handler.. Just mark it up-to-date and
  137. * unlock the buffer.
  138. */
  139. void end_buffer_read_sync(struct buffer_head *bh, int uptodate)
  140. {
  141. __end_buffer_read_notouch(bh, uptodate);
  142. put_bh(bh);
  143. }
  144. EXPORT_SYMBOL(end_buffer_read_sync);
  145. void end_buffer_write_sync(struct buffer_head *bh, int uptodate)
  146. {
  147. if (uptodate) {
  148. set_buffer_uptodate(bh);
  149. } else {
  150. buffer_io_error(bh, ", lost sync page write");
  151. mark_buffer_write_io_error(bh);
  152. clear_buffer_uptodate(bh);
  153. }
  154. unlock_buffer(bh);
  155. put_bh(bh);
  156. }
  157. EXPORT_SYMBOL(end_buffer_write_sync);
  158. /*
  159. * Various filesystems appear to want __find_get_block to be non-blocking.
  160. * But it's the page lock which protects the buffers. To get around this,
  161. * we get exclusion from try_to_free_buffers with the blockdev mapping's
  162. * i_private_lock.
  163. *
  164. * Hack idea: for the blockdev mapping, i_private_lock contention
  165. * may be quite high. This code could TryLock the page, and if that
  166. * succeeds, there is no need to take i_private_lock.
  167. */
  168. static struct buffer_head *
  169. __find_get_block_slow(struct block_device *bdev, sector_t block)
  170. {
  171. struct address_space *bd_mapping = bdev->bd_mapping;
  172. const int blkbits = bd_mapping->host->i_blkbits;
  173. struct buffer_head *ret = NULL;
  174. pgoff_t index;
  175. struct buffer_head *bh;
  176. struct buffer_head *head;
  177. struct folio *folio;
  178. int all_mapped = 1;
  179. static DEFINE_RATELIMIT_STATE(last_warned, HZ, 1);
  180. index = ((loff_t)block << blkbits) / PAGE_SIZE;
  181. folio = __filemap_get_folio(bd_mapping, index, FGP_ACCESSED, 0);
  182. if (IS_ERR(folio))
  183. goto out;
  184. spin_lock(&bd_mapping->i_private_lock);
  185. head = folio_buffers(folio);
  186. if (!head)
  187. goto out_unlock;
  188. bh = head;
  189. do {
  190. if (!buffer_mapped(bh))
  191. all_mapped = 0;
  192. else if (bh->b_blocknr == block) {
  193. ret = bh;
  194. get_bh(bh);
  195. goto out_unlock;
  196. }
  197. bh = bh->b_this_page;
  198. } while (bh != head);
  199. /* we might be here because some of the buffers on this page are
  200. * not mapped. This is due to various races between
  201. * file io on the block device and getblk. It gets dealt with
  202. * elsewhere, don't buffer_error if we had some unmapped buffers
  203. */
  204. ratelimit_set_flags(&last_warned, RATELIMIT_MSG_ON_RELEASE);
  205. if (all_mapped && __ratelimit(&last_warned)) {
  206. printk("__find_get_block_slow() failed. block=%llu, "
  207. "b_blocknr=%llu, b_state=0x%08lx, b_size=%zu, "
  208. "device %pg blocksize: %d\n",
  209. (unsigned long long)block,
  210. (unsigned long long)bh->b_blocknr,
  211. bh->b_state, bh->b_size, bdev,
  212. 1 << blkbits);
  213. }
  214. out_unlock:
  215. spin_unlock(&bd_mapping->i_private_lock);
  216. folio_put(folio);
  217. out:
  218. return ret;
  219. }
  220. static void end_buffer_async_read(struct buffer_head *bh, int uptodate)
  221. {
  222. unsigned long flags;
  223. struct buffer_head *first;
  224. struct buffer_head *tmp;
  225. struct folio *folio;
  226. int folio_uptodate = 1;
  227. BUG_ON(!buffer_async_read(bh));
  228. folio = bh->b_folio;
  229. if (uptodate) {
  230. set_buffer_uptodate(bh);
  231. } else {
  232. clear_buffer_uptodate(bh);
  233. buffer_io_error(bh, ", async page read");
  234. }
  235. /*
  236. * Be _very_ careful from here on. Bad things can happen if
  237. * two buffer heads end IO at almost the same time and both
  238. * decide that the page is now completely done.
  239. */
  240. first = folio_buffers(folio);
  241. spin_lock_irqsave(&first->b_uptodate_lock, flags);
  242. clear_buffer_async_read(bh);
  243. unlock_buffer(bh);
  244. tmp = bh;
  245. do {
  246. if (!buffer_uptodate(tmp))
  247. folio_uptodate = 0;
  248. if (buffer_async_read(tmp)) {
  249. BUG_ON(!buffer_locked(tmp));
  250. goto still_busy;
  251. }
  252. tmp = tmp->b_this_page;
  253. } while (tmp != bh);
  254. spin_unlock_irqrestore(&first->b_uptodate_lock, flags);
  255. folio_end_read(folio, folio_uptodate);
  256. return;
  257. still_busy:
  258. spin_unlock_irqrestore(&first->b_uptodate_lock, flags);
  259. return;
  260. }
  261. struct postprocess_bh_ctx {
  262. struct work_struct work;
  263. struct buffer_head *bh;
  264. };
  265. static void verify_bh(struct work_struct *work)
  266. {
  267. struct postprocess_bh_ctx *ctx =
  268. container_of(work, struct postprocess_bh_ctx, work);
  269. struct buffer_head *bh = ctx->bh;
  270. bool valid;
  271. valid = fsverity_verify_blocks(bh->b_folio, bh->b_size, bh_offset(bh));
  272. end_buffer_async_read(bh, valid);
  273. kfree(ctx);
  274. }
  275. static bool need_fsverity(struct buffer_head *bh)
  276. {
  277. struct folio *folio = bh->b_folio;
  278. struct inode *inode = folio->mapping->host;
  279. return fsverity_active(inode) &&
  280. /* needed by ext4 */
  281. folio->index < DIV_ROUND_UP(inode->i_size, PAGE_SIZE);
  282. }
  283. static void decrypt_bh(struct work_struct *work)
  284. {
  285. struct postprocess_bh_ctx *ctx =
  286. container_of(work, struct postprocess_bh_ctx, work);
  287. struct buffer_head *bh = ctx->bh;
  288. int err;
  289. err = fscrypt_decrypt_pagecache_blocks(bh->b_folio, bh->b_size,
  290. bh_offset(bh));
  291. if (err == 0 && need_fsverity(bh)) {
  292. /*
  293. * We use different work queues for decryption and for verity
  294. * because verity may require reading metadata pages that need
  295. * decryption, and we shouldn't recurse to the same workqueue.
  296. */
  297. INIT_WORK(&ctx->work, verify_bh);
  298. fsverity_enqueue_verify_work(&ctx->work);
  299. return;
  300. }
  301. end_buffer_async_read(bh, err == 0);
  302. kfree(ctx);
  303. }
  304. /*
  305. * I/O completion handler for block_read_full_folio() - pages
  306. * which come unlocked at the end of I/O.
  307. */
  308. static void end_buffer_async_read_io(struct buffer_head *bh, int uptodate)
  309. {
  310. struct inode *inode = bh->b_folio->mapping->host;
  311. bool decrypt = fscrypt_inode_uses_fs_layer_crypto(inode);
  312. bool verify = need_fsverity(bh);
  313. /* Decrypt (with fscrypt) and/or verify (with fsverity) if needed. */
  314. if (uptodate && (decrypt || verify)) {
  315. struct postprocess_bh_ctx *ctx =
  316. kmalloc(sizeof(*ctx), GFP_ATOMIC);
  317. if (ctx) {
  318. ctx->bh = bh;
  319. if (decrypt) {
  320. INIT_WORK(&ctx->work, decrypt_bh);
  321. fscrypt_enqueue_decrypt_work(&ctx->work);
  322. } else {
  323. INIT_WORK(&ctx->work, verify_bh);
  324. fsverity_enqueue_verify_work(&ctx->work);
  325. }
  326. return;
  327. }
  328. uptodate = 0;
  329. }
  330. end_buffer_async_read(bh, uptodate);
  331. }
  332. /*
  333. * Completion handler for block_write_full_folio() - folios which are unlocked
  334. * during I/O, and which have the writeback flag cleared upon I/O completion.
  335. */
  336. static void end_buffer_async_write(struct buffer_head *bh, int uptodate)
  337. {
  338. unsigned long flags;
  339. struct buffer_head *first;
  340. struct buffer_head *tmp;
  341. struct folio *folio;
  342. BUG_ON(!buffer_async_write(bh));
  343. folio = bh->b_folio;
  344. if (uptodate) {
  345. set_buffer_uptodate(bh);
  346. } else {
  347. buffer_io_error(bh, ", lost async page write");
  348. mark_buffer_write_io_error(bh);
  349. clear_buffer_uptodate(bh);
  350. }
  351. first = folio_buffers(folio);
  352. spin_lock_irqsave(&first->b_uptodate_lock, flags);
  353. clear_buffer_async_write(bh);
  354. unlock_buffer(bh);
  355. tmp = bh->b_this_page;
  356. while (tmp != bh) {
  357. if (buffer_async_write(tmp)) {
  358. BUG_ON(!buffer_locked(tmp));
  359. goto still_busy;
  360. }
  361. tmp = tmp->b_this_page;
  362. }
  363. spin_unlock_irqrestore(&first->b_uptodate_lock, flags);
  364. folio_end_writeback(folio);
  365. return;
  366. still_busy:
  367. spin_unlock_irqrestore(&first->b_uptodate_lock, flags);
  368. return;
  369. }
  370. /*
  371. * If a page's buffers are under async readin (end_buffer_async_read
  372. * completion) then there is a possibility that another thread of
  373. * control could lock one of the buffers after it has completed
  374. * but while some of the other buffers have not completed. This
  375. * locked buffer would confuse end_buffer_async_read() into not unlocking
  376. * the page. So the absence of BH_Async_Read tells end_buffer_async_read()
  377. * that this buffer is not under async I/O.
  378. *
  379. * The page comes unlocked when it has no locked buffer_async buffers
  380. * left.
  381. *
  382. * PageLocked prevents anyone starting new async I/O reads any of
  383. * the buffers.
  384. *
  385. * PageWriteback is used to prevent simultaneous writeout of the same
  386. * page.
  387. *
  388. * PageLocked prevents anyone from starting writeback of a page which is
  389. * under read I/O (PageWriteback is only ever set against a locked page).
  390. */
  391. static void mark_buffer_async_read(struct buffer_head *bh)
  392. {
  393. bh->b_end_io = end_buffer_async_read_io;
  394. set_buffer_async_read(bh);
  395. }
  396. static void mark_buffer_async_write_endio(struct buffer_head *bh,
  397. bh_end_io_t *handler)
  398. {
  399. bh->b_end_io = handler;
  400. set_buffer_async_write(bh);
  401. }
  402. void mark_buffer_async_write(struct buffer_head *bh)
  403. {
  404. mark_buffer_async_write_endio(bh, end_buffer_async_write);
  405. }
  406. EXPORT_SYMBOL(mark_buffer_async_write);
  407. /*
  408. * fs/buffer.c contains helper functions for buffer-backed address space's
  409. * fsync functions. A common requirement for buffer-based filesystems is
  410. * that certain data from the backing blockdev needs to be written out for
  411. * a successful fsync(). For example, ext2 indirect blocks need to be
  412. * written back and waited upon before fsync() returns.
  413. *
  414. * The functions mark_buffer_dirty_inode(), fsync_inode_buffers(),
  415. * inode_has_buffers() and invalidate_inode_buffers() are provided for the
  416. * management of a list of dependent buffers at ->i_mapping->i_private_list.
  417. *
  418. * Locking is a little subtle: try_to_free_buffers() will remove buffers
  419. * from their controlling inode's queue when they are being freed. But
  420. * try_to_free_buffers() will be operating against the *blockdev* mapping
  421. * at the time, not against the S_ISREG file which depends on those buffers.
  422. * So the locking for i_private_list is via the i_private_lock in the address_space
  423. * which backs the buffers. Which is different from the address_space
  424. * against which the buffers are listed. So for a particular address_space,
  425. * mapping->i_private_lock does *not* protect mapping->i_private_list! In fact,
  426. * mapping->i_private_list will always be protected by the backing blockdev's
  427. * ->i_private_lock.
  428. *
  429. * Which introduces a requirement: all buffers on an address_space's
  430. * ->i_private_list must be from the same address_space: the blockdev's.
  431. *
  432. * address_spaces which do not place buffers at ->i_private_list via these
  433. * utility functions are free to use i_private_lock and i_private_list for
  434. * whatever they want. The only requirement is that list_empty(i_private_list)
  435. * be true at clear_inode() time.
  436. *
  437. * FIXME: clear_inode should not call invalidate_inode_buffers(). The
  438. * filesystems should do that. invalidate_inode_buffers() should just go
  439. * BUG_ON(!list_empty).
  440. *
  441. * FIXME: mark_buffer_dirty_inode() is a data-plane operation. It should
  442. * take an address_space, not an inode. And it should be called
  443. * mark_buffer_dirty_fsync() to clearly define why those buffers are being
  444. * queued up.
  445. *
  446. * FIXME: mark_buffer_dirty_inode() doesn't need to add the buffer to the
  447. * list if it is already on a list. Because if the buffer is on a list,
  448. * it *must* already be on the right one. If not, the filesystem is being
  449. * silly. This will save a ton of locking. But first we have to ensure
  450. * that buffers are taken *off* the old inode's list when they are freed
  451. * (presumably in truncate). That requires careful auditing of all
  452. * filesystems (do it inside bforget()). It could also be done by bringing
  453. * b_inode back.
  454. */
  455. /*
  456. * The buffer's backing address_space's i_private_lock must be held
  457. */
  458. static void __remove_assoc_queue(struct buffer_head *bh)
  459. {
  460. list_del_init(&bh->b_assoc_buffers);
  461. WARN_ON(!bh->b_assoc_map);
  462. bh->b_assoc_map = NULL;
  463. }
  464. int inode_has_buffers(struct inode *inode)
  465. {
  466. return !list_empty(&inode->i_data.i_private_list);
  467. }
  468. /*
  469. * osync is designed to support O_SYNC io. It waits synchronously for
  470. * all already-submitted IO to complete, but does not queue any new
  471. * writes to the disk.
  472. *
  473. * To do O_SYNC writes, just queue the buffer writes with write_dirty_buffer
  474. * as you dirty the buffers, and then use osync_inode_buffers to wait for
  475. * completion. Any other dirty buffers which are not yet queued for
  476. * write will not be flushed to disk by the osync.
  477. */
  478. static int osync_buffers_list(spinlock_t *lock, struct list_head *list)
  479. {
  480. struct buffer_head *bh;
  481. struct list_head *p;
  482. int err = 0;
  483. spin_lock(lock);
  484. repeat:
  485. list_for_each_prev(p, list) {
  486. bh = BH_ENTRY(p);
  487. if (buffer_locked(bh)) {
  488. get_bh(bh);
  489. spin_unlock(lock);
  490. wait_on_buffer(bh);
  491. if (!buffer_uptodate(bh))
  492. err = -EIO;
  493. brelse(bh);
  494. spin_lock(lock);
  495. goto repeat;
  496. }
  497. }
  498. spin_unlock(lock);
  499. return err;
  500. }
  501. /**
  502. * sync_mapping_buffers - write out & wait upon a mapping's "associated" buffers
  503. * @mapping: the mapping which wants those buffers written
  504. *
  505. * Starts I/O against the buffers at mapping->i_private_list, and waits upon
  506. * that I/O.
  507. *
  508. * Basically, this is a convenience function for fsync().
  509. * @mapping is a file or directory which needs those buffers to be written for
  510. * a successful fsync().
  511. */
  512. int sync_mapping_buffers(struct address_space *mapping)
  513. {
  514. struct address_space *buffer_mapping = mapping->i_private_data;
  515. if (buffer_mapping == NULL || list_empty(&mapping->i_private_list))
  516. return 0;
  517. return fsync_buffers_list(&buffer_mapping->i_private_lock,
  518. &mapping->i_private_list);
  519. }
  520. EXPORT_SYMBOL(sync_mapping_buffers);
  521. /**
  522. * generic_buffers_fsync_noflush - generic buffer fsync implementation
  523. * for simple filesystems with no inode lock
  524. *
  525. * @file: file to synchronize
  526. * @start: start offset in bytes
  527. * @end: end offset in bytes (inclusive)
  528. * @datasync: only synchronize essential metadata if true
  529. *
  530. * This is a generic implementation of the fsync method for simple
  531. * filesystems which track all non-inode metadata in the buffers list
  532. * hanging off the address_space structure.
  533. */
  534. int generic_buffers_fsync_noflush(struct file *file, loff_t start, loff_t end,
  535. bool datasync)
  536. {
  537. struct inode *inode = file->f_mapping->host;
  538. int err;
  539. int ret;
  540. err = file_write_and_wait_range(file, start, end);
  541. if (err)
  542. return err;
  543. ret = sync_mapping_buffers(inode->i_mapping);
  544. if (!(inode->i_state & I_DIRTY_ALL))
  545. goto out;
  546. if (datasync && !(inode->i_state & I_DIRTY_DATASYNC))
  547. goto out;
  548. err = sync_inode_metadata(inode, 1);
  549. if (ret == 0)
  550. ret = err;
  551. out:
  552. /* check and advance again to catch errors after syncing out buffers */
  553. err = file_check_and_advance_wb_err(file);
  554. if (ret == 0)
  555. ret = err;
  556. return ret;
  557. }
  558. EXPORT_SYMBOL(generic_buffers_fsync_noflush);
  559. /**
  560. * generic_buffers_fsync - generic buffer fsync implementation
  561. * for simple filesystems with no inode lock
  562. *
  563. * @file: file to synchronize
  564. * @start: start offset in bytes
  565. * @end: end offset in bytes (inclusive)
  566. * @datasync: only synchronize essential metadata if true
  567. *
  568. * This is a generic implementation of the fsync method for simple
  569. * filesystems which track all non-inode metadata in the buffers list
  570. * hanging off the address_space structure. This also makes sure that
  571. * a device cache flush operation is called at the end.
  572. */
  573. int generic_buffers_fsync(struct file *file, loff_t start, loff_t end,
  574. bool datasync)
  575. {
  576. struct inode *inode = file->f_mapping->host;
  577. int ret;
  578. ret = generic_buffers_fsync_noflush(file, start, end, datasync);
  579. if (!ret)
  580. ret = blkdev_issue_flush(inode->i_sb->s_bdev);
  581. return ret;
  582. }
  583. EXPORT_SYMBOL(generic_buffers_fsync);
  584. /*
  585. * Called when we've recently written block `bblock', and it is known that
  586. * `bblock' was for a buffer_boundary() buffer. This means that the block at
  587. * `bblock + 1' is probably a dirty indirect block. Hunt it down and, if it's
  588. * dirty, schedule it for IO. So that indirects merge nicely with their data.
  589. */
  590. void write_boundary_block(struct block_device *bdev,
  591. sector_t bblock, unsigned blocksize)
  592. {
  593. struct buffer_head *bh = __find_get_block(bdev, bblock + 1, blocksize);
  594. if (bh) {
  595. if (buffer_dirty(bh))
  596. write_dirty_buffer(bh, 0);
  597. put_bh(bh);
  598. }
  599. }
  600. void mark_buffer_dirty_inode(struct buffer_head *bh, struct inode *inode)
  601. {
  602. struct address_space *mapping = inode->i_mapping;
  603. struct address_space *buffer_mapping = bh->b_folio->mapping;
  604. mark_buffer_dirty(bh);
  605. if (!mapping->i_private_data) {
  606. mapping->i_private_data = buffer_mapping;
  607. } else {
  608. BUG_ON(mapping->i_private_data != buffer_mapping);
  609. }
  610. if (!bh->b_assoc_map) {
  611. spin_lock(&buffer_mapping->i_private_lock);
  612. list_move_tail(&bh->b_assoc_buffers,
  613. &mapping->i_private_list);
  614. bh->b_assoc_map = mapping;
  615. spin_unlock(&buffer_mapping->i_private_lock);
  616. }
  617. }
  618. EXPORT_SYMBOL(mark_buffer_dirty_inode);
  619. /**
  620. * block_dirty_folio - Mark a folio as dirty.
  621. * @mapping: The address space containing this folio.
  622. * @folio: The folio to mark dirty.
  623. *
  624. * Filesystems which use buffer_heads can use this function as their
  625. * ->dirty_folio implementation. Some filesystems need to do a little
  626. * work before calling this function. Filesystems which do not use
  627. * buffer_heads should call filemap_dirty_folio() instead.
  628. *
  629. * If the folio has buffers, the uptodate buffers are set dirty, to
  630. * preserve dirty-state coherency between the folio and the buffers.
  631. * Buffers added to a dirty folio are created dirty.
  632. *
  633. * The buffers are dirtied before the folio is dirtied. There's a small
  634. * race window in which writeback may see the folio cleanness but not the
  635. * buffer dirtiness. That's fine. If this code were to set the folio
  636. * dirty before the buffers, writeback could clear the folio dirty flag,
  637. * see a bunch of clean buffers and we'd end up with dirty buffers/clean
  638. * folio on the dirty folio list.
  639. *
  640. * We use i_private_lock to lock against try_to_free_buffers() while
  641. * using the folio's buffer list. This also prevents clean buffers
  642. * being added to the folio after it was set dirty.
  643. *
  644. * Context: May only be called from process context. Does not sleep.
  645. * Caller must ensure that @folio cannot be truncated during this call,
  646. * typically by holding the folio lock or having a page in the folio
  647. * mapped and holding the page table lock.
  648. *
  649. * Return: True if the folio was dirtied; false if it was already dirtied.
  650. */
  651. bool block_dirty_folio(struct address_space *mapping, struct folio *folio)
  652. {
  653. struct buffer_head *head;
  654. bool newly_dirty;
  655. spin_lock(&mapping->i_private_lock);
  656. head = folio_buffers(folio);
  657. if (head) {
  658. struct buffer_head *bh = head;
  659. do {
  660. set_buffer_dirty(bh);
  661. bh = bh->b_this_page;
  662. } while (bh != head);
  663. }
  664. /*
  665. * Lock out page's memcg migration to keep PageDirty
  666. * synchronized with per-memcg dirty page counters.
  667. */
  668. folio_memcg_lock(folio);
  669. newly_dirty = !folio_test_set_dirty(folio);
  670. spin_unlock(&mapping->i_private_lock);
  671. if (newly_dirty)
  672. __folio_mark_dirty(folio, mapping, 1);
  673. folio_memcg_unlock(folio);
  674. if (newly_dirty)
  675. __mark_inode_dirty(mapping->host, I_DIRTY_PAGES);
  676. return newly_dirty;
  677. }
  678. EXPORT_SYMBOL(block_dirty_folio);
  679. /*
  680. * Write out and wait upon a list of buffers.
  681. *
  682. * We have conflicting pressures: we want to make sure that all
  683. * initially dirty buffers get waited on, but that any subsequently
  684. * dirtied buffers don't. After all, we don't want fsync to last
  685. * forever if somebody is actively writing to the file.
  686. *
  687. * Do this in two main stages: first we copy dirty buffers to a
  688. * temporary inode list, queueing the writes as we go. Then we clean
  689. * up, waiting for those writes to complete.
  690. *
  691. * During this second stage, any subsequent updates to the file may end
  692. * up refiling the buffer on the original inode's dirty list again, so
  693. * there is a chance we will end up with a buffer queued for write but
  694. * not yet completed on that list. So, as a final cleanup we go through
  695. * the osync code to catch these locked, dirty buffers without requeuing
  696. * any newly dirty buffers for write.
  697. */
  698. static int fsync_buffers_list(spinlock_t *lock, struct list_head *list)
  699. {
  700. struct buffer_head *bh;
  701. struct address_space *mapping;
  702. int err = 0, err2;
  703. struct blk_plug plug;
  704. LIST_HEAD(tmp);
  705. blk_start_plug(&plug);
  706. spin_lock(lock);
  707. while (!list_empty(list)) {
  708. bh = BH_ENTRY(list->next);
  709. mapping = bh->b_assoc_map;
  710. __remove_assoc_queue(bh);
  711. /* Avoid race with mark_buffer_dirty_inode() which does
  712. * a lockless check and we rely on seeing the dirty bit */
  713. smp_mb();
  714. if (buffer_dirty(bh) || buffer_locked(bh)) {
  715. list_add(&bh->b_assoc_buffers, &tmp);
  716. bh->b_assoc_map = mapping;
  717. if (buffer_dirty(bh)) {
  718. get_bh(bh);
  719. spin_unlock(lock);
  720. /*
  721. * Ensure any pending I/O completes so that
  722. * write_dirty_buffer() actually writes the
  723. * current contents - it is a noop if I/O is
  724. * still in flight on potentially older
  725. * contents.
  726. */
  727. write_dirty_buffer(bh, REQ_SYNC);
  728. /*
  729. * Kick off IO for the previous mapping. Note
  730. * that we will not run the very last mapping,
  731. * wait_on_buffer() will do that for us
  732. * through sync_buffer().
  733. */
  734. brelse(bh);
  735. spin_lock(lock);
  736. }
  737. }
  738. }
  739. spin_unlock(lock);
  740. blk_finish_plug(&plug);
  741. spin_lock(lock);
  742. while (!list_empty(&tmp)) {
  743. bh = BH_ENTRY(tmp.prev);
  744. get_bh(bh);
  745. mapping = bh->b_assoc_map;
  746. __remove_assoc_queue(bh);
  747. /* Avoid race with mark_buffer_dirty_inode() which does
  748. * a lockless check and we rely on seeing the dirty bit */
  749. smp_mb();
  750. if (buffer_dirty(bh)) {
  751. list_add(&bh->b_assoc_buffers,
  752. &mapping->i_private_list);
  753. bh->b_assoc_map = mapping;
  754. }
  755. spin_unlock(lock);
  756. wait_on_buffer(bh);
  757. if (!buffer_uptodate(bh))
  758. err = -EIO;
  759. brelse(bh);
  760. spin_lock(lock);
  761. }
  762. spin_unlock(lock);
  763. err2 = osync_buffers_list(lock, list);
  764. if (err)
  765. return err;
  766. else
  767. return err2;
  768. }
  769. /*
  770. * Invalidate any and all dirty buffers on a given inode. We are
  771. * probably unmounting the fs, but that doesn't mean we have already
  772. * done a sync(). Just drop the buffers from the inode list.
  773. *
  774. * NOTE: we take the inode's blockdev's mapping's i_private_lock. Which
  775. * assumes that all the buffers are against the blockdev. Not true
  776. * for reiserfs.
  777. */
  778. void invalidate_inode_buffers(struct inode *inode)
  779. {
  780. if (inode_has_buffers(inode)) {
  781. struct address_space *mapping = &inode->i_data;
  782. struct list_head *list = &mapping->i_private_list;
  783. struct address_space *buffer_mapping = mapping->i_private_data;
  784. spin_lock(&buffer_mapping->i_private_lock);
  785. while (!list_empty(list))
  786. __remove_assoc_queue(BH_ENTRY(list->next));
  787. spin_unlock(&buffer_mapping->i_private_lock);
  788. }
  789. }
  790. EXPORT_SYMBOL(invalidate_inode_buffers);
  791. /*
  792. * Remove any clean buffers from the inode's buffer list. This is called
  793. * when we're trying to free the inode itself. Those buffers can pin it.
  794. *
  795. * Returns true if all buffers were removed.
  796. */
  797. int remove_inode_buffers(struct inode *inode)
  798. {
  799. int ret = 1;
  800. if (inode_has_buffers(inode)) {
  801. struct address_space *mapping = &inode->i_data;
  802. struct list_head *list = &mapping->i_private_list;
  803. struct address_space *buffer_mapping = mapping->i_private_data;
  804. spin_lock(&buffer_mapping->i_private_lock);
  805. while (!list_empty(list)) {
  806. struct buffer_head *bh = BH_ENTRY(list->next);
  807. if (buffer_dirty(bh)) {
  808. ret = 0;
  809. break;
  810. }
  811. __remove_assoc_queue(bh);
  812. }
  813. spin_unlock(&buffer_mapping->i_private_lock);
  814. }
  815. return ret;
  816. }
  817. /*
  818. * Create the appropriate buffers when given a folio for data area and
  819. * the size of each buffer.. Use the bh->b_this_page linked list to
  820. * follow the buffers created. Return NULL if unable to create more
  821. * buffers.
  822. *
  823. * The retry flag is used to differentiate async IO (paging, swapping)
  824. * which may not fail from ordinary buffer allocations.
  825. */
  826. struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size,
  827. gfp_t gfp)
  828. {
  829. struct buffer_head *bh, *head;
  830. long offset;
  831. struct mem_cgroup *memcg, *old_memcg;
  832. /* The folio lock pins the memcg */
  833. memcg = folio_memcg(folio);
  834. old_memcg = set_active_memcg(memcg);
  835. head = NULL;
  836. offset = folio_size(folio);
  837. while ((offset -= size) >= 0) {
  838. bh = alloc_buffer_head(gfp);
  839. if (!bh)
  840. goto no_grow;
  841. bh->b_this_page = head;
  842. bh->b_blocknr = -1;
  843. head = bh;
  844. bh->b_size = size;
  845. /* Link the buffer to its folio */
  846. folio_set_bh(bh, folio, offset);
  847. }
  848. out:
  849. set_active_memcg(old_memcg);
  850. return head;
  851. /*
  852. * In case anything failed, we just free everything we got.
  853. */
  854. no_grow:
  855. if (head) {
  856. do {
  857. bh = head;
  858. head = head->b_this_page;
  859. free_buffer_head(bh);
  860. } while (head);
  861. }
  862. goto out;
  863. }
  864. EXPORT_SYMBOL_GPL(folio_alloc_buffers);
  865. struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size)
  866. {
  867. gfp_t gfp = GFP_NOFS | __GFP_ACCOUNT;
  868. return folio_alloc_buffers(page_folio(page), size, gfp);
  869. }
  870. EXPORT_SYMBOL_GPL(alloc_page_buffers);
  871. static inline void link_dev_buffers(struct folio *folio,
  872. struct buffer_head *head)
  873. {
  874. struct buffer_head *bh, *tail;
  875. bh = head;
  876. do {
  877. tail = bh;
  878. bh = bh->b_this_page;
  879. } while (bh);
  880. tail->b_this_page = head;
  881. folio_attach_private(folio, head);
  882. }
  883. static sector_t blkdev_max_block(struct block_device *bdev, unsigned int size)
  884. {
  885. sector_t retval = ~((sector_t)0);
  886. loff_t sz = bdev_nr_bytes(bdev);
  887. if (sz) {
  888. unsigned int sizebits = blksize_bits(size);
  889. retval = (sz >> sizebits);
  890. }
  891. return retval;
  892. }
  893. /*
  894. * Initialise the state of a blockdev folio's buffers.
  895. */
  896. static sector_t folio_init_buffers(struct folio *folio,
  897. struct block_device *bdev, unsigned size)
  898. {
  899. struct buffer_head *head = folio_buffers(folio);
  900. struct buffer_head *bh = head;
  901. bool uptodate = folio_test_uptodate(folio);
  902. sector_t block = div_u64(folio_pos(folio), size);
  903. sector_t end_block = blkdev_max_block(bdev, size);
  904. do {
  905. if (!buffer_mapped(bh)) {
  906. bh->b_end_io = NULL;
  907. bh->b_private = NULL;
  908. bh->b_bdev = bdev;
  909. bh->b_blocknr = block;
  910. if (uptodate)
  911. set_buffer_uptodate(bh);
  912. if (block < end_block)
  913. set_buffer_mapped(bh);
  914. }
  915. block++;
  916. bh = bh->b_this_page;
  917. } while (bh != head);
  918. /*
  919. * Caller needs to validate requested block against end of device.
  920. */
  921. return end_block;
  922. }
  923. /*
  924. * Create the page-cache folio that contains the requested block.
  925. *
  926. * This is used purely for blockdev mappings.
  927. *
  928. * Returns false if we have a failure which cannot be cured by retrying
  929. * without sleeping. Returns true if we succeeded, or the caller should retry.
  930. */
  931. static bool grow_dev_folio(struct block_device *bdev, sector_t block,
  932. pgoff_t index, unsigned size, gfp_t gfp)
  933. {
  934. struct address_space *mapping = bdev->bd_mapping;
  935. struct folio *folio;
  936. struct buffer_head *bh;
  937. sector_t end_block = 0;
  938. folio = __filemap_get_folio(mapping, index,
  939. FGP_LOCK | FGP_ACCESSED | FGP_CREAT, gfp);
  940. if (IS_ERR(folio))
  941. return false;
  942. bh = folio_buffers(folio);
  943. if (bh) {
  944. if (bh->b_size == size) {
  945. end_block = folio_init_buffers(folio, bdev, size);
  946. goto unlock;
  947. }
  948. /*
  949. * Retrying may succeed; for example the folio may finish
  950. * writeback, or buffers may be cleaned. This should not
  951. * happen very often; maybe we have old buffers attached to
  952. * this blockdev's page cache and we're trying to change
  953. * the block size?
  954. */
  955. if (!try_to_free_buffers(folio)) {
  956. end_block = ~0ULL;
  957. goto unlock;
  958. }
  959. }
  960. bh = folio_alloc_buffers(folio, size, gfp | __GFP_ACCOUNT);
  961. if (!bh)
  962. goto unlock;
  963. /*
  964. * Link the folio to the buffers and initialise them. Take the
  965. * lock to be atomic wrt __find_get_block(), which does not
  966. * run under the folio lock.
  967. */
  968. spin_lock(&mapping->i_private_lock);
  969. link_dev_buffers(folio, bh);
  970. end_block = folio_init_buffers(folio, bdev, size);
  971. spin_unlock(&mapping->i_private_lock);
  972. unlock:
  973. folio_unlock(folio);
  974. folio_put(folio);
  975. return block < end_block;
  976. }
  977. /*
  978. * Create buffers for the specified block device block's folio. If
  979. * that folio was dirty, the buffers are set dirty also. Returns false
  980. * if we've hit a permanent error.
  981. */
  982. static bool grow_buffers(struct block_device *bdev, sector_t block,
  983. unsigned size, gfp_t gfp)
  984. {
  985. loff_t pos;
  986. /*
  987. * Check for a block which lies outside our maximum possible
  988. * pagecache index.
  989. */
  990. if (check_mul_overflow(block, (sector_t)size, &pos) || pos > MAX_LFS_FILESIZE) {
  991. printk(KERN_ERR "%s: requested out-of-range block %llu for device %pg\n",
  992. __func__, (unsigned long long)block,
  993. bdev);
  994. return false;
  995. }
  996. /* Create a folio with the proper size buffers */
  997. return grow_dev_folio(bdev, block, pos / PAGE_SIZE, size, gfp);
  998. }
  999. static struct buffer_head *
  1000. __getblk_slow(struct block_device *bdev, sector_t block,
  1001. unsigned size, gfp_t gfp)
  1002. {
  1003. /* Size must be multiple of hard sectorsize */
  1004. if (unlikely(size & (bdev_logical_block_size(bdev)-1) ||
  1005. (size < 512 || size > PAGE_SIZE))) {
  1006. printk(KERN_ERR "getblk(): invalid block size %d requested\n",
  1007. size);
  1008. printk(KERN_ERR "logical block size: %d\n",
  1009. bdev_logical_block_size(bdev));
  1010. dump_stack();
  1011. return NULL;
  1012. }
  1013. for (;;) {
  1014. struct buffer_head *bh;
  1015. bh = __find_get_block(bdev, block, size);
  1016. if (bh)
  1017. return bh;
  1018. if (!grow_buffers(bdev, block, size, gfp))
  1019. return NULL;
  1020. }
  1021. }
  1022. /*
  1023. * The relationship between dirty buffers and dirty pages:
  1024. *
  1025. * Whenever a page has any dirty buffers, the page's dirty bit is set, and
  1026. * the page is tagged dirty in the page cache.
  1027. *
  1028. * At all times, the dirtiness of the buffers represents the dirtiness of
  1029. * subsections of the page. If the page has buffers, the page dirty bit is
  1030. * merely a hint about the true dirty state.
  1031. *
  1032. * When a page is set dirty in its entirety, all its buffers are marked dirty
  1033. * (if the page has buffers).
  1034. *
  1035. * When a buffer is marked dirty, its page is dirtied, but the page's other
  1036. * buffers are not.
  1037. *
  1038. * Also. When blockdev buffers are explicitly read with bread(), they
  1039. * individually become uptodate. But their backing page remains not
  1040. * uptodate - even if all of its buffers are uptodate. A subsequent
  1041. * block_read_full_folio() against that folio will discover all the uptodate
  1042. * buffers, will set the folio uptodate and will perform no I/O.
  1043. */
  1044. /**
  1045. * mark_buffer_dirty - mark a buffer_head as needing writeout
  1046. * @bh: the buffer_head to mark dirty
  1047. *
  1048. * mark_buffer_dirty() will set the dirty bit against the buffer, then set
  1049. * its backing page dirty, then tag the page as dirty in the page cache
  1050. * and then attach the address_space's inode to its superblock's dirty
  1051. * inode list.
  1052. *
  1053. * mark_buffer_dirty() is atomic. It takes bh->b_folio->mapping->i_private_lock,
  1054. * i_pages lock and mapping->host->i_lock.
  1055. */
  1056. void mark_buffer_dirty(struct buffer_head *bh)
  1057. {
  1058. WARN_ON_ONCE(!buffer_uptodate(bh));
  1059. trace_block_dirty_buffer(bh);
  1060. /*
  1061. * Very *carefully* optimize the it-is-already-dirty case.
  1062. *
  1063. * Don't let the final "is it dirty" escape to before we
  1064. * perhaps modified the buffer.
  1065. */
  1066. if (buffer_dirty(bh)) {
  1067. smp_mb();
  1068. if (buffer_dirty(bh))
  1069. return;
  1070. }
  1071. if (!test_set_buffer_dirty(bh)) {
  1072. struct folio *folio = bh->b_folio;
  1073. struct address_space *mapping = NULL;
  1074. folio_memcg_lock(folio);
  1075. if (!folio_test_set_dirty(folio)) {
  1076. mapping = folio->mapping;
  1077. if (mapping)
  1078. __folio_mark_dirty(folio, mapping, 0);
  1079. }
  1080. folio_memcg_unlock(folio);
  1081. if (mapping)
  1082. __mark_inode_dirty(mapping->host, I_DIRTY_PAGES);
  1083. }
  1084. }
  1085. EXPORT_SYMBOL(mark_buffer_dirty);
  1086. void mark_buffer_write_io_error(struct buffer_head *bh)
  1087. {
  1088. set_buffer_write_io_error(bh);
  1089. /* FIXME: do we need to set this in both places? */
  1090. if (bh->b_folio && bh->b_folio->mapping)
  1091. mapping_set_error(bh->b_folio->mapping, -EIO);
  1092. if (bh->b_assoc_map) {
  1093. mapping_set_error(bh->b_assoc_map, -EIO);
  1094. errseq_set(&bh->b_assoc_map->host->i_sb->s_wb_err, -EIO);
  1095. }
  1096. }
  1097. EXPORT_SYMBOL(mark_buffer_write_io_error);
  1098. /**
  1099. * __brelse - Release a buffer.
  1100. * @bh: The buffer to release.
  1101. *
  1102. * This variant of brelse() can be called if @bh is guaranteed to not be NULL.
  1103. */
  1104. void __brelse(struct buffer_head *bh)
  1105. {
  1106. if (atomic_read(&bh->b_count)) {
  1107. put_bh(bh);
  1108. return;
  1109. }
  1110. WARN(1, KERN_ERR "VFS: brelse: Trying to free free buffer\n");
  1111. }
  1112. EXPORT_SYMBOL(__brelse);
  1113. /**
  1114. * __bforget - Discard any dirty data in a buffer.
  1115. * @bh: The buffer to forget.
  1116. *
  1117. * This variant of bforget() can be called if @bh is guaranteed to not
  1118. * be NULL.
  1119. */
  1120. void __bforget(struct buffer_head *bh)
  1121. {
  1122. clear_buffer_dirty(bh);
  1123. if (bh->b_assoc_map) {
  1124. struct address_space *buffer_mapping = bh->b_folio->mapping;
  1125. spin_lock(&buffer_mapping->i_private_lock);
  1126. list_del_init(&bh->b_assoc_buffers);
  1127. bh->b_assoc_map = NULL;
  1128. spin_unlock(&buffer_mapping->i_private_lock);
  1129. }
  1130. __brelse(bh);
  1131. }
  1132. EXPORT_SYMBOL(__bforget);
  1133. static struct buffer_head *__bread_slow(struct buffer_head *bh)
  1134. {
  1135. lock_buffer(bh);
  1136. if (buffer_uptodate(bh)) {
  1137. unlock_buffer(bh);
  1138. return bh;
  1139. } else {
  1140. get_bh(bh);
  1141. bh->b_end_io = end_buffer_read_sync;
  1142. submit_bh(REQ_OP_READ, bh);
  1143. wait_on_buffer(bh);
  1144. if (buffer_uptodate(bh))
  1145. return bh;
  1146. }
  1147. brelse(bh);
  1148. return NULL;
  1149. }
  1150. /*
  1151. * Per-cpu buffer LRU implementation. To reduce the cost of __find_get_block().
  1152. * The bhs[] array is sorted - newest buffer is at bhs[0]. Buffers have their
  1153. * refcount elevated by one when they're in an LRU. A buffer can only appear
  1154. * once in a particular CPU's LRU. A single buffer can be present in multiple
  1155. * CPU's LRUs at the same time.
  1156. *
  1157. * This is a transparent caching front-end to sb_bread(), sb_getblk() and
  1158. * sb_find_get_block().
  1159. *
  1160. * The LRUs themselves only need locking against invalidate_bh_lrus. We use
  1161. * a local interrupt disable for that.
  1162. */
  1163. #define BH_LRU_SIZE 16
  1164. struct bh_lru {
  1165. struct buffer_head *bhs[BH_LRU_SIZE];
  1166. };
  1167. static DEFINE_PER_CPU(struct bh_lru, bh_lrus) = {{ NULL }};
  1168. #ifdef CONFIG_SMP
  1169. #define bh_lru_lock() local_irq_disable()
  1170. #define bh_lru_unlock() local_irq_enable()
  1171. #else
  1172. #define bh_lru_lock() preempt_disable()
  1173. #define bh_lru_unlock() preempt_enable()
  1174. #endif
  1175. static inline void check_irqs_on(void)
  1176. {
  1177. #ifdef irqs_disabled
  1178. BUG_ON(irqs_disabled());
  1179. #endif
  1180. }
  1181. /*
  1182. * Install a buffer_head into this cpu's LRU. If not already in the LRU, it is
  1183. * inserted at the front, and the buffer_head at the back if any is evicted.
  1184. * Or, if already in the LRU it is moved to the front.
  1185. */
  1186. static void bh_lru_install(struct buffer_head *bh)
  1187. {
  1188. struct buffer_head *evictee = bh;
  1189. struct bh_lru *b;
  1190. int i;
  1191. check_irqs_on();
  1192. bh_lru_lock();
  1193. /*
  1194. * the refcount of buffer_head in bh_lru prevents dropping the
  1195. * attached page(i.e., try_to_free_buffers) so it could cause
  1196. * failing page migration.
  1197. * Skip putting upcoming bh into bh_lru until migration is done.
  1198. */
  1199. if (lru_cache_disabled() || cpu_is_isolated(smp_processor_id())) {
  1200. bh_lru_unlock();
  1201. return;
  1202. }
  1203. b = this_cpu_ptr(&bh_lrus);
  1204. for (i = 0; i < BH_LRU_SIZE; i++) {
  1205. swap(evictee, b->bhs[i]);
  1206. if (evictee == bh) {
  1207. bh_lru_unlock();
  1208. return;
  1209. }
  1210. }
  1211. get_bh(bh);
  1212. bh_lru_unlock();
  1213. brelse(evictee);
  1214. }
  1215. /*
  1216. * Look up the bh in this cpu's LRU. If it's there, move it to the head.
  1217. */
  1218. static struct buffer_head *
  1219. lookup_bh_lru(struct block_device *bdev, sector_t block, unsigned size)
  1220. {
  1221. struct buffer_head *ret = NULL;
  1222. unsigned int i;
  1223. check_irqs_on();
  1224. bh_lru_lock();
  1225. if (cpu_is_isolated(smp_processor_id())) {
  1226. bh_lru_unlock();
  1227. return NULL;
  1228. }
  1229. for (i = 0; i < BH_LRU_SIZE; i++) {
  1230. struct buffer_head *bh = __this_cpu_read(bh_lrus.bhs[i]);
  1231. if (bh && bh->b_blocknr == block && bh->b_bdev == bdev &&
  1232. bh->b_size == size) {
  1233. if (i) {
  1234. while (i) {
  1235. __this_cpu_write(bh_lrus.bhs[i],
  1236. __this_cpu_read(bh_lrus.bhs[i - 1]));
  1237. i--;
  1238. }
  1239. __this_cpu_write(bh_lrus.bhs[0], bh);
  1240. }
  1241. get_bh(bh);
  1242. ret = bh;
  1243. break;
  1244. }
  1245. }
  1246. bh_lru_unlock();
  1247. return ret;
  1248. }
  1249. /*
  1250. * Perform a pagecache lookup for the matching buffer. If it's there, refresh
  1251. * it in the LRU and mark it as accessed. If it is not present then return
  1252. * NULL
  1253. */
  1254. struct buffer_head *
  1255. __find_get_block(struct block_device *bdev, sector_t block, unsigned size)
  1256. {
  1257. struct buffer_head *bh = lookup_bh_lru(bdev, block, size);
  1258. if (bh == NULL) {
  1259. /* __find_get_block_slow will mark the page accessed */
  1260. bh = __find_get_block_slow(bdev, block);
  1261. if (bh)
  1262. bh_lru_install(bh);
  1263. } else
  1264. touch_buffer(bh);
  1265. return bh;
  1266. }
  1267. EXPORT_SYMBOL(__find_get_block);
  1268. /**
  1269. * bdev_getblk - Get a buffer_head in a block device's buffer cache.
  1270. * @bdev: The block device.
  1271. * @block: The block number.
  1272. * @size: The size of buffer_heads for this @bdev.
  1273. * @gfp: The memory allocation flags to use.
  1274. *
  1275. * The returned buffer head has its reference count incremented, but is
  1276. * not locked. The caller should call brelse() when it has finished
  1277. * with the buffer. The buffer may not be uptodate. If needed, the
  1278. * caller can bring it uptodate either by reading it or overwriting it.
  1279. *
  1280. * Return: The buffer head, or NULL if memory could not be allocated.
  1281. */
  1282. struct buffer_head *bdev_getblk(struct block_device *bdev, sector_t block,
  1283. unsigned size, gfp_t gfp)
  1284. {
  1285. struct buffer_head *bh = __find_get_block(bdev, block, size);
  1286. might_alloc(gfp);
  1287. if (bh)
  1288. return bh;
  1289. return __getblk_slow(bdev, block, size, gfp);
  1290. }
  1291. EXPORT_SYMBOL(bdev_getblk);
  1292. /*
  1293. * Do async read-ahead on a buffer..
  1294. */
  1295. void __breadahead(struct block_device *bdev, sector_t block, unsigned size)
  1296. {
  1297. struct buffer_head *bh = bdev_getblk(bdev, block, size,
  1298. GFP_NOWAIT | __GFP_MOVABLE);
  1299. if (likely(bh)) {
  1300. bh_readahead(bh, REQ_RAHEAD);
  1301. brelse(bh);
  1302. }
  1303. }
  1304. EXPORT_SYMBOL(__breadahead);
  1305. /**
  1306. * __bread_gfp() - Read a block.
  1307. * @bdev: The block device to read from.
  1308. * @block: Block number in units of block size.
  1309. * @size: The block size of this device in bytes.
  1310. * @gfp: Not page allocation flags; see below.
  1311. *
  1312. * You are not expected to call this function. You should use one of
  1313. * sb_bread(), sb_bread_unmovable() or __bread().
  1314. *
  1315. * Read a specified block, and return the buffer head that refers to it.
  1316. * If @gfp is 0, the memory will be allocated using the block device's
  1317. * default GFP flags. If @gfp is __GFP_MOVABLE, the memory may be
  1318. * allocated from a movable area. Do not pass in a complete set of
  1319. * GFP flags.
  1320. *
  1321. * The returned buffer head has its refcount increased. The caller should
  1322. * call brelse() when it has finished with the buffer.
  1323. *
  1324. * Context: May sleep waiting for I/O.
  1325. * Return: NULL if the block was unreadable.
  1326. */
  1327. struct buffer_head *__bread_gfp(struct block_device *bdev, sector_t block,
  1328. unsigned size, gfp_t gfp)
  1329. {
  1330. struct buffer_head *bh;
  1331. gfp |= mapping_gfp_constraint(bdev->bd_mapping, ~__GFP_FS);
  1332. /*
  1333. * Prefer looping in the allocator rather than here, at least that
  1334. * code knows what it's doing.
  1335. */
  1336. gfp |= __GFP_NOFAIL;
  1337. bh = bdev_getblk(bdev, block, size, gfp);
  1338. if (likely(bh) && !buffer_uptodate(bh))
  1339. bh = __bread_slow(bh);
  1340. return bh;
  1341. }
  1342. EXPORT_SYMBOL(__bread_gfp);
  1343. static void __invalidate_bh_lrus(struct bh_lru *b)
  1344. {
  1345. int i;
  1346. for (i = 0; i < BH_LRU_SIZE; i++) {
  1347. brelse(b->bhs[i]);
  1348. b->bhs[i] = NULL;
  1349. }
  1350. }
  1351. /*
  1352. * invalidate_bh_lrus() is called rarely - but not only at unmount.
  1353. * This doesn't race because it runs in each cpu either in irq
  1354. * or with preempt disabled.
  1355. */
  1356. static void invalidate_bh_lru(void *arg)
  1357. {
  1358. struct bh_lru *b = &get_cpu_var(bh_lrus);
  1359. __invalidate_bh_lrus(b);
  1360. put_cpu_var(bh_lrus);
  1361. }
  1362. bool has_bh_in_lru(int cpu, void *dummy)
  1363. {
  1364. struct bh_lru *b = per_cpu_ptr(&bh_lrus, cpu);
  1365. int i;
  1366. for (i = 0; i < BH_LRU_SIZE; i++) {
  1367. if (b->bhs[i])
  1368. return true;
  1369. }
  1370. return false;
  1371. }
  1372. void invalidate_bh_lrus(void)
  1373. {
  1374. on_each_cpu_cond(has_bh_in_lru, invalidate_bh_lru, NULL, 1);
  1375. }
  1376. EXPORT_SYMBOL_GPL(invalidate_bh_lrus);
  1377. /*
  1378. * It's called from workqueue context so we need a bh_lru_lock to close
  1379. * the race with preemption/irq.
  1380. */
  1381. void invalidate_bh_lrus_cpu(void)
  1382. {
  1383. struct bh_lru *b;
  1384. bh_lru_lock();
  1385. b = this_cpu_ptr(&bh_lrus);
  1386. __invalidate_bh_lrus(b);
  1387. bh_lru_unlock();
  1388. }
  1389. void folio_set_bh(struct buffer_head *bh, struct folio *folio,
  1390. unsigned long offset)
  1391. {
  1392. bh->b_folio = folio;
  1393. BUG_ON(offset >= folio_size(folio));
  1394. if (folio_test_highmem(folio))
  1395. /*
  1396. * This catches illegal uses and preserves the offset:
  1397. */
  1398. bh->b_data = (char *)(0 + offset);
  1399. else
  1400. bh->b_data = folio_address(folio) + offset;
  1401. }
  1402. EXPORT_SYMBOL(folio_set_bh);
  1403. /*
  1404. * Called when truncating a buffer on a page completely.
  1405. */
  1406. /* Bits that are cleared during an invalidate */
  1407. #define BUFFER_FLAGS_DISCARD \
  1408. (1 << BH_Mapped | 1 << BH_New | 1 << BH_Req | \
  1409. 1 << BH_Delay | 1 << BH_Unwritten)
  1410. static void discard_buffer(struct buffer_head * bh)
  1411. {
  1412. unsigned long b_state;
  1413. lock_buffer(bh);
  1414. clear_buffer_dirty(bh);
  1415. bh->b_bdev = NULL;
  1416. b_state = READ_ONCE(bh->b_state);
  1417. do {
  1418. } while (!try_cmpxchg(&bh->b_state, &b_state,
  1419. b_state & ~BUFFER_FLAGS_DISCARD));
  1420. unlock_buffer(bh);
  1421. }
  1422. /**
  1423. * block_invalidate_folio - Invalidate part or all of a buffer-backed folio.
  1424. * @folio: The folio which is affected.
  1425. * @offset: start of the range to invalidate
  1426. * @length: length of the range to invalidate
  1427. *
  1428. * block_invalidate_folio() is called when all or part of the folio has been
  1429. * invalidated by a truncate operation.
  1430. *
  1431. * block_invalidate_folio() does not have to release all buffers, but it must
  1432. * ensure that no dirty buffer is left outside @offset and that no I/O
  1433. * is underway against any of the blocks which are outside the truncation
  1434. * point. Because the caller is about to free (and possibly reuse) those
  1435. * blocks on-disk.
  1436. */
  1437. void block_invalidate_folio(struct folio *folio, size_t offset, size_t length)
  1438. {
  1439. struct buffer_head *head, *bh, *next;
  1440. size_t curr_off = 0;
  1441. size_t stop = length + offset;
  1442. BUG_ON(!folio_test_locked(folio));
  1443. /*
  1444. * Check for overflow
  1445. */
  1446. BUG_ON(stop > folio_size(folio) || stop < length);
  1447. head = folio_buffers(folio);
  1448. if (!head)
  1449. return;
  1450. bh = head;
  1451. do {
  1452. size_t next_off = curr_off + bh->b_size;
  1453. next = bh->b_this_page;
  1454. /*
  1455. * Are we still fully in range ?
  1456. */
  1457. if (next_off > stop)
  1458. goto out;
  1459. /*
  1460. * is this block fully invalidated?
  1461. */
  1462. if (offset <= curr_off)
  1463. discard_buffer(bh);
  1464. curr_off = next_off;
  1465. bh = next;
  1466. } while (bh != head);
  1467. /*
  1468. * We release buffers only if the entire folio is being invalidated.
  1469. * The get_block cached value has been unconditionally invalidated,
  1470. * so real IO is not possible anymore.
  1471. */
  1472. if (length == folio_size(folio))
  1473. filemap_release_folio(folio, 0);
  1474. out:
  1475. return;
  1476. }
  1477. EXPORT_SYMBOL(block_invalidate_folio);
  1478. /*
  1479. * We attach and possibly dirty the buffers atomically wrt
  1480. * block_dirty_folio() via i_private_lock. try_to_free_buffers
  1481. * is already excluded via the folio lock.
  1482. */
  1483. struct buffer_head *create_empty_buffers(struct folio *folio,
  1484. unsigned long blocksize, unsigned long b_state)
  1485. {
  1486. struct buffer_head *bh, *head, *tail;
  1487. gfp_t gfp = GFP_NOFS | __GFP_ACCOUNT | __GFP_NOFAIL;
  1488. head = folio_alloc_buffers(folio, blocksize, gfp);
  1489. bh = head;
  1490. do {
  1491. bh->b_state |= b_state;
  1492. tail = bh;
  1493. bh = bh->b_this_page;
  1494. } while (bh);
  1495. tail->b_this_page = head;
  1496. spin_lock(&folio->mapping->i_private_lock);
  1497. if (folio_test_uptodate(folio) || folio_test_dirty(folio)) {
  1498. bh = head;
  1499. do {
  1500. if (folio_test_dirty(folio))
  1501. set_buffer_dirty(bh);
  1502. if (folio_test_uptodate(folio))
  1503. set_buffer_uptodate(bh);
  1504. bh = bh->b_this_page;
  1505. } while (bh != head);
  1506. }
  1507. folio_attach_private(folio, head);
  1508. spin_unlock(&folio->mapping->i_private_lock);
  1509. return head;
  1510. }
  1511. EXPORT_SYMBOL(create_empty_buffers);
  1512. /**
  1513. * clean_bdev_aliases: clean a range of buffers in block device
  1514. * @bdev: Block device to clean buffers in
  1515. * @block: Start of a range of blocks to clean
  1516. * @len: Number of blocks to clean
  1517. *
  1518. * We are taking a range of blocks for data and we don't want writeback of any
  1519. * buffer-cache aliases starting from return from this function and until the
  1520. * moment when something will explicitly mark the buffer dirty (hopefully that
  1521. * will not happen until we will free that block ;-) We don't even need to mark
  1522. * it not-uptodate - nobody can expect anything from a newly allocated buffer
  1523. * anyway. We used to use unmap_buffer() for such invalidation, but that was
  1524. * wrong. We definitely don't want to mark the alias unmapped, for example - it
  1525. * would confuse anyone who might pick it with bread() afterwards...
  1526. *
  1527. * Also.. Note that bforget() doesn't lock the buffer. So there can be
  1528. * writeout I/O going on against recently-freed buffers. We don't wait on that
  1529. * I/O in bforget() - it's more efficient to wait on the I/O only if we really
  1530. * need to. That happens here.
  1531. */
  1532. void clean_bdev_aliases(struct block_device *bdev, sector_t block, sector_t len)
  1533. {
  1534. struct address_space *bd_mapping = bdev->bd_mapping;
  1535. const int blkbits = bd_mapping->host->i_blkbits;
  1536. struct folio_batch fbatch;
  1537. pgoff_t index = ((loff_t)block << blkbits) / PAGE_SIZE;
  1538. pgoff_t end;
  1539. int i, count;
  1540. struct buffer_head *bh;
  1541. struct buffer_head *head;
  1542. end = ((loff_t)(block + len - 1) << blkbits) / PAGE_SIZE;
  1543. folio_batch_init(&fbatch);
  1544. while (filemap_get_folios(bd_mapping, &index, end, &fbatch)) {
  1545. count = folio_batch_count(&fbatch);
  1546. for (i = 0; i < count; i++) {
  1547. struct folio *folio = fbatch.folios[i];
  1548. if (!folio_buffers(folio))
  1549. continue;
  1550. /*
  1551. * We use folio lock instead of bd_mapping->i_private_lock
  1552. * to pin buffers here since we can afford to sleep and
  1553. * it scales better than a global spinlock lock.
  1554. */
  1555. folio_lock(folio);
  1556. /* Recheck when the folio is locked which pins bhs */
  1557. head = folio_buffers(folio);
  1558. if (!head)
  1559. goto unlock_page;
  1560. bh = head;
  1561. do {
  1562. if (!buffer_mapped(bh) || (bh->b_blocknr < block))
  1563. goto next;
  1564. if (bh->b_blocknr >= block + len)
  1565. break;
  1566. clear_buffer_dirty(bh);
  1567. wait_on_buffer(bh);
  1568. clear_buffer_req(bh);
  1569. next:
  1570. bh = bh->b_this_page;
  1571. } while (bh != head);
  1572. unlock_page:
  1573. folio_unlock(folio);
  1574. }
  1575. folio_batch_release(&fbatch);
  1576. cond_resched();
  1577. /* End of range already reached? */
  1578. if (index > end || !index)
  1579. break;
  1580. }
  1581. }
  1582. EXPORT_SYMBOL(clean_bdev_aliases);
  1583. static struct buffer_head *folio_create_buffers(struct folio *folio,
  1584. struct inode *inode,
  1585. unsigned int b_state)
  1586. {
  1587. struct buffer_head *bh;
  1588. BUG_ON(!folio_test_locked(folio));
  1589. bh = folio_buffers(folio);
  1590. if (!bh)
  1591. bh = create_empty_buffers(folio,
  1592. 1 << READ_ONCE(inode->i_blkbits), b_state);
  1593. return bh;
  1594. }
  1595. /*
  1596. * NOTE! All mapped/uptodate combinations are valid:
  1597. *
  1598. * Mapped Uptodate Meaning
  1599. *
  1600. * No No "unknown" - must do get_block()
  1601. * No Yes "hole" - zero-filled
  1602. * Yes No "allocated" - allocated on disk, not read in
  1603. * Yes Yes "valid" - allocated and up-to-date in memory.
  1604. *
  1605. * "Dirty" is valid only with the last case (mapped+uptodate).
  1606. */
  1607. /*
  1608. * While block_write_full_folio is writing back the dirty buffers under
  1609. * the page lock, whoever dirtied the buffers may decide to clean them
  1610. * again at any time. We handle that by only looking at the buffer
  1611. * state inside lock_buffer().
  1612. *
  1613. * If block_write_full_folio() is called for regular writeback
  1614. * (wbc->sync_mode == WB_SYNC_NONE) then it will redirty a page which has a
  1615. * locked buffer. This only can happen if someone has written the buffer
  1616. * directly, with submit_bh(). At the address_space level PageWriteback
  1617. * prevents this contention from occurring.
  1618. *
  1619. * If block_write_full_folio() is called with wbc->sync_mode ==
  1620. * WB_SYNC_ALL, the writes are posted using REQ_SYNC; this
  1621. * causes the writes to be flagged as synchronous writes.
  1622. */
  1623. int __block_write_full_folio(struct inode *inode, struct folio *folio,
  1624. get_block_t *get_block, struct writeback_control *wbc)
  1625. {
  1626. int err;
  1627. sector_t block;
  1628. sector_t last_block;
  1629. struct buffer_head *bh, *head;
  1630. size_t blocksize;
  1631. int nr_underway = 0;
  1632. blk_opf_t write_flags = wbc_to_write_flags(wbc);
  1633. head = folio_create_buffers(folio, inode,
  1634. (1 << BH_Dirty) | (1 << BH_Uptodate));
  1635. /*
  1636. * Be very careful. We have no exclusion from block_dirty_folio
  1637. * here, and the (potentially unmapped) buffers may become dirty at
  1638. * any time. If a buffer becomes dirty here after we've inspected it
  1639. * then we just miss that fact, and the folio stays dirty.
  1640. *
  1641. * Buffers outside i_size may be dirtied by block_dirty_folio;
  1642. * handle that here by just cleaning them.
  1643. */
  1644. bh = head;
  1645. blocksize = bh->b_size;
  1646. block = div_u64(folio_pos(folio), blocksize);
  1647. last_block = div_u64(i_size_read(inode) - 1, blocksize);
  1648. /*
  1649. * Get all the dirty buffers mapped to disk addresses and
  1650. * handle any aliases from the underlying blockdev's mapping.
  1651. */
  1652. do {
  1653. if (block > last_block) {
  1654. /*
  1655. * mapped buffers outside i_size will occur, because
  1656. * this folio can be outside i_size when there is a
  1657. * truncate in progress.
  1658. */
  1659. /*
  1660. * The buffer was zeroed by block_write_full_folio()
  1661. */
  1662. clear_buffer_dirty(bh);
  1663. set_buffer_uptodate(bh);
  1664. } else if ((!buffer_mapped(bh) || buffer_delay(bh)) &&
  1665. buffer_dirty(bh)) {
  1666. WARN_ON(bh->b_size != blocksize);
  1667. err = get_block(inode, block, bh, 1);
  1668. if (err)
  1669. goto recover;
  1670. clear_buffer_delay(bh);
  1671. if (buffer_new(bh)) {
  1672. /* blockdev mappings never come here */
  1673. clear_buffer_new(bh);
  1674. clean_bdev_bh_alias(bh);
  1675. }
  1676. }
  1677. bh = bh->b_this_page;
  1678. block++;
  1679. } while (bh != head);
  1680. do {
  1681. if (!buffer_mapped(bh))
  1682. continue;
  1683. /*
  1684. * If it's a fully non-blocking write attempt and we cannot
  1685. * lock the buffer then redirty the folio. Note that this can
  1686. * potentially cause a busy-wait loop from writeback threads
  1687. * and kswapd activity, but those code paths have their own
  1688. * higher-level throttling.
  1689. */
  1690. if (wbc->sync_mode != WB_SYNC_NONE) {
  1691. lock_buffer(bh);
  1692. } else if (!trylock_buffer(bh)) {
  1693. folio_redirty_for_writepage(wbc, folio);
  1694. continue;
  1695. }
  1696. if (test_clear_buffer_dirty(bh)) {
  1697. mark_buffer_async_write_endio(bh,
  1698. end_buffer_async_write);
  1699. } else {
  1700. unlock_buffer(bh);
  1701. }
  1702. } while ((bh = bh->b_this_page) != head);
  1703. /*
  1704. * The folio and its buffers are protected by the writeback flag,
  1705. * so we can drop the bh refcounts early.
  1706. */
  1707. BUG_ON(folio_test_writeback(folio));
  1708. folio_start_writeback(folio);
  1709. do {
  1710. struct buffer_head *next = bh->b_this_page;
  1711. if (buffer_async_write(bh)) {
  1712. submit_bh_wbc(REQ_OP_WRITE | write_flags, bh,
  1713. inode->i_write_hint, wbc);
  1714. nr_underway++;
  1715. }
  1716. bh = next;
  1717. } while (bh != head);
  1718. folio_unlock(folio);
  1719. err = 0;
  1720. done:
  1721. if (nr_underway == 0) {
  1722. /*
  1723. * The folio was marked dirty, but the buffers were
  1724. * clean. Someone wrote them back by hand with
  1725. * write_dirty_buffer/submit_bh. A rare case.
  1726. */
  1727. folio_end_writeback(folio);
  1728. /*
  1729. * The folio and buffer_heads can be released at any time from
  1730. * here on.
  1731. */
  1732. }
  1733. return err;
  1734. recover:
  1735. /*
  1736. * ENOSPC, or some other error. We may already have added some
  1737. * blocks to the file, so we need to write these out to avoid
  1738. * exposing stale data.
  1739. * The folio is currently locked and not marked for writeback
  1740. */
  1741. bh = head;
  1742. /* Recovery: lock and submit the mapped buffers */
  1743. do {
  1744. if (buffer_mapped(bh) && buffer_dirty(bh) &&
  1745. !buffer_delay(bh)) {
  1746. lock_buffer(bh);
  1747. mark_buffer_async_write_endio(bh,
  1748. end_buffer_async_write);
  1749. } else {
  1750. /*
  1751. * The buffer may have been set dirty during
  1752. * attachment to a dirty folio.
  1753. */
  1754. clear_buffer_dirty(bh);
  1755. }
  1756. } while ((bh = bh->b_this_page) != head);
  1757. BUG_ON(folio_test_writeback(folio));
  1758. mapping_set_error(folio->mapping, err);
  1759. folio_start_writeback(folio);
  1760. do {
  1761. struct buffer_head *next = bh->b_this_page;
  1762. if (buffer_async_write(bh)) {
  1763. clear_buffer_dirty(bh);
  1764. submit_bh_wbc(REQ_OP_WRITE | write_flags, bh,
  1765. inode->i_write_hint, wbc);
  1766. nr_underway++;
  1767. }
  1768. bh = next;
  1769. } while (bh != head);
  1770. folio_unlock(folio);
  1771. goto done;
  1772. }
  1773. EXPORT_SYMBOL(__block_write_full_folio);
  1774. /*
  1775. * If a folio has any new buffers, zero them out here, and mark them uptodate
  1776. * and dirty so they'll be written out (in order to prevent uninitialised
  1777. * block data from leaking). And clear the new bit.
  1778. */
  1779. void folio_zero_new_buffers(struct folio *folio, size_t from, size_t to)
  1780. {
  1781. size_t block_start, block_end;
  1782. struct buffer_head *head, *bh;
  1783. BUG_ON(!folio_test_locked(folio));
  1784. head = folio_buffers(folio);
  1785. if (!head)
  1786. return;
  1787. bh = head;
  1788. block_start = 0;
  1789. do {
  1790. block_end = block_start + bh->b_size;
  1791. if (buffer_new(bh)) {
  1792. if (block_end > from && block_start < to) {
  1793. if (!folio_test_uptodate(folio)) {
  1794. size_t start, xend;
  1795. start = max(from, block_start);
  1796. xend = min(to, block_end);
  1797. folio_zero_segment(folio, start, xend);
  1798. set_buffer_uptodate(bh);
  1799. }
  1800. clear_buffer_new(bh);
  1801. mark_buffer_dirty(bh);
  1802. }
  1803. }
  1804. block_start = block_end;
  1805. bh = bh->b_this_page;
  1806. } while (bh != head);
  1807. }
  1808. EXPORT_SYMBOL(folio_zero_new_buffers);
  1809. static int
  1810. iomap_to_bh(struct inode *inode, sector_t block, struct buffer_head *bh,
  1811. const struct iomap *iomap)
  1812. {
  1813. loff_t offset = (loff_t)block << inode->i_blkbits;
  1814. bh->b_bdev = iomap->bdev;
  1815. /*
  1816. * Block points to offset in file we need to map, iomap contains
  1817. * the offset at which the map starts. If the map ends before the
  1818. * current block, then do not map the buffer and let the caller
  1819. * handle it.
  1820. */
  1821. if (offset >= iomap->offset + iomap->length)
  1822. return -EIO;
  1823. switch (iomap->type) {
  1824. case IOMAP_HOLE:
  1825. /*
  1826. * If the buffer is not up to date or beyond the current EOF,
  1827. * we need to mark it as new to ensure sub-block zeroing is
  1828. * executed if necessary.
  1829. */
  1830. if (!buffer_uptodate(bh) ||
  1831. (offset >= i_size_read(inode)))
  1832. set_buffer_new(bh);
  1833. return 0;
  1834. case IOMAP_DELALLOC:
  1835. if (!buffer_uptodate(bh) ||
  1836. (offset >= i_size_read(inode)))
  1837. set_buffer_new(bh);
  1838. set_buffer_uptodate(bh);
  1839. set_buffer_mapped(bh);
  1840. set_buffer_delay(bh);
  1841. return 0;
  1842. case IOMAP_UNWRITTEN:
  1843. /*
  1844. * For unwritten regions, we always need to ensure that regions
  1845. * in the block we are not writing to are zeroed. Mark the
  1846. * buffer as new to ensure this.
  1847. */
  1848. set_buffer_new(bh);
  1849. set_buffer_unwritten(bh);
  1850. fallthrough;
  1851. case IOMAP_MAPPED:
  1852. if ((iomap->flags & IOMAP_F_NEW) ||
  1853. offset >= i_size_read(inode)) {
  1854. /*
  1855. * This can happen if truncating the block device races
  1856. * with the check in the caller as i_size updates on
  1857. * block devices aren't synchronized by i_rwsem for
  1858. * block devices.
  1859. */
  1860. if (S_ISBLK(inode->i_mode))
  1861. return -EIO;
  1862. set_buffer_new(bh);
  1863. }
  1864. bh->b_blocknr = (iomap->addr + offset - iomap->offset) >>
  1865. inode->i_blkbits;
  1866. set_buffer_mapped(bh);
  1867. return 0;
  1868. default:
  1869. WARN_ON_ONCE(1);
  1870. return -EIO;
  1871. }
  1872. }
  1873. int __block_write_begin_int(struct folio *folio, loff_t pos, unsigned len,
  1874. get_block_t *get_block, const struct iomap *iomap)
  1875. {
  1876. size_t from = offset_in_folio(folio, pos);
  1877. size_t to = from + len;
  1878. struct inode *inode = folio->mapping->host;
  1879. size_t block_start, block_end;
  1880. sector_t block;
  1881. int err = 0;
  1882. size_t blocksize;
  1883. struct buffer_head *bh, *head, *wait[2], **wait_bh=wait;
  1884. BUG_ON(!folio_test_locked(folio));
  1885. BUG_ON(to > folio_size(folio));
  1886. BUG_ON(from > to);
  1887. head = folio_create_buffers(folio, inode, 0);
  1888. blocksize = head->b_size;
  1889. block = div_u64(folio_pos(folio), blocksize);
  1890. for (bh = head, block_start = 0; bh != head || !block_start;
  1891. block++, block_start=block_end, bh = bh->b_this_page) {
  1892. block_end = block_start + blocksize;
  1893. if (block_end <= from || block_start >= to) {
  1894. if (folio_test_uptodate(folio)) {
  1895. if (!buffer_uptodate(bh))
  1896. set_buffer_uptodate(bh);
  1897. }
  1898. continue;
  1899. }
  1900. if (buffer_new(bh))
  1901. clear_buffer_new(bh);
  1902. if (!buffer_mapped(bh)) {
  1903. WARN_ON(bh->b_size != blocksize);
  1904. if (get_block)
  1905. err = get_block(inode, block, bh, 1);
  1906. else
  1907. err = iomap_to_bh(inode, block, bh, iomap);
  1908. if (err)
  1909. break;
  1910. if (buffer_new(bh)) {
  1911. clean_bdev_bh_alias(bh);
  1912. if (folio_test_uptodate(folio)) {
  1913. clear_buffer_new(bh);
  1914. set_buffer_uptodate(bh);
  1915. mark_buffer_dirty(bh);
  1916. continue;
  1917. }
  1918. if (block_end > to || block_start < from)
  1919. folio_zero_segments(folio,
  1920. to, block_end,
  1921. block_start, from);
  1922. continue;
  1923. }
  1924. }
  1925. if (folio_test_uptodate(folio)) {
  1926. if (!buffer_uptodate(bh))
  1927. set_buffer_uptodate(bh);
  1928. continue;
  1929. }
  1930. if (!buffer_uptodate(bh) && !buffer_delay(bh) &&
  1931. !buffer_unwritten(bh) &&
  1932. (block_start < from || block_end > to)) {
  1933. bh_read_nowait(bh, 0);
  1934. *wait_bh++=bh;
  1935. }
  1936. }
  1937. /*
  1938. * If we issued read requests - let them complete.
  1939. */
  1940. while(wait_bh > wait) {
  1941. wait_on_buffer(*--wait_bh);
  1942. if (!buffer_uptodate(*wait_bh))
  1943. err = -EIO;
  1944. }
  1945. if (unlikely(err))
  1946. folio_zero_new_buffers(folio, from, to);
  1947. return err;
  1948. }
  1949. int __block_write_begin(struct folio *folio, loff_t pos, unsigned len,
  1950. get_block_t *get_block)
  1951. {
  1952. return __block_write_begin_int(folio, pos, len, get_block, NULL);
  1953. }
  1954. EXPORT_SYMBOL(__block_write_begin);
  1955. static void __block_commit_write(struct folio *folio, size_t from, size_t to)
  1956. {
  1957. size_t block_start, block_end;
  1958. bool partial = false;
  1959. unsigned blocksize;
  1960. struct buffer_head *bh, *head;
  1961. bh = head = folio_buffers(folio);
  1962. if (!bh)
  1963. return;
  1964. blocksize = bh->b_size;
  1965. block_start = 0;
  1966. do {
  1967. block_end = block_start + blocksize;
  1968. if (block_end <= from || block_start >= to) {
  1969. if (!buffer_uptodate(bh))
  1970. partial = true;
  1971. } else {
  1972. set_buffer_uptodate(bh);
  1973. mark_buffer_dirty(bh);
  1974. }
  1975. if (buffer_new(bh))
  1976. clear_buffer_new(bh);
  1977. block_start = block_end;
  1978. bh = bh->b_this_page;
  1979. } while (bh != head);
  1980. /*
  1981. * If this is a partial write which happened to make all buffers
  1982. * uptodate then we can optimize away a bogus read_folio() for
  1983. * the next read(). Here we 'discover' whether the folio went
  1984. * uptodate as a result of this (potentially partial) write.
  1985. */
  1986. if (!partial)
  1987. folio_mark_uptodate(folio);
  1988. }
  1989. /*
  1990. * block_write_begin takes care of the basic task of block allocation and
  1991. * bringing partial write blocks uptodate first.
  1992. *
  1993. * The filesystem needs to handle block truncation upon failure.
  1994. */
  1995. int block_write_begin(struct address_space *mapping, loff_t pos, unsigned len,
  1996. struct folio **foliop, get_block_t *get_block)
  1997. {
  1998. pgoff_t index = pos >> PAGE_SHIFT;
  1999. struct folio *folio;
  2000. int status;
  2001. folio = __filemap_get_folio(mapping, index, FGP_WRITEBEGIN,
  2002. mapping_gfp_mask(mapping));
  2003. if (IS_ERR(folio))
  2004. return PTR_ERR(folio);
  2005. status = __block_write_begin_int(folio, pos, len, get_block, NULL);
  2006. if (unlikely(status)) {
  2007. folio_unlock(folio);
  2008. folio_put(folio);
  2009. folio = NULL;
  2010. }
  2011. *foliop = folio;
  2012. return status;
  2013. }
  2014. EXPORT_SYMBOL(block_write_begin);
  2015. int block_write_end(struct file *file, struct address_space *mapping,
  2016. loff_t pos, unsigned len, unsigned copied,
  2017. struct folio *folio, void *fsdata)
  2018. {
  2019. size_t start = pos - folio_pos(folio);
  2020. if (unlikely(copied < len)) {
  2021. /*
  2022. * The buffers that were written will now be uptodate, so
  2023. * we don't have to worry about a read_folio reading them
  2024. * and overwriting a partial write. However if we have
  2025. * encountered a short write and only partially written
  2026. * into a buffer, it will not be marked uptodate, so a
  2027. * read_folio might come in and destroy our partial write.
  2028. *
  2029. * Do the simplest thing, and just treat any short write to a
  2030. * non uptodate folio as a zero-length write, and force the
  2031. * caller to redo the whole thing.
  2032. */
  2033. if (!folio_test_uptodate(folio))
  2034. copied = 0;
  2035. folio_zero_new_buffers(folio, start+copied, start+len);
  2036. }
  2037. flush_dcache_folio(folio);
  2038. /* This could be a short (even 0-length) commit */
  2039. __block_commit_write(folio, start, start + copied);
  2040. return copied;
  2041. }
  2042. EXPORT_SYMBOL(block_write_end);
  2043. int generic_write_end(struct file *file, struct address_space *mapping,
  2044. loff_t pos, unsigned len, unsigned copied,
  2045. struct folio *folio, void *fsdata)
  2046. {
  2047. struct inode *inode = mapping->host;
  2048. loff_t old_size = inode->i_size;
  2049. bool i_size_changed = false;
  2050. copied = block_write_end(file, mapping, pos, len, copied, folio, fsdata);
  2051. /*
  2052. * No need to use i_size_read() here, the i_size cannot change under us
  2053. * because we hold i_rwsem.
  2054. *
  2055. * But it's important to update i_size while still holding folio lock:
  2056. * page writeout could otherwise come in and zero beyond i_size.
  2057. */
  2058. if (pos + copied > inode->i_size) {
  2059. i_size_write(inode, pos + copied);
  2060. i_size_changed = true;
  2061. }
  2062. folio_unlock(folio);
  2063. folio_put(folio);
  2064. if (old_size < pos)
  2065. pagecache_isize_extended(inode, old_size, pos);
  2066. /*
  2067. * Don't mark the inode dirty under page lock. First, it unnecessarily
  2068. * makes the holding time of page lock longer. Second, it forces lock
  2069. * ordering of page lock and transaction start for journaling
  2070. * filesystems.
  2071. */
  2072. if (i_size_changed)
  2073. mark_inode_dirty(inode);
  2074. return copied;
  2075. }
  2076. EXPORT_SYMBOL(generic_write_end);
  2077. /*
  2078. * block_is_partially_uptodate checks whether buffers within a folio are
  2079. * uptodate or not.
  2080. *
  2081. * Returns true if all buffers which correspond to the specified part
  2082. * of the folio are uptodate.
  2083. */
  2084. bool block_is_partially_uptodate(struct folio *folio, size_t from, size_t count)
  2085. {
  2086. unsigned block_start, block_end, blocksize;
  2087. unsigned to;
  2088. struct buffer_head *bh, *head;
  2089. bool ret = true;
  2090. head = folio_buffers(folio);
  2091. if (!head)
  2092. return false;
  2093. blocksize = head->b_size;
  2094. to = min_t(unsigned, folio_size(folio) - from, count);
  2095. to = from + to;
  2096. if (from < blocksize && to > folio_size(folio) - blocksize)
  2097. return false;
  2098. bh = head;
  2099. block_start = 0;
  2100. do {
  2101. block_end = block_start + blocksize;
  2102. if (block_end > from && block_start < to) {
  2103. if (!buffer_uptodate(bh)) {
  2104. ret = false;
  2105. break;
  2106. }
  2107. if (block_end >= to)
  2108. break;
  2109. }
  2110. block_start = block_end;
  2111. bh = bh->b_this_page;
  2112. } while (bh != head);
  2113. return ret;
  2114. }
  2115. EXPORT_SYMBOL(block_is_partially_uptodate);
  2116. /*
  2117. * Generic "read_folio" function for block devices that have the normal
  2118. * get_block functionality. This is most of the block device filesystems.
  2119. * Reads the folio asynchronously --- the unlock_buffer() and
  2120. * set/clear_buffer_uptodate() functions propagate buffer state into the
  2121. * folio once IO has completed.
  2122. */
  2123. int block_read_full_folio(struct folio *folio, get_block_t *get_block)
  2124. {
  2125. struct inode *inode = folio->mapping->host;
  2126. sector_t iblock, lblock;
  2127. struct buffer_head *bh, *head, *arr[MAX_BUF_PER_PAGE];
  2128. size_t blocksize;
  2129. int nr, i;
  2130. int fully_mapped = 1;
  2131. bool page_error = false;
  2132. loff_t limit = i_size_read(inode);
  2133. /* This is needed for ext4. */
  2134. if (IS_ENABLED(CONFIG_FS_VERITY) && IS_VERITY(inode))
  2135. limit = inode->i_sb->s_maxbytes;
  2136. VM_BUG_ON_FOLIO(folio_test_large(folio), folio);
  2137. head = folio_create_buffers(folio, inode, 0);
  2138. blocksize = head->b_size;
  2139. iblock = div_u64(folio_pos(folio), blocksize);
  2140. lblock = div_u64(limit + blocksize - 1, blocksize);
  2141. bh = head;
  2142. nr = 0;
  2143. i = 0;
  2144. do {
  2145. if (buffer_uptodate(bh))
  2146. continue;
  2147. if (!buffer_mapped(bh)) {
  2148. int err = 0;
  2149. fully_mapped = 0;
  2150. if (iblock < lblock) {
  2151. WARN_ON(bh->b_size != blocksize);
  2152. err = get_block(inode, iblock, bh, 0);
  2153. if (err)
  2154. page_error = true;
  2155. }
  2156. if (!buffer_mapped(bh)) {
  2157. folio_zero_range(folio, i * blocksize,
  2158. blocksize);
  2159. if (!err)
  2160. set_buffer_uptodate(bh);
  2161. continue;
  2162. }
  2163. /*
  2164. * get_block() might have updated the buffer
  2165. * synchronously
  2166. */
  2167. if (buffer_uptodate(bh))
  2168. continue;
  2169. }
  2170. arr[nr++] = bh;
  2171. } while (i++, iblock++, (bh = bh->b_this_page) != head);
  2172. if (fully_mapped)
  2173. folio_set_mappedtodisk(folio);
  2174. if (!nr) {
  2175. /*
  2176. * All buffers are uptodate or get_block() returned an
  2177. * error when trying to map them - we can finish the read.
  2178. */
  2179. folio_end_read(folio, !page_error);
  2180. return 0;
  2181. }
  2182. /* Stage two: lock the buffers */
  2183. for (i = 0; i < nr; i++) {
  2184. bh = arr[i];
  2185. lock_buffer(bh);
  2186. mark_buffer_async_read(bh);
  2187. }
  2188. /*
  2189. * Stage 3: start the IO. Check for uptodateness
  2190. * inside the buffer lock in case another process reading
  2191. * the underlying blockdev brought it uptodate (the sct fix).
  2192. */
  2193. for (i = 0; i < nr; i++) {
  2194. bh = arr[i];
  2195. if (buffer_uptodate(bh))
  2196. end_buffer_async_read(bh, 1);
  2197. else
  2198. submit_bh(REQ_OP_READ, bh);
  2199. }
  2200. return 0;
  2201. }
  2202. EXPORT_SYMBOL(block_read_full_folio);
  2203. /* utility function for filesystems that need to do work on expanding
  2204. * truncates. Uses filesystem pagecache writes to allow the filesystem to
  2205. * deal with the hole.
  2206. */
  2207. int generic_cont_expand_simple(struct inode *inode, loff_t size)
  2208. {
  2209. struct address_space *mapping = inode->i_mapping;
  2210. const struct address_space_operations *aops = mapping->a_ops;
  2211. struct folio *folio;
  2212. void *fsdata = NULL;
  2213. int err;
  2214. err = inode_newsize_ok(inode, size);
  2215. if (err)
  2216. goto out;
  2217. err = aops->write_begin(NULL, mapping, size, 0, &folio, &fsdata);
  2218. if (err)
  2219. goto out;
  2220. err = aops->write_end(NULL, mapping, size, 0, 0, folio, fsdata);
  2221. BUG_ON(err > 0);
  2222. out:
  2223. return err;
  2224. }
  2225. EXPORT_SYMBOL(generic_cont_expand_simple);
  2226. static int cont_expand_zero(struct file *file, struct address_space *mapping,
  2227. loff_t pos, loff_t *bytes)
  2228. {
  2229. struct inode *inode = mapping->host;
  2230. const struct address_space_operations *aops = mapping->a_ops;
  2231. unsigned int blocksize = i_blocksize(inode);
  2232. struct folio *folio;
  2233. void *fsdata = NULL;
  2234. pgoff_t index, curidx;
  2235. loff_t curpos;
  2236. unsigned zerofrom, offset, len;
  2237. int err = 0;
  2238. index = pos >> PAGE_SHIFT;
  2239. offset = pos & ~PAGE_MASK;
  2240. while (index > (curidx = (curpos = *bytes)>>PAGE_SHIFT)) {
  2241. zerofrom = curpos & ~PAGE_MASK;
  2242. if (zerofrom & (blocksize-1)) {
  2243. *bytes |= (blocksize-1);
  2244. (*bytes)++;
  2245. }
  2246. len = PAGE_SIZE - zerofrom;
  2247. err = aops->write_begin(file, mapping, curpos, len,
  2248. &folio, &fsdata);
  2249. if (err)
  2250. goto out;
  2251. folio_zero_range(folio, offset_in_folio(folio, curpos), len);
  2252. err = aops->write_end(file, mapping, curpos, len, len,
  2253. folio, fsdata);
  2254. if (err < 0)
  2255. goto out;
  2256. BUG_ON(err != len);
  2257. err = 0;
  2258. balance_dirty_pages_ratelimited(mapping);
  2259. if (fatal_signal_pending(current)) {
  2260. err = -EINTR;
  2261. goto out;
  2262. }
  2263. }
  2264. /* page covers the boundary, find the boundary offset */
  2265. if (index == curidx) {
  2266. zerofrom = curpos & ~PAGE_MASK;
  2267. /* if we will expand the thing last block will be filled */
  2268. if (offset <= zerofrom) {
  2269. goto out;
  2270. }
  2271. if (zerofrom & (blocksize-1)) {
  2272. *bytes |= (blocksize-1);
  2273. (*bytes)++;
  2274. }
  2275. len = offset - zerofrom;
  2276. err = aops->write_begin(file, mapping, curpos, len,
  2277. &folio, &fsdata);
  2278. if (err)
  2279. goto out;
  2280. folio_zero_range(folio, offset_in_folio(folio, curpos), len);
  2281. err = aops->write_end(file, mapping, curpos, len, len,
  2282. folio, fsdata);
  2283. if (err < 0)
  2284. goto out;
  2285. BUG_ON(err != len);
  2286. err = 0;
  2287. }
  2288. out:
  2289. return err;
  2290. }
  2291. /*
  2292. * For moronic filesystems that do not allow holes in file.
  2293. * We may have to extend the file.
  2294. */
  2295. int cont_write_begin(struct file *file, struct address_space *mapping,
  2296. loff_t pos, unsigned len,
  2297. struct folio **foliop, void **fsdata,
  2298. get_block_t *get_block, loff_t *bytes)
  2299. {
  2300. struct inode *inode = mapping->host;
  2301. unsigned int blocksize = i_blocksize(inode);
  2302. unsigned int zerofrom;
  2303. int err;
  2304. err = cont_expand_zero(file, mapping, pos, bytes);
  2305. if (err)
  2306. return err;
  2307. zerofrom = *bytes & ~PAGE_MASK;
  2308. if (pos+len > *bytes && zerofrom & (blocksize-1)) {
  2309. *bytes |= (blocksize-1);
  2310. (*bytes)++;
  2311. }
  2312. return block_write_begin(mapping, pos, len, foliop, get_block);
  2313. }
  2314. EXPORT_SYMBOL(cont_write_begin);
  2315. void block_commit_write(struct page *page, unsigned from, unsigned to)
  2316. {
  2317. struct folio *folio = page_folio(page);
  2318. __block_commit_write(folio, from, to);
  2319. }
  2320. EXPORT_SYMBOL(block_commit_write);
  2321. /*
  2322. * block_page_mkwrite() is not allowed to change the file size as it gets
  2323. * called from a page fault handler when a page is first dirtied. Hence we must
  2324. * be careful to check for EOF conditions here. We set the page up correctly
  2325. * for a written page which means we get ENOSPC checking when writing into
  2326. * holes and correct delalloc and unwritten extent mapping on filesystems that
  2327. * support these features.
  2328. *
  2329. * We are not allowed to take the i_mutex here so we have to play games to
  2330. * protect against truncate races as the page could now be beyond EOF. Because
  2331. * truncate writes the inode size before removing pages, once we have the
  2332. * page lock we can determine safely if the page is beyond EOF. If it is not
  2333. * beyond EOF, then the page is guaranteed safe against truncation until we
  2334. * unlock the page.
  2335. *
  2336. * Direct callers of this function should protect against filesystem freezing
  2337. * using sb_start_pagefault() - sb_end_pagefault() functions.
  2338. */
  2339. int block_page_mkwrite(struct vm_area_struct *vma, struct vm_fault *vmf,
  2340. get_block_t get_block)
  2341. {
  2342. struct folio *folio = page_folio(vmf->page);
  2343. struct inode *inode = file_inode(vma->vm_file);
  2344. unsigned long end;
  2345. loff_t size;
  2346. int ret;
  2347. folio_lock(folio);
  2348. size = i_size_read(inode);
  2349. if ((folio->mapping != inode->i_mapping) ||
  2350. (folio_pos(folio) >= size)) {
  2351. /* We overload EFAULT to mean page got truncated */
  2352. ret = -EFAULT;
  2353. goto out_unlock;
  2354. }
  2355. end = folio_size(folio);
  2356. /* folio is wholly or partially inside EOF */
  2357. if (folio_pos(folio) + end > size)
  2358. end = size - folio_pos(folio);
  2359. ret = __block_write_begin_int(folio, 0, end, get_block, NULL);
  2360. if (unlikely(ret))
  2361. goto out_unlock;
  2362. __block_commit_write(folio, 0, end);
  2363. folio_mark_dirty(folio);
  2364. folio_wait_stable(folio);
  2365. return 0;
  2366. out_unlock:
  2367. folio_unlock(folio);
  2368. return ret;
  2369. }
  2370. EXPORT_SYMBOL(block_page_mkwrite);
  2371. int block_truncate_page(struct address_space *mapping,
  2372. loff_t from, get_block_t *get_block)
  2373. {
  2374. pgoff_t index = from >> PAGE_SHIFT;
  2375. unsigned blocksize;
  2376. sector_t iblock;
  2377. size_t offset, length, pos;
  2378. struct inode *inode = mapping->host;
  2379. struct folio *folio;
  2380. struct buffer_head *bh;
  2381. int err = 0;
  2382. blocksize = i_blocksize(inode);
  2383. length = from & (blocksize - 1);
  2384. /* Block boundary? Nothing to do */
  2385. if (!length)
  2386. return 0;
  2387. length = blocksize - length;
  2388. iblock = ((loff_t)index * PAGE_SIZE) >> inode->i_blkbits;
  2389. folio = filemap_grab_folio(mapping, index);
  2390. if (IS_ERR(folio))
  2391. return PTR_ERR(folio);
  2392. bh = folio_buffers(folio);
  2393. if (!bh)
  2394. bh = create_empty_buffers(folio, blocksize, 0);
  2395. /* Find the buffer that contains "offset" */
  2396. offset = offset_in_folio(folio, from);
  2397. pos = blocksize;
  2398. while (offset >= pos) {
  2399. bh = bh->b_this_page;
  2400. iblock++;
  2401. pos += blocksize;
  2402. }
  2403. if (!buffer_mapped(bh)) {
  2404. WARN_ON(bh->b_size != blocksize);
  2405. err = get_block(inode, iblock, bh, 0);
  2406. if (err)
  2407. goto unlock;
  2408. /* unmapped? It's a hole - nothing to do */
  2409. if (!buffer_mapped(bh))
  2410. goto unlock;
  2411. }
  2412. /* Ok, it's mapped. Make sure it's up-to-date */
  2413. if (folio_test_uptodate(folio))
  2414. set_buffer_uptodate(bh);
  2415. if (!buffer_uptodate(bh) && !buffer_delay(bh) && !buffer_unwritten(bh)) {
  2416. err = bh_read(bh, 0);
  2417. /* Uhhuh. Read error. Complain and punt. */
  2418. if (err < 0)
  2419. goto unlock;
  2420. }
  2421. folio_zero_range(folio, offset, length);
  2422. mark_buffer_dirty(bh);
  2423. unlock:
  2424. folio_unlock(folio);
  2425. folio_put(folio);
  2426. return err;
  2427. }
  2428. EXPORT_SYMBOL(block_truncate_page);
  2429. /*
  2430. * The generic ->writepage function for buffer-backed address_spaces
  2431. */
  2432. int block_write_full_folio(struct folio *folio, struct writeback_control *wbc,
  2433. void *get_block)
  2434. {
  2435. struct inode * const inode = folio->mapping->host;
  2436. loff_t i_size = i_size_read(inode);
  2437. /* Is the folio fully inside i_size? */
  2438. if (folio_pos(folio) + folio_size(folio) <= i_size)
  2439. return __block_write_full_folio(inode, folio, get_block, wbc);
  2440. /* Is the folio fully outside i_size? (truncate in progress) */
  2441. if (folio_pos(folio) >= i_size) {
  2442. folio_unlock(folio);
  2443. return 0; /* don't care */
  2444. }
  2445. /*
  2446. * The folio straddles i_size. It must be zeroed out on each and every
  2447. * writepage invocation because it may be mmapped. "A file is mapped
  2448. * in multiples of the page size. For a file that is not a multiple of
  2449. * the page size, the remaining memory is zeroed when mapped, and
  2450. * writes to that region are not written out to the file."
  2451. */
  2452. folio_zero_segment(folio, offset_in_folio(folio, i_size),
  2453. folio_size(folio));
  2454. return __block_write_full_folio(inode, folio, get_block, wbc);
  2455. }
  2456. sector_t generic_block_bmap(struct address_space *mapping, sector_t block,
  2457. get_block_t *get_block)
  2458. {
  2459. struct inode *inode = mapping->host;
  2460. struct buffer_head tmp = {
  2461. .b_size = i_blocksize(inode),
  2462. };
  2463. get_block(inode, block, &tmp, 0);
  2464. return tmp.b_blocknr;
  2465. }
  2466. EXPORT_SYMBOL(generic_block_bmap);
  2467. static void end_bio_bh_io_sync(struct bio *bio)
  2468. {
  2469. struct buffer_head *bh = bio->bi_private;
  2470. if (unlikely(bio_flagged(bio, BIO_QUIET)))
  2471. set_bit(BH_Quiet, &bh->b_state);
  2472. bh->b_end_io(bh, !bio->bi_status);
  2473. bio_put(bio);
  2474. }
  2475. static void submit_bh_wbc(blk_opf_t opf, struct buffer_head *bh,
  2476. enum rw_hint write_hint,
  2477. struct writeback_control *wbc)
  2478. {
  2479. const enum req_op op = opf & REQ_OP_MASK;
  2480. struct bio *bio;
  2481. BUG_ON(!buffer_locked(bh));
  2482. BUG_ON(!buffer_mapped(bh));
  2483. BUG_ON(!bh->b_end_io);
  2484. BUG_ON(buffer_delay(bh));
  2485. BUG_ON(buffer_unwritten(bh));
  2486. /*
  2487. * Only clear out a write error when rewriting
  2488. */
  2489. if (test_set_buffer_req(bh) && (op == REQ_OP_WRITE))
  2490. clear_buffer_write_io_error(bh);
  2491. if (buffer_meta(bh))
  2492. opf |= REQ_META;
  2493. if (buffer_prio(bh))
  2494. opf |= REQ_PRIO;
  2495. bio = bio_alloc(bh->b_bdev, 1, opf, GFP_NOIO);
  2496. fscrypt_set_bio_crypt_ctx_bh(bio, bh, GFP_NOIO);
  2497. bio->bi_iter.bi_sector = bh->b_blocknr * (bh->b_size >> 9);
  2498. bio->bi_write_hint = write_hint;
  2499. bio_add_folio_nofail(bio, bh->b_folio, bh->b_size, bh_offset(bh));
  2500. bio->bi_end_io = end_bio_bh_io_sync;
  2501. bio->bi_private = bh;
  2502. /* Take care of bh's that straddle the end of the device */
  2503. guard_bio_eod(bio);
  2504. if (wbc) {
  2505. wbc_init_bio(wbc, bio);
  2506. wbc_account_cgroup_owner(wbc, bh->b_folio, bh->b_size);
  2507. }
  2508. submit_bio(bio);
  2509. }
  2510. void submit_bh(blk_opf_t opf, struct buffer_head *bh)
  2511. {
  2512. submit_bh_wbc(opf, bh, WRITE_LIFE_NOT_SET, NULL);
  2513. }
  2514. EXPORT_SYMBOL(submit_bh);
  2515. void write_dirty_buffer(struct buffer_head *bh, blk_opf_t op_flags)
  2516. {
  2517. lock_buffer(bh);
  2518. if (!test_clear_buffer_dirty(bh)) {
  2519. unlock_buffer(bh);
  2520. return;
  2521. }
  2522. bh->b_end_io = end_buffer_write_sync;
  2523. get_bh(bh);
  2524. submit_bh(REQ_OP_WRITE | op_flags, bh);
  2525. }
  2526. EXPORT_SYMBOL(write_dirty_buffer);
  2527. /*
  2528. * For a data-integrity writeout, we need to wait upon any in-progress I/O
  2529. * and then start new I/O and then wait upon it. The caller must have a ref on
  2530. * the buffer_head.
  2531. */
  2532. int __sync_dirty_buffer(struct buffer_head *bh, blk_opf_t op_flags)
  2533. {
  2534. WARN_ON(atomic_read(&bh->b_count) < 1);
  2535. lock_buffer(bh);
  2536. if (test_clear_buffer_dirty(bh)) {
  2537. /*
  2538. * The bh should be mapped, but it might not be if the
  2539. * device was hot-removed. Not much we can do but fail the I/O.
  2540. */
  2541. if (!buffer_mapped(bh)) {
  2542. unlock_buffer(bh);
  2543. return -EIO;
  2544. }
  2545. get_bh(bh);
  2546. bh->b_end_io = end_buffer_write_sync;
  2547. submit_bh(REQ_OP_WRITE | op_flags, bh);
  2548. wait_on_buffer(bh);
  2549. if (!buffer_uptodate(bh))
  2550. return -EIO;
  2551. } else {
  2552. unlock_buffer(bh);
  2553. }
  2554. return 0;
  2555. }
  2556. EXPORT_SYMBOL(__sync_dirty_buffer);
  2557. int sync_dirty_buffer(struct buffer_head *bh)
  2558. {
  2559. return __sync_dirty_buffer(bh, REQ_SYNC);
  2560. }
  2561. EXPORT_SYMBOL(sync_dirty_buffer);
  2562. static inline int buffer_busy(struct buffer_head *bh)
  2563. {
  2564. return atomic_read(&bh->b_count) |
  2565. (bh->b_state & ((1 << BH_Dirty) | (1 << BH_Lock)));
  2566. }
  2567. static bool
  2568. drop_buffers(struct folio *folio, struct buffer_head **buffers_to_free)
  2569. {
  2570. struct buffer_head *head = folio_buffers(folio);
  2571. struct buffer_head *bh;
  2572. bh = head;
  2573. do {
  2574. if (buffer_busy(bh))
  2575. goto failed;
  2576. bh = bh->b_this_page;
  2577. } while (bh != head);
  2578. do {
  2579. struct buffer_head *next = bh->b_this_page;
  2580. if (bh->b_assoc_map)
  2581. __remove_assoc_queue(bh);
  2582. bh = next;
  2583. } while (bh != head);
  2584. *buffers_to_free = head;
  2585. folio_detach_private(folio);
  2586. return true;
  2587. failed:
  2588. return false;
  2589. }
  2590. /**
  2591. * try_to_free_buffers - Release buffers attached to this folio.
  2592. * @folio: The folio.
  2593. *
  2594. * If any buffers are in use (dirty, under writeback, elevated refcount),
  2595. * no buffers will be freed.
  2596. *
  2597. * If the folio is dirty but all the buffers are clean then we need to
  2598. * be sure to mark the folio clean as well. This is because the folio
  2599. * may be against a block device, and a later reattachment of buffers
  2600. * to a dirty folio will set *all* buffers dirty. Which would corrupt
  2601. * filesystem data on the same device.
  2602. *
  2603. * The same applies to regular filesystem folios: if all the buffers are
  2604. * clean then we set the folio clean and proceed. To do that, we require
  2605. * total exclusion from block_dirty_folio(). That is obtained with
  2606. * i_private_lock.
  2607. *
  2608. * Exclusion against try_to_free_buffers may be obtained by either
  2609. * locking the folio or by holding its mapping's i_private_lock.
  2610. *
  2611. * Context: Process context. @folio must be locked. Will not sleep.
  2612. * Return: true if all buffers attached to this folio were freed.
  2613. */
  2614. bool try_to_free_buffers(struct folio *folio)
  2615. {
  2616. struct address_space * const mapping = folio->mapping;
  2617. struct buffer_head *buffers_to_free = NULL;
  2618. bool ret = 0;
  2619. BUG_ON(!folio_test_locked(folio));
  2620. if (folio_test_writeback(folio))
  2621. return false;
  2622. if (mapping == NULL) { /* can this still happen? */
  2623. ret = drop_buffers(folio, &buffers_to_free);
  2624. goto out;
  2625. }
  2626. spin_lock(&mapping->i_private_lock);
  2627. ret = drop_buffers(folio, &buffers_to_free);
  2628. /*
  2629. * If the filesystem writes its buffers by hand (eg ext3)
  2630. * then we can have clean buffers against a dirty folio. We
  2631. * clean the folio here; otherwise the VM will never notice
  2632. * that the filesystem did any IO at all.
  2633. *
  2634. * Also, during truncate, discard_buffer will have marked all
  2635. * the folio's buffers clean. We discover that here and clean
  2636. * the folio also.
  2637. *
  2638. * i_private_lock must be held over this entire operation in order
  2639. * to synchronise against block_dirty_folio and prevent the
  2640. * dirty bit from being lost.
  2641. */
  2642. if (ret)
  2643. folio_cancel_dirty(folio);
  2644. spin_unlock(&mapping->i_private_lock);
  2645. out:
  2646. if (buffers_to_free) {
  2647. struct buffer_head *bh = buffers_to_free;
  2648. do {
  2649. struct buffer_head *next = bh->b_this_page;
  2650. free_buffer_head(bh);
  2651. bh = next;
  2652. } while (bh != buffers_to_free);
  2653. }
  2654. return ret;
  2655. }
  2656. EXPORT_SYMBOL(try_to_free_buffers);
  2657. /*
  2658. * Buffer-head allocation
  2659. */
  2660. static struct kmem_cache *bh_cachep __ro_after_init;
  2661. /*
  2662. * Once the number of bh's in the machine exceeds this level, we start
  2663. * stripping them in writeback.
  2664. */
  2665. static unsigned long max_buffer_heads __ro_after_init;
  2666. int buffer_heads_over_limit;
  2667. struct bh_accounting {
  2668. int nr; /* Number of live bh's */
  2669. int ratelimit; /* Limit cacheline bouncing */
  2670. };
  2671. static DEFINE_PER_CPU(struct bh_accounting, bh_accounting) = {0, 0};
  2672. static void recalc_bh_state(void)
  2673. {
  2674. int i;
  2675. int tot = 0;
  2676. if (__this_cpu_inc_return(bh_accounting.ratelimit) - 1 < 4096)
  2677. return;
  2678. __this_cpu_write(bh_accounting.ratelimit, 0);
  2679. for_each_online_cpu(i)
  2680. tot += per_cpu(bh_accounting, i).nr;
  2681. buffer_heads_over_limit = (tot > max_buffer_heads);
  2682. }
  2683. struct buffer_head *alloc_buffer_head(gfp_t gfp_flags)
  2684. {
  2685. struct buffer_head *ret = kmem_cache_zalloc(bh_cachep, gfp_flags);
  2686. if (ret) {
  2687. INIT_LIST_HEAD(&ret->b_assoc_buffers);
  2688. spin_lock_init(&ret->b_uptodate_lock);
  2689. preempt_disable();
  2690. __this_cpu_inc(bh_accounting.nr);
  2691. recalc_bh_state();
  2692. preempt_enable();
  2693. }
  2694. return ret;
  2695. }
  2696. EXPORT_SYMBOL(alloc_buffer_head);
  2697. void free_buffer_head(struct buffer_head *bh)
  2698. {
  2699. BUG_ON(!list_empty(&bh->b_assoc_buffers));
  2700. kmem_cache_free(bh_cachep, bh);
  2701. preempt_disable();
  2702. __this_cpu_dec(bh_accounting.nr);
  2703. recalc_bh_state();
  2704. preempt_enable();
  2705. }
  2706. EXPORT_SYMBOL(free_buffer_head);
  2707. static int buffer_exit_cpu_dead(unsigned int cpu)
  2708. {
  2709. int i;
  2710. struct bh_lru *b = &per_cpu(bh_lrus, cpu);
  2711. for (i = 0; i < BH_LRU_SIZE; i++) {
  2712. brelse(b->bhs[i]);
  2713. b->bhs[i] = NULL;
  2714. }
  2715. this_cpu_add(bh_accounting.nr, per_cpu(bh_accounting, cpu).nr);
  2716. per_cpu(bh_accounting, cpu).nr = 0;
  2717. return 0;
  2718. }
  2719. /**
  2720. * bh_uptodate_or_lock - Test whether the buffer is uptodate
  2721. * @bh: struct buffer_head
  2722. *
  2723. * Return true if the buffer is up-to-date and false,
  2724. * with the buffer locked, if not.
  2725. */
  2726. int bh_uptodate_or_lock(struct buffer_head *bh)
  2727. {
  2728. if (!buffer_uptodate(bh)) {
  2729. lock_buffer(bh);
  2730. if (!buffer_uptodate(bh))
  2731. return 0;
  2732. unlock_buffer(bh);
  2733. }
  2734. return 1;
  2735. }
  2736. EXPORT_SYMBOL(bh_uptodate_or_lock);
  2737. /**
  2738. * __bh_read - Submit read for a locked buffer
  2739. * @bh: struct buffer_head
  2740. * @op_flags: appending REQ_OP_* flags besides REQ_OP_READ
  2741. * @wait: wait until reading finish
  2742. *
  2743. * Returns zero on success or don't wait, and -EIO on error.
  2744. */
  2745. int __bh_read(struct buffer_head *bh, blk_opf_t op_flags, bool wait)
  2746. {
  2747. int ret = 0;
  2748. BUG_ON(!buffer_locked(bh));
  2749. get_bh(bh);
  2750. bh->b_end_io = end_buffer_read_sync;
  2751. submit_bh(REQ_OP_READ | op_flags, bh);
  2752. if (wait) {
  2753. wait_on_buffer(bh);
  2754. if (!buffer_uptodate(bh))
  2755. ret = -EIO;
  2756. }
  2757. return ret;
  2758. }
  2759. EXPORT_SYMBOL(__bh_read);
  2760. /**
  2761. * __bh_read_batch - Submit read for a batch of unlocked buffers
  2762. * @nr: entry number of the buffer batch
  2763. * @bhs: a batch of struct buffer_head
  2764. * @op_flags: appending REQ_OP_* flags besides REQ_OP_READ
  2765. * @force_lock: force to get a lock on the buffer if set, otherwise drops any
  2766. * buffer that cannot lock.
  2767. *
  2768. * Returns zero on success or don't wait, and -EIO on error.
  2769. */
  2770. void __bh_read_batch(int nr, struct buffer_head *bhs[],
  2771. blk_opf_t op_flags, bool force_lock)
  2772. {
  2773. int i;
  2774. for (i = 0; i < nr; i++) {
  2775. struct buffer_head *bh = bhs[i];
  2776. if (buffer_uptodate(bh))
  2777. continue;
  2778. if (force_lock)
  2779. lock_buffer(bh);
  2780. else
  2781. if (!trylock_buffer(bh))
  2782. continue;
  2783. if (buffer_uptodate(bh)) {
  2784. unlock_buffer(bh);
  2785. continue;
  2786. }
  2787. bh->b_end_io = end_buffer_read_sync;
  2788. get_bh(bh);
  2789. submit_bh(REQ_OP_READ | op_flags, bh);
  2790. }
  2791. }
  2792. EXPORT_SYMBOL(__bh_read_batch);
  2793. void __init buffer_init(void)
  2794. {
  2795. unsigned long nrpages;
  2796. int ret;
  2797. bh_cachep = KMEM_CACHE(buffer_head,
  2798. SLAB_RECLAIM_ACCOUNT|SLAB_PANIC);
  2799. /*
  2800. * Limit the bh occupancy to 10% of ZONE_NORMAL
  2801. */
  2802. nrpages = (nr_free_buffer_pages() * 10) / 100;
  2803. max_buffer_heads = nrpages * (PAGE_SIZE / sizeof(struct buffer_head));
  2804. ret = cpuhp_setup_state_nocalls(CPUHP_FS_BUFF_DEAD, "fs/buffer:dead",
  2805. NULL, buffer_exit_cpu_dead);
  2806. WARN_ON(ret < 0);
  2807. }