skbuff.c 182 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402240324042405240624072408240924102411241224132414241524162417241824192420242124222423242424252426242724282429243024312432243324342435243624372438243924402441244224432444244524462447244824492450245124522453245424552456245724582459246024612462246324642465246624672468246924702471247224732474247524762477247824792480248124822483248424852486248724882489249024912492249324942495249624972498249925002501250225032504250525062507250825092510251125122513251425152516251725182519252025212522252325242525252625272528252925302531253225332534253525362537253825392540254125422543254425452546254725482549255025512552255325542555255625572558255925602561256225632564256525662567256825692570257125722573257425752576257725782579258025812582258325842585258625872588258925902591259225932594259525962597259825992600260126022603260426052606260726082609261026112612261326142615261626172618261926202621262226232624262526262627262826292630263126322633263426352636263726382639264026412642264326442645264626472648264926502651265226532654265526562657265826592660266126622663266426652666266726682669267026712672267326742675267626772678267926802681268226832684268526862687268826892690269126922693269426952696269726982699270027012702270327042705270627072708270927102711271227132714271527162717271827192720272127222723272427252726272727282729273027312732273327342735273627372738273927402741274227432744274527462747274827492750275127522753275427552756275727582759276027612762276327642765276627672768276927702771277227732774277527762777277827792780278127822783278427852786278727882789279027912792279327942795279627972798279928002801280228032804280528062807280828092810281128122813281428152816281728182819282028212822282328242825282628272828282928302831283228332834283528362837283828392840284128422843284428452846284728482849285028512852285328542855285628572858285928602861286228632864286528662867286828692870287128722873287428752876287728782879288028812882288328842885288628872888288928902891289228932894289528962897289828992900290129022903290429052906290729082909291029112912291329142915291629172918291929202921292229232924292529262927292829292930293129322933293429352936293729382939294029412942294329442945294629472948294929502951295229532954295529562957295829592960296129622963296429652966296729682969297029712972297329742975297629772978297929802981298229832984298529862987298829892990299129922993299429952996299729982999300030013002300330043005300630073008300930103011301230133014301530163017301830193020302130223023302430253026302730283029303030313032303330343035303630373038303930403041304230433044304530463047304830493050305130523053305430553056305730583059306030613062306330643065306630673068306930703071307230733074307530763077307830793080308130823083308430853086308730883089309030913092309330943095309630973098309931003101310231033104310531063107310831093110311131123113311431153116311731183119312031213122312331243125312631273128312931303131313231333134313531363137313831393140314131423143314431453146314731483149315031513152315331543155315631573158315931603161316231633164316531663167316831693170317131723173317431753176317731783179318031813182318331843185318631873188318931903191319231933194319531963197319831993200320132023203320432053206320732083209321032113212321332143215321632173218321932203221322232233224322532263227322832293230323132323233323432353236323732383239324032413242324332443245324632473248324932503251325232533254325532563257325832593260326132623263326432653266326732683269327032713272327332743275327632773278327932803281328232833284328532863287328832893290329132923293329432953296329732983299330033013302330333043305330633073308330933103311331233133314331533163317331833193320332133223323332433253326332733283329333033313332333333343335333633373338333933403341334233433344334533463347334833493350335133523353335433553356335733583359336033613362336333643365336633673368336933703371337233733374337533763377337833793380338133823383338433853386338733883389339033913392339333943395339633973398339934003401340234033404340534063407340834093410341134123413341434153416341734183419342034213422342334243425342634273428342934303431343234333434343534363437343834393440344134423443344434453446344734483449345034513452345334543455345634573458345934603461346234633464346534663467346834693470347134723473347434753476347734783479348034813482348334843485348634873488348934903491349234933494349534963497349834993500350135023503350435053506350735083509351035113512351335143515351635173518351935203521352235233524352535263527352835293530353135323533353435353536353735383539354035413542354335443545354635473548354935503551355235533554355535563557355835593560356135623563356435653566356735683569357035713572357335743575357635773578357935803581358235833584358535863587358835893590359135923593359435953596359735983599360036013602360336043605360636073608360936103611361236133614361536163617361836193620362136223623362436253626362736283629363036313632363336343635363636373638363936403641364236433644364536463647364836493650365136523653365436553656365736583659366036613662366336643665366636673668366936703671367236733674367536763677367836793680368136823683368436853686368736883689369036913692369336943695369636973698369937003701370237033704370537063707370837093710371137123713371437153716371737183719372037213722372337243725372637273728372937303731373237333734373537363737373837393740374137423743374437453746374737483749375037513752375337543755375637573758375937603761376237633764376537663767376837693770377137723773377437753776377737783779378037813782378337843785378637873788378937903791379237933794379537963797379837993800380138023803380438053806380738083809381038113812381338143815381638173818381938203821382238233824382538263827382838293830383138323833383438353836383738383839384038413842384338443845384638473848384938503851385238533854385538563857385838593860386138623863386438653866386738683869387038713872387338743875387638773878387938803881388238833884388538863887388838893890389138923893389438953896389738983899390039013902390339043905390639073908390939103911391239133914391539163917391839193920392139223923392439253926392739283929393039313932393339343935393639373938393939403941394239433944394539463947394839493950395139523953395439553956395739583959396039613962396339643965396639673968396939703971397239733974397539763977397839793980398139823983398439853986398739883989399039913992399339943995399639973998399940004001400240034004400540064007400840094010401140124013401440154016401740184019402040214022402340244025402640274028402940304031403240334034403540364037403840394040404140424043404440454046404740484049405040514052405340544055405640574058405940604061406240634064406540664067406840694070407140724073407440754076407740784079408040814082408340844085408640874088408940904091409240934094409540964097409840994100410141024103410441054106410741084109411041114112411341144115411641174118411941204121412241234124412541264127412841294130413141324133413441354136413741384139414041414142414341444145414641474148414941504151415241534154415541564157415841594160416141624163416441654166416741684169417041714172417341744175417641774178417941804181418241834184418541864187418841894190419141924193419441954196419741984199420042014202420342044205420642074208420942104211421242134214421542164217421842194220422142224223422442254226422742284229423042314232423342344235423642374238423942404241424242434244424542464247424842494250425142524253425442554256425742584259426042614262426342644265426642674268426942704271427242734274427542764277427842794280428142824283428442854286428742884289429042914292429342944295429642974298429943004301430243034304430543064307430843094310431143124313431443154316431743184319432043214322432343244325432643274328432943304331433243334334433543364337433843394340434143424343434443454346434743484349435043514352435343544355435643574358435943604361436243634364436543664367436843694370437143724373437443754376437743784379438043814382438343844385438643874388438943904391439243934394439543964397439843994400440144024403440444054406440744084409441044114412441344144415441644174418441944204421442244234424442544264427442844294430443144324433443444354436443744384439444044414442444344444445444644474448444944504451445244534454445544564457445844594460446144624463446444654466446744684469447044714472447344744475447644774478447944804481448244834484448544864487448844894490449144924493449444954496449744984499450045014502450345044505450645074508450945104511451245134514451545164517451845194520452145224523452445254526452745284529453045314532453345344535453645374538453945404541454245434544454545464547454845494550455145524553455445554556455745584559456045614562456345644565456645674568456945704571457245734574457545764577457845794580458145824583458445854586458745884589459045914592459345944595459645974598459946004601460246034604460546064607460846094610461146124613461446154616461746184619462046214622462346244625462646274628462946304631463246334634463546364637463846394640464146424643464446454646464746484649465046514652465346544655465646574658465946604661466246634664466546664667466846694670467146724673467446754676467746784679468046814682468346844685468646874688468946904691469246934694469546964697469846994700470147024703470447054706470747084709471047114712471347144715471647174718471947204721472247234724472547264727472847294730473147324733473447354736473747384739474047414742474347444745474647474748474947504751475247534754475547564757475847594760476147624763476447654766476747684769477047714772477347744775477647774778477947804781478247834784478547864787478847894790479147924793479447954796479747984799480048014802480348044805480648074808480948104811481248134814481548164817481848194820482148224823482448254826482748284829483048314832483348344835483648374838483948404841484248434844484548464847484848494850485148524853485448554856485748584859486048614862486348644865486648674868486948704871487248734874487548764877487848794880488148824883488448854886488748884889489048914892489348944895489648974898489949004901490249034904490549064907490849094910491149124913491449154916491749184919492049214922492349244925492649274928492949304931493249334934493549364937493849394940494149424943494449454946494749484949495049514952495349544955495649574958495949604961496249634964496549664967496849694970497149724973497449754976497749784979498049814982498349844985498649874988498949904991499249934994499549964997499849995000500150025003500450055006500750085009501050115012501350145015501650175018501950205021502250235024502550265027502850295030503150325033503450355036503750385039504050415042504350445045504650475048504950505051505250535054505550565057505850595060506150625063506450655066506750685069507050715072507350745075507650775078507950805081508250835084508550865087508850895090509150925093509450955096509750985099510051015102510351045105510651075108510951105111511251135114511551165117511851195120512151225123512451255126512751285129513051315132513351345135513651375138513951405141514251435144514551465147514851495150515151525153515451555156515751585159516051615162516351645165516651675168516951705171517251735174517551765177517851795180518151825183518451855186518751885189519051915192519351945195519651975198519952005201520252035204520552065207520852095210521152125213521452155216521752185219522052215222522352245225522652275228522952305231523252335234523552365237523852395240524152425243524452455246524752485249525052515252525352545255525652575258525952605261526252635264526552665267526852695270527152725273527452755276527752785279528052815282528352845285528652875288528952905291529252935294529552965297529852995300530153025303530453055306530753085309531053115312531353145315531653175318531953205321532253235324532553265327532853295330533153325333533453355336533753385339534053415342534353445345534653475348534953505351535253535354535553565357535853595360536153625363536453655366536753685369537053715372537353745375537653775378537953805381538253835384538553865387538853895390539153925393539453955396539753985399540054015402540354045405540654075408540954105411541254135414541554165417541854195420542154225423542454255426542754285429543054315432543354345435543654375438543954405441544254435444544554465447544854495450545154525453545454555456545754585459546054615462546354645465546654675468546954705471547254735474547554765477547854795480548154825483548454855486548754885489549054915492549354945495549654975498549955005501550255035504550555065507550855095510551155125513551455155516551755185519552055215522552355245525552655275528552955305531553255335534553555365537553855395540554155425543554455455546554755485549555055515552555355545555555655575558555955605561556255635564556555665567556855695570557155725573557455755576557755785579558055815582558355845585558655875588558955905591559255935594559555965597559855995600560156025603560456055606560756085609561056115612561356145615561656175618561956205621562256235624562556265627562856295630563156325633563456355636563756385639564056415642564356445645564656475648564956505651565256535654565556565657565856595660566156625663566456655666566756685669567056715672567356745675567656775678567956805681568256835684568556865687568856895690569156925693569456955696569756985699570057015702570357045705570657075708570957105711571257135714571557165717571857195720572157225723572457255726572757285729573057315732573357345735573657375738573957405741574257435744574557465747574857495750575157525753575457555756575757585759576057615762576357645765576657675768576957705771577257735774577557765777577857795780578157825783578457855786578757885789579057915792579357945795579657975798579958005801580258035804580558065807580858095810581158125813581458155816581758185819582058215822582358245825582658275828582958305831583258335834583558365837583858395840584158425843584458455846584758485849585058515852585358545855585658575858585958605861586258635864586558665867586858695870587158725873587458755876587758785879588058815882588358845885588658875888588958905891589258935894589558965897589858995900590159025903590459055906590759085909591059115912591359145915591659175918591959205921592259235924592559265927592859295930593159325933593459355936593759385939594059415942594359445945594659475948594959505951595259535954595559565957595859595960596159625963596459655966596759685969597059715972597359745975597659775978597959805981598259835984598559865987598859895990599159925993599459955996599759985999600060016002600360046005600660076008600960106011601260136014601560166017601860196020602160226023602460256026602760286029603060316032603360346035603660376038603960406041604260436044604560466047604860496050605160526053605460556056605760586059606060616062606360646065606660676068606960706071607260736074607560766077607860796080608160826083608460856086608760886089609060916092609360946095609660976098609961006101610261036104610561066107610861096110611161126113611461156116611761186119612061216122612361246125612661276128612961306131613261336134613561366137613861396140614161426143614461456146614761486149615061516152615361546155615661576158615961606161616261636164616561666167616861696170617161726173617461756176617761786179618061816182618361846185618661876188618961906191619261936194619561966197619861996200620162026203620462056206620762086209621062116212621362146215621662176218621962206221622262236224622562266227622862296230623162326233623462356236623762386239624062416242624362446245624662476248624962506251625262536254625562566257625862596260626162626263626462656266626762686269627062716272627362746275627662776278627962806281628262836284628562866287628862896290629162926293629462956296629762986299630063016302630363046305630663076308630963106311631263136314631563166317631863196320632163226323632463256326632763286329633063316332633363346335633663376338633963406341634263436344634563466347634863496350635163526353635463556356635763586359636063616362636363646365636663676368636963706371637263736374637563766377637863796380638163826383638463856386638763886389639063916392639363946395639663976398639964006401640264036404640564066407640864096410641164126413641464156416641764186419642064216422642364246425642664276428642964306431643264336434643564366437643864396440644164426443644464456446644764486449645064516452645364546455645664576458645964606461646264636464646564666467646864696470647164726473647464756476647764786479648064816482648364846485648664876488648964906491649264936494649564966497649864996500650165026503650465056506650765086509651065116512651365146515651665176518651965206521652265236524652565266527652865296530653165326533653465356536653765386539654065416542654365446545654665476548654965506551655265536554655565566557655865596560656165626563656465656566656765686569657065716572657365746575657665776578657965806581658265836584658565866587658865896590659165926593659465956596659765986599660066016602660366046605660666076608660966106611661266136614661566166617661866196620662166226623662466256626662766286629663066316632663366346635663666376638663966406641664266436644664566466647664866496650665166526653665466556656665766586659666066616662666366646665666666676668666966706671667266736674667566766677667866796680668166826683668466856686668766886689669066916692669366946695669666976698669967006701670267036704670567066707670867096710671167126713671467156716671767186719672067216722672367246725672667276728672967306731673267336734673567366737673867396740674167426743674467456746674767486749675067516752675367546755675667576758675967606761676267636764676567666767676867696770677167726773677467756776677767786779678067816782678367846785678667876788678967906791679267936794679567966797679867996800680168026803680468056806680768086809681068116812681368146815681668176818681968206821682268236824682568266827682868296830683168326833683468356836683768386839684068416842684368446845684668476848684968506851685268536854685568566857685868596860686168626863686468656866686768686869687068716872687368746875687668776878687968806881688268836884688568866887688868896890689168926893689468956896689768986899690069016902690369046905690669076908690969106911691269136914691569166917691869196920692169226923692469256926692769286929693069316932693369346935693669376938693969406941694269436944694569466947694869496950695169526953695469556956695769586959696069616962696369646965696669676968696969706971697269736974697569766977697869796980698169826983698469856986698769886989699069916992699369946995699669976998699970007001700270037004700570067007700870097010701170127013701470157016701770187019702070217022702370247025702670277028702970307031703270337034703570367037703870397040704170427043704470457046704770487049705070517052705370547055705670577058705970607061706270637064706570667067706870697070707170727073707470757076707770787079708070817082708370847085708670877088708970907091709270937094709570967097709870997100710171027103710471057106710771087109711071117112711371147115711671177118711971207121712271237124712571267127712871297130713171327133713471357136713771387139714071417142714371447145714671477148714971507151715271537154715571567157715871597160716171627163716471657166716771687169717071717172717371747175717671777178717971807181718271837184718571867187718871897190719171927193719471957196719771987199720072017202720372047205720672077208720972107211721272137214721572167217721872197220722172227223722472257226722772287229723072317232723372347235723672377238723972407241724272437244724572467247724872497250725172527253725472557256725772587259726072617262726372647265726672677268726972707271727272737274727572767277727872797280728172827283728472857286728772887289729072917292
  1. // SPDX-License-Identifier: GPL-2.0-or-later
  2. /*
  3. * Routines having to do with the 'struct sk_buff' memory handlers.
  4. *
  5. * Authors: Alan Cox <alan@lxorguk.ukuu.org.uk>
  6. * Florian La Roche <rzsfl@rz.uni-sb.de>
  7. *
  8. * Fixes:
  9. * Alan Cox : Fixed the worst of the load
  10. * balancer bugs.
  11. * Dave Platt : Interrupt stacking fix.
  12. * Richard Kooijman : Timestamp fixes.
  13. * Alan Cox : Changed buffer format.
  14. * Alan Cox : destructor hook for AF_UNIX etc.
  15. * Linus Torvalds : Better skb_clone.
  16. * Alan Cox : Added skb_copy.
  17. * Alan Cox : Added all the changed routines Linus
  18. * only put in the headers
  19. * Ray VanTassle : Fixed --skb->lock in free
  20. * Alan Cox : skb_copy copy arp field
  21. * Andi Kleen : slabified it.
  22. * Robert Olsson : Removed skb_head_pool
  23. *
  24. * NOTE:
  25. * The __skb_ routines should be called with interrupts
  26. * disabled, or you better be *real* sure that the operation is atomic
  27. * with respect to whatever list is being frobbed (e.g. via lock_sock()
  28. * or via disabling bottom half handlers, etc).
  29. */
  30. /*
  31. * The functions in this file will not compile correctly with gcc 2.4.x
  32. */
  33. #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
  34. #include <linux/module.h>
  35. #include <linux/types.h>
  36. #include <linux/kernel.h>
  37. #include <linux/mm.h>
  38. #include <linux/interrupt.h>
  39. #include <linux/in.h>
  40. #include <linux/inet.h>
  41. #include <linux/slab.h>
  42. #include <linux/tcp.h>
  43. #include <linux/udp.h>
  44. #include <linux/sctp.h>
  45. #include <linux/netdevice.h>
  46. #ifdef CONFIG_NET_CLS_ACT
  47. #include <net/pkt_sched.h>
  48. #endif
  49. #include <linux/string.h>
  50. #include <linux/skbuff.h>
  51. #include <linux/skbuff_ref.h>
  52. #include <linux/splice.h>
  53. #include <linux/cache.h>
  54. #include <linux/rtnetlink.h>
  55. #include <linux/init.h>
  56. #include <linux/scatterlist.h>
  57. #include <linux/errqueue.h>
  58. #include <linux/prefetch.h>
  59. #include <linux/bitfield.h>
  60. #include <linux/if_vlan.h>
  61. #include <linux/mpls.h>
  62. #include <linux/kcov.h>
  63. #include <linux/iov_iter.h>
  64. #include <net/protocol.h>
  65. #include <net/dst.h>
  66. #include <net/sock.h>
  67. #include <net/checksum.h>
  68. #include <net/gso.h>
  69. #include <net/hotdata.h>
  70. #include <net/ip6_checksum.h>
  71. #include <net/xfrm.h>
  72. #include <net/mpls.h>
  73. #include <net/mptcp.h>
  74. #include <net/mctp.h>
  75. #include <net/page_pool/helpers.h>
  76. #include <net/dropreason.h>
  77. #include <linux/uaccess.h>
  78. #include <trace/events/skb.h>
  79. #include <linux/highmem.h>
  80. #include <linux/capability.h>
  81. #include <linux/user_namespace.h>
  82. #include <linux/indirect_call_wrapper.h>
  83. #include <linux/textsearch.h>
  84. #include "dev.h"
  85. #include "netmem_priv.h"
  86. #include "sock_destructor.h"
  87. #ifdef CONFIG_SKB_EXTENSIONS
  88. static struct kmem_cache *skbuff_ext_cache __ro_after_init;
  89. #endif
  90. #define SKB_SMALL_HEAD_SIZE SKB_HEAD_ALIGN(MAX_TCP_HEADER)
  91. /* We want SKB_SMALL_HEAD_CACHE_SIZE to not be a power of two.
  92. * This should ensure that SKB_SMALL_HEAD_HEADROOM is a unique
  93. * size, and we can differentiate heads from skb_small_head_cache
  94. * vs system slabs by looking at their size (skb_end_offset()).
  95. */
  96. #define SKB_SMALL_HEAD_CACHE_SIZE \
  97. (is_power_of_2(SKB_SMALL_HEAD_SIZE) ? \
  98. (SKB_SMALL_HEAD_SIZE + L1_CACHE_BYTES) : \
  99. SKB_SMALL_HEAD_SIZE)
  100. #define SKB_SMALL_HEAD_HEADROOM \
  101. SKB_WITH_OVERHEAD(SKB_SMALL_HEAD_CACHE_SIZE)
  102. /* kcm_write_msgs() relies on casting paged frags to bio_vec to use
  103. * iov_iter_bvec(). These static asserts ensure the cast is valid is long as the
  104. * netmem is a page.
  105. */
  106. static_assert(offsetof(struct bio_vec, bv_page) ==
  107. offsetof(skb_frag_t, netmem));
  108. static_assert(sizeof_field(struct bio_vec, bv_page) ==
  109. sizeof_field(skb_frag_t, netmem));
  110. static_assert(offsetof(struct bio_vec, bv_len) == offsetof(skb_frag_t, len));
  111. static_assert(sizeof_field(struct bio_vec, bv_len) ==
  112. sizeof_field(skb_frag_t, len));
  113. static_assert(offsetof(struct bio_vec, bv_offset) ==
  114. offsetof(skb_frag_t, offset));
  115. static_assert(sizeof_field(struct bio_vec, bv_offset) ==
  116. sizeof_field(skb_frag_t, offset));
  117. #undef FN
  118. #define FN(reason) [SKB_DROP_REASON_##reason] = #reason,
  119. static const char * const drop_reasons[] = {
  120. [SKB_CONSUMED] = "CONSUMED",
  121. DEFINE_DROP_REASON(FN, FN)
  122. };
  123. static const struct drop_reason_list drop_reasons_core = {
  124. .reasons = drop_reasons,
  125. .n_reasons = ARRAY_SIZE(drop_reasons),
  126. };
  127. const struct drop_reason_list __rcu *
  128. drop_reasons_by_subsys[SKB_DROP_REASON_SUBSYS_NUM] = {
  129. [SKB_DROP_REASON_SUBSYS_CORE] = RCU_INITIALIZER(&drop_reasons_core),
  130. };
  131. EXPORT_SYMBOL(drop_reasons_by_subsys);
  132. /**
  133. * drop_reasons_register_subsys - register another drop reason subsystem
  134. * @subsys: the subsystem to register, must not be the core
  135. * @list: the list of drop reasons within the subsystem, must point to
  136. * a statically initialized list
  137. */
  138. void drop_reasons_register_subsys(enum skb_drop_reason_subsys subsys,
  139. const struct drop_reason_list *list)
  140. {
  141. if (WARN(subsys <= SKB_DROP_REASON_SUBSYS_CORE ||
  142. subsys >= ARRAY_SIZE(drop_reasons_by_subsys),
  143. "invalid subsystem %d\n", subsys))
  144. return;
  145. /* must point to statically allocated memory, so INIT is OK */
  146. RCU_INIT_POINTER(drop_reasons_by_subsys[subsys], list);
  147. }
  148. EXPORT_SYMBOL_GPL(drop_reasons_register_subsys);
  149. /**
  150. * drop_reasons_unregister_subsys - unregister a drop reason subsystem
  151. * @subsys: the subsystem to remove, must not be the core
  152. *
  153. * Note: This will synchronize_rcu() to ensure no users when it returns.
  154. */
  155. void drop_reasons_unregister_subsys(enum skb_drop_reason_subsys subsys)
  156. {
  157. if (WARN(subsys <= SKB_DROP_REASON_SUBSYS_CORE ||
  158. subsys >= ARRAY_SIZE(drop_reasons_by_subsys),
  159. "invalid subsystem %d\n", subsys))
  160. return;
  161. RCU_INIT_POINTER(drop_reasons_by_subsys[subsys], NULL);
  162. synchronize_rcu();
  163. }
  164. EXPORT_SYMBOL_GPL(drop_reasons_unregister_subsys);
  165. /**
  166. * skb_panic - private function for out-of-line support
  167. * @skb: buffer
  168. * @sz: size
  169. * @addr: address
  170. * @msg: skb_over_panic or skb_under_panic
  171. *
  172. * Out-of-line support for skb_put() and skb_push().
  173. * Called via the wrapper skb_over_panic() or skb_under_panic().
  174. * Keep out of line to prevent kernel bloat.
  175. * __builtin_return_address is not used because it is not always reliable.
  176. */
  177. static void skb_panic(struct sk_buff *skb, unsigned int sz, void *addr,
  178. const char msg[])
  179. {
  180. pr_emerg("%s: text:%px len:%d put:%d head:%px data:%px tail:%#lx end:%#lx dev:%s\n",
  181. msg, addr, skb->len, sz, skb->head, skb->data,
  182. (unsigned long)skb->tail, (unsigned long)skb->end,
  183. skb->dev ? skb->dev->name : "<NULL>");
  184. BUG();
  185. }
  186. static void skb_over_panic(struct sk_buff *skb, unsigned int sz, void *addr)
  187. {
  188. skb_panic(skb, sz, addr, __func__);
  189. }
  190. static void skb_under_panic(struct sk_buff *skb, unsigned int sz, void *addr)
  191. {
  192. skb_panic(skb, sz, addr, __func__);
  193. }
  194. #define NAPI_SKB_CACHE_SIZE 64
  195. #define NAPI_SKB_CACHE_BULK 16
  196. #define NAPI_SKB_CACHE_HALF (NAPI_SKB_CACHE_SIZE / 2)
  197. #if PAGE_SIZE == SZ_4K
  198. #define NAPI_HAS_SMALL_PAGE_FRAG 1
  199. #define NAPI_SMALL_PAGE_PFMEMALLOC(nc) ((nc).pfmemalloc)
  200. /* specialized page frag allocator using a single order 0 page
  201. * and slicing it into 1K sized fragment. Constrained to systems
  202. * with a very limited amount of 1K fragments fitting a single
  203. * page - to avoid excessive truesize underestimation
  204. */
  205. struct page_frag_1k {
  206. void *va;
  207. u16 offset;
  208. bool pfmemalloc;
  209. };
  210. static void *page_frag_alloc_1k(struct page_frag_1k *nc, gfp_t gfp)
  211. {
  212. struct page *page;
  213. int offset;
  214. offset = nc->offset - SZ_1K;
  215. if (likely(offset >= 0))
  216. goto use_frag;
  217. page = alloc_pages_node(NUMA_NO_NODE, gfp, 0);
  218. if (!page)
  219. return NULL;
  220. nc->va = page_address(page);
  221. nc->pfmemalloc = page_is_pfmemalloc(page);
  222. offset = PAGE_SIZE - SZ_1K;
  223. page_ref_add(page, offset / SZ_1K);
  224. use_frag:
  225. nc->offset = offset;
  226. return nc->va + offset;
  227. }
  228. #else
  229. /* the small page is actually unused in this build; add dummy helpers
  230. * to please the compiler and avoid later preprocessor's conditionals
  231. */
  232. #define NAPI_HAS_SMALL_PAGE_FRAG 0
  233. #define NAPI_SMALL_PAGE_PFMEMALLOC(nc) false
  234. struct page_frag_1k {
  235. };
  236. static void *page_frag_alloc_1k(struct page_frag_1k *nc, gfp_t gfp_mask)
  237. {
  238. return NULL;
  239. }
  240. #endif
  241. struct napi_alloc_cache {
  242. local_lock_t bh_lock;
  243. struct page_frag_cache page;
  244. struct page_frag_1k page_small;
  245. unsigned int skb_count;
  246. void *skb_cache[NAPI_SKB_CACHE_SIZE];
  247. };
  248. static DEFINE_PER_CPU(struct page_frag_cache, netdev_alloc_cache);
  249. static DEFINE_PER_CPU(struct napi_alloc_cache, napi_alloc_cache) = {
  250. .bh_lock = INIT_LOCAL_LOCK(bh_lock),
  251. };
  252. /* Double check that napi_get_frags() allocates skbs with
  253. * skb->head being backed by slab, not a page fragment.
  254. * This is to make sure bug fixed in 3226b158e67c
  255. * ("net: avoid 32 x truesize under-estimation for tiny skbs")
  256. * does not accidentally come back.
  257. */
  258. void napi_get_frags_check(struct napi_struct *napi)
  259. {
  260. struct sk_buff *skb;
  261. local_bh_disable();
  262. skb = napi_get_frags(napi);
  263. WARN_ON_ONCE(!NAPI_HAS_SMALL_PAGE_FRAG && skb && skb->head_frag);
  264. napi_free_frags(napi);
  265. local_bh_enable();
  266. }
  267. void *__napi_alloc_frag_align(unsigned int fragsz, unsigned int align_mask)
  268. {
  269. struct napi_alloc_cache *nc = this_cpu_ptr(&napi_alloc_cache);
  270. void *data;
  271. fragsz = SKB_DATA_ALIGN(fragsz);
  272. local_lock_nested_bh(&napi_alloc_cache.bh_lock);
  273. data = __page_frag_alloc_align(&nc->page, fragsz,
  274. GFP_ATOMIC | __GFP_NOWARN, align_mask);
  275. local_unlock_nested_bh(&napi_alloc_cache.bh_lock);
  276. return data;
  277. }
  278. EXPORT_SYMBOL(__napi_alloc_frag_align);
  279. void *__netdev_alloc_frag_align(unsigned int fragsz, unsigned int align_mask)
  280. {
  281. void *data;
  282. if (in_hardirq() || irqs_disabled()) {
  283. struct page_frag_cache *nc = this_cpu_ptr(&netdev_alloc_cache);
  284. fragsz = SKB_DATA_ALIGN(fragsz);
  285. data = __page_frag_alloc_align(nc, fragsz,
  286. GFP_ATOMIC | __GFP_NOWARN,
  287. align_mask);
  288. } else {
  289. local_bh_disable();
  290. data = __napi_alloc_frag_align(fragsz, align_mask);
  291. local_bh_enable();
  292. }
  293. return data;
  294. }
  295. EXPORT_SYMBOL(__netdev_alloc_frag_align);
  296. static struct sk_buff *napi_skb_cache_get(void)
  297. {
  298. struct napi_alloc_cache *nc = this_cpu_ptr(&napi_alloc_cache);
  299. struct sk_buff *skb;
  300. local_lock_nested_bh(&napi_alloc_cache.bh_lock);
  301. if (unlikely(!nc->skb_count)) {
  302. nc->skb_count = kmem_cache_alloc_bulk(net_hotdata.skbuff_cache,
  303. GFP_ATOMIC | __GFP_NOWARN,
  304. NAPI_SKB_CACHE_BULK,
  305. nc->skb_cache);
  306. if (unlikely(!nc->skb_count)) {
  307. local_unlock_nested_bh(&napi_alloc_cache.bh_lock);
  308. return NULL;
  309. }
  310. }
  311. skb = nc->skb_cache[--nc->skb_count];
  312. local_unlock_nested_bh(&napi_alloc_cache.bh_lock);
  313. kasan_mempool_unpoison_object(skb, kmem_cache_size(net_hotdata.skbuff_cache));
  314. return skb;
  315. }
  316. static inline void __finalize_skb_around(struct sk_buff *skb, void *data,
  317. unsigned int size)
  318. {
  319. struct skb_shared_info *shinfo;
  320. size -= SKB_DATA_ALIGN(sizeof(struct skb_shared_info));
  321. /* Assumes caller memset cleared SKB */
  322. skb->truesize = SKB_TRUESIZE(size);
  323. refcount_set(&skb->users, 1);
  324. skb->head = data;
  325. skb->data = data;
  326. skb_reset_tail_pointer(skb);
  327. skb_set_end_offset(skb, size);
  328. skb->mac_header = (typeof(skb->mac_header))~0U;
  329. skb->transport_header = (typeof(skb->transport_header))~0U;
  330. skb->alloc_cpu = raw_smp_processor_id();
  331. /* make sure we initialize shinfo sequentially */
  332. shinfo = skb_shinfo(skb);
  333. memset(shinfo, 0, offsetof(struct skb_shared_info, dataref));
  334. atomic_set(&shinfo->dataref, 1);
  335. skb_set_kcov_handle(skb, kcov_common_handle());
  336. }
  337. static inline void *__slab_build_skb(struct sk_buff *skb, void *data,
  338. unsigned int *size)
  339. {
  340. void *resized;
  341. /* Must find the allocation size (and grow it to match). */
  342. *size = ksize(data);
  343. /* krealloc() will immediately return "data" when
  344. * "ksize(data)" is requested: it is the existing upper
  345. * bounds. As a result, GFP_ATOMIC will be ignored. Note
  346. * that this "new" pointer needs to be passed back to the
  347. * caller for use so the __alloc_size hinting will be
  348. * tracked correctly.
  349. */
  350. resized = krealloc(data, *size, GFP_ATOMIC);
  351. WARN_ON_ONCE(resized != data);
  352. return resized;
  353. }
  354. /* build_skb() variant which can operate on slab buffers.
  355. * Note that this should be used sparingly as slab buffers
  356. * cannot be combined efficiently by GRO!
  357. */
  358. struct sk_buff *slab_build_skb(void *data)
  359. {
  360. struct sk_buff *skb;
  361. unsigned int size;
  362. skb = kmem_cache_alloc(net_hotdata.skbuff_cache,
  363. GFP_ATOMIC | __GFP_NOWARN);
  364. if (unlikely(!skb))
  365. return NULL;
  366. memset(skb, 0, offsetof(struct sk_buff, tail));
  367. data = __slab_build_skb(skb, data, &size);
  368. __finalize_skb_around(skb, data, size);
  369. return skb;
  370. }
  371. EXPORT_SYMBOL(slab_build_skb);
  372. /* Caller must provide SKB that is memset cleared */
  373. static void __build_skb_around(struct sk_buff *skb, void *data,
  374. unsigned int frag_size)
  375. {
  376. unsigned int size = frag_size;
  377. /* frag_size == 0 is considered deprecated now. Callers
  378. * using slab buffer should use slab_build_skb() instead.
  379. */
  380. if (WARN_ONCE(size == 0, "Use slab_build_skb() instead"))
  381. data = __slab_build_skb(skb, data, &size);
  382. __finalize_skb_around(skb, data, size);
  383. }
  384. /**
  385. * __build_skb - build a network buffer
  386. * @data: data buffer provided by caller
  387. * @frag_size: size of data (must not be 0)
  388. *
  389. * Allocate a new &sk_buff. Caller provides space holding head and
  390. * skb_shared_info. @data must have been allocated from the page
  391. * allocator or vmalloc(). (A @frag_size of 0 to indicate a kmalloc()
  392. * allocation is deprecated, and callers should use slab_build_skb()
  393. * instead.)
  394. * The return is the new skb buffer.
  395. * On a failure the return is %NULL, and @data is not freed.
  396. * Notes :
  397. * Before IO, driver allocates only data buffer where NIC put incoming frame
  398. * Driver should add room at head (NET_SKB_PAD) and
  399. * MUST add room at tail (SKB_DATA_ALIGN(skb_shared_info))
  400. * After IO, driver calls build_skb(), to allocate sk_buff and populate it
  401. * before giving packet to stack.
  402. * RX rings only contains data buffers, not full skbs.
  403. */
  404. struct sk_buff *__build_skb(void *data, unsigned int frag_size)
  405. {
  406. struct sk_buff *skb;
  407. skb = kmem_cache_alloc(net_hotdata.skbuff_cache,
  408. GFP_ATOMIC | __GFP_NOWARN);
  409. if (unlikely(!skb))
  410. return NULL;
  411. memset(skb, 0, offsetof(struct sk_buff, tail));
  412. __build_skb_around(skb, data, frag_size);
  413. return skb;
  414. }
  415. /* build_skb() is wrapper over __build_skb(), that specifically
  416. * takes care of skb->head and skb->pfmemalloc
  417. */
  418. struct sk_buff *build_skb(void *data, unsigned int frag_size)
  419. {
  420. struct sk_buff *skb = __build_skb(data, frag_size);
  421. if (likely(skb && frag_size)) {
  422. skb->head_frag = 1;
  423. skb_propagate_pfmemalloc(virt_to_head_page(data), skb);
  424. }
  425. return skb;
  426. }
  427. EXPORT_SYMBOL(build_skb);
  428. /**
  429. * build_skb_around - build a network buffer around provided skb
  430. * @skb: sk_buff provide by caller, must be memset cleared
  431. * @data: data buffer provided by caller
  432. * @frag_size: size of data
  433. */
  434. struct sk_buff *build_skb_around(struct sk_buff *skb,
  435. void *data, unsigned int frag_size)
  436. {
  437. if (unlikely(!skb))
  438. return NULL;
  439. __build_skb_around(skb, data, frag_size);
  440. if (frag_size) {
  441. skb->head_frag = 1;
  442. skb_propagate_pfmemalloc(virt_to_head_page(data), skb);
  443. }
  444. return skb;
  445. }
  446. EXPORT_SYMBOL(build_skb_around);
  447. /**
  448. * __napi_build_skb - build a network buffer
  449. * @data: data buffer provided by caller
  450. * @frag_size: size of data
  451. *
  452. * Version of __build_skb() that uses NAPI percpu caches to obtain
  453. * skbuff_head instead of inplace allocation.
  454. *
  455. * Returns a new &sk_buff on success, %NULL on allocation failure.
  456. */
  457. static struct sk_buff *__napi_build_skb(void *data, unsigned int frag_size)
  458. {
  459. struct sk_buff *skb;
  460. skb = napi_skb_cache_get();
  461. if (unlikely(!skb))
  462. return NULL;
  463. memset(skb, 0, offsetof(struct sk_buff, tail));
  464. __build_skb_around(skb, data, frag_size);
  465. return skb;
  466. }
  467. /**
  468. * napi_build_skb - build a network buffer
  469. * @data: data buffer provided by caller
  470. * @frag_size: size of data
  471. *
  472. * Version of __napi_build_skb() that takes care of skb->head_frag
  473. * and skb->pfmemalloc when the data is a page or page fragment.
  474. *
  475. * Returns a new &sk_buff on success, %NULL on allocation failure.
  476. */
  477. struct sk_buff *napi_build_skb(void *data, unsigned int frag_size)
  478. {
  479. struct sk_buff *skb = __napi_build_skb(data, frag_size);
  480. if (likely(skb) && frag_size) {
  481. skb->head_frag = 1;
  482. skb_propagate_pfmemalloc(virt_to_head_page(data), skb);
  483. }
  484. return skb;
  485. }
  486. EXPORT_SYMBOL(napi_build_skb);
  487. /*
  488. * kmalloc_reserve is a wrapper around kmalloc_node_track_caller that tells
  489. * the caller if emergency pfmemalloc reserves are being used. If it is and
  490. * the socket is later found to be SOCK_MEMALLOC then PFMEMALLOC reserves
  491. * may be used. Otherwise, the packet data may be discarded until enough
  492. * memory is free
  493. */
  494. static void *kmalloc_reserve(unsigned int *size, gfp_t flags, int node,
  495. bool *pfmemalloc)
  496. {
  497. bool ret_pfmemalloc = false;
  498. size_t obj_size;
  499. void *obj;
  500. obj_size = SKB_HEAD_ALIGN(*size);
  501. if (obj_size <= SKB_SMALL_HEAD_CACHE_SIZE &&
  502. !(flags & KMALLOC_NOT_NORMAL_BITS)) {
  503. obj = kmem_cache_alloc_node(net_hotdata.skb_small_head_cache,
  504. flags | __GFP_NOMEMALLOC | __GFP_NOWARN,
  505. node);
  506. *size = SKB_SMALL_HEAD_CACHE_SIZE;
  507. if (obj || !(gfp_pfmemalloc_allowed(flags)))
  508. goto out;
  509. /* Try again but now we are using pfmemalloc reserves */
  510. ret_pfmemalloc = true;
  511. obj = kmem_cache_alloc_node(net_hotdata.skb_small_head_cache, flags, node);
  512. goto out;
  513. }
  514. obj_size = kmalloc_size_roundup(obj_size);
  515. /* The following cast might truncate high-order bits of obj_size, this
  516. * is harmless because kmalloc(obj_size >= 2^32) will fail anyway.
  517. */
  518. *size = (unsigned int)obj_size;
  519. /*
  520. * Try a regular allocation, when that fails and we're not entitled
  521. * to the reserves, fail.
  522. */
  523. obj = kmalloc_node_track_caller(obj_size,
  524. flags | __GFP_NOMEMALLOC | __GFP_NOWARN,
  525. node);
  526. if (obj || !(gfp_pfmemalloc_allowed(flags)))
  527. goto out;
  528. /* Try again but now we are using pfmemalloc reserves */
  529. ret_pfmemalloc = true;
  530. obj = kmalloc_node_track_caller(obj_size, flags, node);
  531. out:
  532. if (pfmemalloc)
  533. *pfmemalloc = ret_pfmemalloc;
  534. return obj;
  535. }
  536. /* Allocate a new skbuff. We do this ourselves so we can fill in a few
  537. * 'private' fields and also do memory statistics to find all the
  538. * [BEEP] leaks.
  539. *
  540. */
  541. /**
  542. * __alloc_skb - allocate a network buffer
  543. * @size: size to allocate
  544. * @gfp_mask: allocation mask
  545. * @flags: If SKB_ALLOC_FCLONE is set, allocate from fclone cache
  546. * instead of head cache and allocate a cloned (child) skb.
  547. * If SKB_ALLOC_RX is set, __GFP_MEMALLOC will be used for
  548. * allocations in case the data is required for writeback
  549. * @node: numa node to allocate memory on
  550. *
  551. * Allocate a new &sk_buff. The returned buffer has no headroom and a
  552. * tail room of at least size bytes. The object has a reference count
  553. * of one. The return is the buffer. On a failure the return is %NULL.
  554. *
  555. * Buffers may only be allocated from interrupts using a @gfp_mask of
  556. * %GFP_ATOMIC.
  557. */
  558. struct sk_buff *__alloc_skb(unsigned int size, gfp_t gfp_mask,
  559. int flags, int node)
  560. {
  561. struct kmem_cache *cache;
  562. struct sk_buff *skb;
  563. bool pfmemalloc;
  564. u8 *data;
  565. cache = (flags & SKB_ALLOC_FCLONE)
  566. ? net_hotdata.skbuff_fclone_cache : net_hotdata.skbuff_cache;
  567. if (sk_memalloc_socks() && (flags & SKB_ALLOC_RX))
  568. gfp_mask |= __GFP_MEMALLOC;
  569. /* Get the HEAD */
  570. if ((flags & (SKB_ALLOC_FCLONE | SKB_ALLOC_NAPI)) == SKB_ALLOC_NAPI &&
  571. likely(node == NUMA_NO_NODE || node == numa_mem_id()))
  572. skb = napi_skb_cache_get();
  573. else
  574. skb = kmem_cache_alloc_node(cache, gfp_mask & ~GFP_DMA, node);
  575. if (unlikely(!skb))
  576. return NULL;
  577. prefetchw(skb);
  578. /* We do our best to align skb_shared_info on a separate cache
  579. * line. It usually works because kmalloc(X > SMP_CACHE_BYTES) gives
  580. * aligned memory blocks, unless SLUB/SLAB debug is enabled.
  581. * Both skb->head and skb_shared_info are cache line aligned.
  582. */
  583. data = kmalloc_reserve(&size, gfp_mask, node, &pfmemalloc);
  584. if (unlikely(!data))
  585. goto nodata;
  586. /* kmalloc_size_roundup() might give us more room than requested.
  587. * Put skb_shared_info exactly at the end of allocated zone,
  588. * to allow max possible filling before reallocation.
  589. */
  590. prefetchw(data + SKB_WITH_OVERHEAD(size));
  591. /*
  592. * Only clear those fields we need to clear, not those that we will
  593. * actually initialise below. Hence, don't put any more fields after
  594. * the tail pointer in struct sk_buff!
  595. */
  596. memset(skb, 0, offsetof(struct sk_buff, tail));
  597. __build_skb_around(skb, data, size);
  598. skb->pfmemalloc = pfmemalloc;
  599. if (flags & SKB_ALLOC_FCLONE) {
  600. struct sk_buff_fclones *fclones;
  601. fclones = container_of(skb, struct sk_buff_fclones, skb1);
  602. skb->fclone = SKB_FCLONE_ORIG;
  603. refcount_set(&fclones->fclone_ref, 1);
  604. }
  605. return skb;
  606. nodata:
  607. kmem_cache_free(cache, skb);
  608. return NULL;
  609. }
  610. EXPORT_SYMBOL(__alloc_skb);
  611. /**
  612. * __netdev_alloc_skb - allocate an skbuff for rx on a specific device
  613. * @dev: network device to receive on
  614. * @len: length to allocate
  615. * @gfp_mask: get_free_pages mask, passed to alloc_skb
  616. *
  617. * Allocate a new &sk_buff and assign it a usage count of one. The
  618. * buffer has NET_SKB_PAD headroom built in. Users should allocate
  619. * the headroom they think they need without accounting for the
  620. * built in space. The built in space is used for optimisations.
  621. *
  622. * %NULL is returned if there is no free memory.
  623. */
  624. struct sk_buff *__netdev_alloc_skb(struct net_device *dev, unsigned int len,
  625. gfp_t gfp_mask)
  626. {
  627. struct page_frag_cache *nc;
  628. struct sk_buff *skb;
  629. bool pfmemalloc;
  630. void *data;
  631. len += NET_SKB_PAD;
  632. /* If requested length is either too small or too big,
  633. * we use kmalloc() for skb->head allocation.
  634. */
  635. if (len <= SKB_WITH_OVERHEAD(1024) ||
  636. len > SKB_WITH_OVERHEAD(PAGE_SIZE) ||
  637. (gfp_mask & (__GFP_DIRECT_RECLAIM | GFP_DMA))) {
  638. skb = __alloc_skb(len, gfp_mask, SKB_ALLOC_RX, NUMA_NO_NODE);
  639. if (!skb)
  640. goto skb_fail;
  641. goto skb_success;
  642. }
  643. len = SKB_HEAD_ALIGN(len);
  644. if (sk_memalloc_socks())
  645. gfp_mask |= __GFP_MEMALLOC;
  646. if (in_hardirq() || irqs_disabled()) {
  647. nc = this_cpu_ptr(&netdev_alloc_cache);
  648. data = page_frag_alloc(nc, len, gfp_mask);
  649. pfmemalloc = nc->pfmemalloc;
  650. } else {
  651. local_bh_disable();
  652. local_lock_nested_bh(&napi_alloc_cache.bh_lock);
  653. nc = this_cpu_ptr(&napi_alloc_cache.page);
  654. data = page_frag_alloc(nc, len, gfp_mask);
  655. pfmemalloc = nc->pfmemalloc;
  656. local_unlock_nested_bh(&napi_alloc_cache.bh_lock);
  657. local_bh_enable();
  658. }
  659. if (unlikely(!data))
  660. return NULL;
  661. skb = __build_skb(data, len);
  662. if (unlikely(!skb)) {
  663. skb_free_frag(data);
  664. return NULL;
  665. }
  666. if (pfmemalloc)
  667. skb->pfmemalloc = 1;
  668. skb->head_frag = 1;
  669. skb_success:
  670. skb_reserve(skb, NET_SKB_PAD);
  671. skb->dev = dev;
  672. skb_fail:
  673. return skb;
  674. }
  675. EXPORT_SYMBOL(__netdev_alloc_skb);
  676. /**
  677. * napi_alloc_skb - allocate skbuff for rx in a specific NAPI instance
  678. * @napi: napi instance this buffer was allocated for
  679. * @len: length to allocate
  680. *
  681. * Allocate a new sk_buff for use in NAPI receive. This buffer will
  682. * attempt to allocate the head from a special reserved region used
  683. * only for NAPI Rx allocation. By doing this we can save several
  684. * CPU cycles by avoiding having to disable and re-enable IRQs.
  685. *
  686. * %NULL is returned if there is no free memory.
  687. */
  688. struct sk_buff *napi_alloc_skb(struct napi_struct *napi, unsigned int len)
  689. {
  690. gfp_t gfp_mask = GFP_ATOMIC | __GFP_NOWARN;
  691. struct napi_alloc_cache *nc;
  692. struct sk_buff *skb;
  693. bool pfmemalloc;
  694. void *data;
  695. DEBUG_NET_WARN_ON_ONCE(!in_softirq());
  696. len += NET_SKB_PAD + NET_IP_ALIGN;
  697. /* If requested length is either too small or too big,
  698. * we use kmalloc() for skb->head allocation.
  699. * When the small frag allocator is available, prefer it over kmalloc
  700. * for small fragments
  701. */
  702. if ((!NAPI_HAS_SMALL_PAGE_FRAG && len <= SKB_WITH_OVERHEAD(1024)) ||
  703. len > SKB_WITH_OVERHEAD(PAGE_SIZE) ||
  704. (gfp_mask & (__GFP_DIRECT_RECLAIM | GFP_DMA))) {
  705. skb = __alloc_skb(len, gfp_mask, SKB_ALLOC_RX | SKB_ALLOC_NAPI,
  706. NUMA_NO_NODE);
  707. if (!skb)
  708. goto skb_fail;
  709. goto skb_success;
  710. }
  711. if (sk_memalloc_socks())
  712. gfp_mask |= __GFP_MEMALLOC;
  713. local_lock_nested_bh(&napi_alloc_cache.bh_lock);
  714. nc = this_cpu_ptr(&napi_alloc_cache);
  715. if (NAPI_HAS_SMALL_PAGE_FRAG && len <= SKB_WITH_OVERHEAD(1024)) {
  716. /* we are artificially inflating the allocation size, but
  717. * that is not as bad as it may look like, as:
  718. * - 'len' less than GRO_MAX_HEAD makes little sense
  719. * - On most systems, larger 'len' values lead to fragment
  720. * size above 512 bytes
  721. * - kmalloc would use the kmalloc-1k slab for such values
  722. * - Builds with smaller GRO_MAX_HEAD will very likely do
  723. * little networking, as that implies no WiFi and no
  724. * tunnels support, and 32 bits arches.
  725. */
  726. len = SZ_1K;
  727. data = page_frag_alloc_1k(&nc->page_small, gfp_mask);
  728. pfmemalloc = NAPI_SMALL_PAGE_PFMEMALLOC(nc->page_small);
  729. } else {
  730. len = SKB_HEAD_ALIGN(len);
  731. data = page_frag_alloc(&nc->page, len, gfp_mask);
  732. pfmemalloc = nc->page.pfmemalloc;
  733. }
  734. local_unlock_nested_bh(&napi_alloc_cache.bh_lock);
  735. if (unlikely(!data))
  736. return NULL;
  737. skb = __napi_build_skb(data, len);
  738. if (unlikely(!skb)) {
  739. skb_free_frag(data);
  740. return NULL;
  741. }
  742. if (pfmemalloc)
  743. skb->pfmemalloc = 1;
  744. skb->head_frag = 1;
  745. skb_success:
  746. skb_reserve(skb, NET_SKB_PAD + NET_IP_ALIGN);
  747. skb->dev = napi->dev;
  748. skb_fail:
  749. return skb;
  750. }
  751. EXPORT_SYMBOL(napi_alloc_skb);
  752. void skb_add_rx_frag_netmem(struct sk_buff *skb, int i, netmem_ref netmem,
  753. int off, int size, unsigned int truesize)
  754. {
  755. DEBUG_NET_WARN_ON_ONCE(size > truesize);
  756. skb_fill_netmem_desc(skb, i, netmem, off, size);
  757. skb->len += size;
  758. skb->data_len += size;
  759. skb->truesize += truesize;
  760. }
  761. EXPORT_SYMBOL(skb_add_rx_frag_netmem);
  762. void skb_coalesce_rx_frag(struct sk_buff *skb, int i, int size,
  763. unsigned int truesize)
  764. {
  765. skb_frag_t *frag = &skb_shinfo(skb)->frags[i];
  766. DEBUG_NET_WARN_ON_ONCE(size > truesize);
  767. skb_frag_size_add(frag, size);
  768. skb->len += size;
  769. skb->data_len += size;
  770. skb->truesize += truesize;
  771. }
  772. EXPORT_SYMBOL(skb_coalesce_rx_frag);
  773. static void skb_drop_list(struct sk_buff **listp)
  774. {
  775. kfree_skb_list(*listp);
  776. *listp = NULL;
  777. }
  778. static inline void skb_drop_fraglist(struct sk_buff *skb)
  779. {
  780. skb_drop_list(&skb_shinfo(skb)->frag_list);
  781. }
  782. static void skb_clone_fraglist(struct sk_buff *skb)
  783. {
  784. struct sk_buff *list;
  785. skb_walk_frags(skb, list)
  786. skb_get(list);
  787. }
  788. static bool is_pp_netmem(netmem_ref netmem)
  789. {
  790. return (netmem_get_pp_magic(netmem) & ~0x3UL) == PP_SIGNATURE;
  791. }
  792. int skb_pp_cow_data(struct page_pool *pool, struct sk_buff **pskb,
  793. unsigned int headroom)
  794. {
  795. #if IS_ENABLED(CONFIG_PAGE_POOL)
  796. u32 size, truesize, len, max_head_size, off;
  797. struct sk_buff *skb = *pskb, *nskb;
  798. int err, i, head_off;
  799. void *data;
  800. /* XDP does not support fraglist so we need to linearize
  801. * the skb.
  802. */
  803. if (skb_has_frag_list(skb))
  804. return -EOPNOTSUPP;
  805. max_head_size = SKB_WITH_OVERHEAD(PAGE_SIZE - headroom);
  806. if (skb->len > max_head_size + MAX_SKB_FRAGS * PAGE_SIZE)
  807. return -ENOMEM;
  808. size = min_t(u32, skb->len, max_head_size);
  809. truesize = SKB_HEAD_ALIGN(size) + headroom;
  810. data = page_pool_dev_alloc_va(pool, &truesize);
  811. if (!data)
  812. return -ENOMEM;
  813. nskb = napi_build_skb(data, truesize);
  814. if (!nskb) {
  815. page_pool_free_va(pool, data, true);
  816. return -ENOMEM;
  817. }
  818. skb_reserve(nskb, headroom);
  819. skb_copy_header(nskb, skb);
  820. skb_mark_for_recycle(nskb);
  821. err = skb_copy_bits(skb, 0, nskb->data, size);
  822. if (err) {
  823. consume_skb(nskb);
  824. return err;
  825. }
  826. skb_put(nskb, size);
  827. head_off = skb_headroom(nskb) - skb_headroom(skb);
  828. skb_headers_offset_update(nskb, head_off);
  829. off = size;
  830. len = skb->len - off;
  831. for (i = 0; i < MAX_SKB_FRAGS && off < skb->len; i++) {
  832. struct page *page;
  833. u32 page_off;
  834. size = min_t(u32, len, PAGE_SIZE);
  835. truesize = size;
  836. page = page_pool_dev_alloc(pool, &page_off, &truesize);
  837. if (!page) {
  838. consume_skb(nskb);
  839. return -ENOMEM;
  840. }
  841. skb_add_rx_frag(nskb, i, page, page_off, size, truesize);
  842. err = skb_copy_bits(skb, off, page_address(page) + page_off,
  843. size);
  844. if (err) {
  845. consume_skb(nskb);
  846. return err;
  847. }
  848. len -= size;
  849. off += size;
  850. }
  851. consume_skb(skb);
  852. *pskb = nskb;
  853. return 0;
  854. #else
  855. return -EOPNOTSUPP;
  856. #endif
  857. }
  858. EXPORT_SYMBOL(skb_pp_cow_data);
  859. int skb_cow_data_for_xdp(struct page_pool *pool, struct sk_buff **pskb,
  860. struct bpf_prog *prog)
  861. {
  862. if (!prog->aux->xdp_has_frags)
  863. return -EINVAL;
  864. return skb_pp_cow_data(pool, pskb, XDP_PACKET_HEADROOM);
  865. }
  866. EXPORT_SYMBOL(skb_cow_data_for_xdp);
  867. #if IS_ENABLED(CONFIG_PAGE_POOL)
  868. bool napi_pp_put_page(netmem_ref netmem)
  869. {
  870. netmem = netmem_compound_head(netmem);
  871. /* page->pp_magic is OR'ed with PP_SIGNATURE after the allocation
  872. * in order to preserve any existing bits, such as bit 0 for the
  873. * head page of compound page and bit 1 for pfmemalloc page, so
  874. * mask those bits for freeing side when doing below checking,
  875. * and page_is_pfmemalloc() is checked in __page_pool_put_page()
  876. * to avoid recycling the pfmemalloc page.
  877. */
  878. if (unlikely(!is_pp_netmem(netmem)))
  879. return false;
  880. page_pool_put_full_netmem(netmem_get_pp(netmem), netmem, false);
  881. return true;
  882. }
  883. EXPORT_SYMBOL(napi_pp_put_page);
  884. #endif
  885. static bool skb_pp_recycle(struct sk_buff *skb, void *data)
  886. {
  887. if (!IS_ENABLED(CONFIG_PAGE_POOL) || !skb->pp_recycle)
  888. return false;
  889. return napi_pp_put_page(page_to_netmem(virt_to_page(data)));
  890. }
  891. /**
  892. * skb_pp_frag_ref() - Increase fragment references of a page pool aware skb
  893. * @skb: page pool aware skb
  894. *
  895. * Increase the fragment reference count (pp_ref_count) of a skb. This is
  896. * intended to gain fragment references only for page pool aware skbs,
  897. * i.e. when skb->pp_recycle is true, and not for fragments in a
  898. * non-pp-recycling skb. It has a fallback to increase references on normal
  899. * pages, as page pool aware skbs may also have normal page fragments.
  900. */
  901. static int skb_pp_frag_ref(struct sk_buff *skb)
  902. {
  903. struct skb_shared_info *shinfo;
  904. netmem_ref head_netmem;
  905. int i;
  906. if (!skb->pp_recycle)
  907. return -EINVAL;
  908. shinfo = skb_shinfo(skb);
  909. for (i = 0; i < shinfo->nr_frags; i++) {
  910. head_netmem = netmem_compound_head(shinfo->frags[i].netmem);
  911. if (likely(is_pp_netmem(head_netmem)))
  912. page_pool_ref_netmem(head_netmem);
  913. else
  914. page_ref_inc(netmem_to_page(head_netmem));
  915. }
  916. return 0;
  917. }
  918. static void skb_kfree_head(void *head, unsigned int end_offset)
  919. {
  920. if (end_offset == SKB_SMALL_HEAD_HEADROOM)
  921. kmem_cache_free(net_hotdata.skb_small_head_cache, head);
  922. else
  923. kfree(head);
  924. }
  925. static void skb_free_head(struct sk_buff *skb)
  926. {
  927. unsigned char *head = skb->head;
  928. if (skb->head_frag) {
  929. if (skb_pp_recycle(skb, head))
  930. return;
  931. skb_free_frag(head);
  932. } else {
  933. skb_kfree_head(head, skb_end_offset(skb));
  934. }
  935. }
  936. static void skb_release_data(struct sk_buff *skb, enum skb_drop_reason reason)
  937. {
  938. struct skb_shared_info *shinfo = skb_shinfo(skb);
  939. int i;
  940. if (!skb_data_unref(skb, shinfo))
  941. goto exit;
  942. if (skb_zcopy(skb)) {
  943. bool skip_unref = shinfo->flags & SKBFL_MANAGED_FRAG_REFS;
  944. skb_zcopy_clear(skb, true);
  945. if (skip_unref)
  946. goto free_head;
  947. }
  948. for (i = 0; i < shinfo->nr_frags; i++)
  949. __skb_frag_unref(&shinfo->frags[i], skb->pp_recycle);
  950. free_head:
  951. if (shinfo->frag_list)
  952. kfree_skb_list_reason(shinfo->frag_list, reason);
  953. skb_free_head(skb);
  954. exit:
  955. /* When we clone an SKB we copy the reycling bit. The pp_recycle
  956. * bit is only set on the head though, so in order to avoid races
  957. * while trying to recycle fragments on __skb_frag_unref() we need
  958. * to make one SKB responsible for triggering the recycle path.
  959. * So disable the recycling bit if an SKB is cloned and we have
  960. * additional references to the fragmented part of the SKB.
  961. * Eventually the last SKB will have the recycling bit set and it's
  962. * dataref set to 0, which will trigger the recycling
  963. */
  964. skb->pp_recycle = 0;
  965. }
  966. /*
  967. * Free an skbuff by memory without cleaning the state.
  968. */
  969. static void kfree_skbmem(struct sk_buff *skb)
  970. {
  971. struct sk_buff_fclones *fclones;
  972. switch (skb->fclone) {
  973. case SKB_FCLONE_UNAVAILABLE:
  974. kmem_cache_free(net_hotdata.skbuff_cache, skb);
  975. return;
  976. case SKB_FCLONE_ORIG:
  977. fclones = container_of(skb, struct sk_buff_fclones, skb1);
  978. /* We usually free the clone (TX completion) before original skb
  979. * This test would have no chance to be true for the clone,
  980. * while here, branch prediction will be good.
  981. */
  982. if (refcount_read(&fclones->fclone_ref) == 1)
  983. goto fastpath;
  984. break;
  985. default: /* SKB_FCLONE_CLONE */
  986. fclones = container_of(skb, struct sk_buff_fclones, skb2);
  987. break;
  988. }
  989. if (!refcount_dec_and_test(&fclones->fclone_ref))
  990. return;
  991. fastpath:
  992. kmem_cache_free(net_hotdata.skbuff_fclone_cache, fclones);
  993. }
  994. void skb_release_head_state(struct sk_buff *skb)
  995. {
  996. skb_dst_drop(skb);
  997. if (skb->destructor) {
  998. DEBUG_NET_WARN_ON_ONCE(in_hardirq());
  999. skb->destructor(skb);
  1000. }
  1001. #if IS_ENABLED(CONFIG_NF_CONNTRACK)
  1002. nf_conntrack_put(skb_nfct(skb));
  1003. #endif
  1004. skb_ext_put(skb);
  1005. }
  1006. /* Free everything but the sk_buff shell. */
  1007. static void skb_release_all(struct sk_buff *skb, enum skb_drop_reason reason)
  1008. {
  1009. skb_release_head_state(skb);
  1010. if (likely(skb->head))
  1011. skb_release_data(skb, reason);
  1012. }
  1013. /**
  1014. * __kfree_skb - private function
  1015. * @skb: buffer
  1016. *
  1017. * Free an sk_buff. Release anything attached to the buffer.
  1018. * Clean the state. This is an internal helper function. Users should
  1019. * always call kfree_skb
  1020. */
  1021. void __kfree_skb(struct sk_buff *skb)
  1022. {
  1023. skb_release_all(skb, SKB_DROP_REASON_NOT_SPECIFIED);
  1024. kfree_skbmem(skb);
  1025. }
  1026. EXPORT_SYMBOL(__kfree_skb);
  1027. static __always_inline
  1028. bool __sk_skb_reason_drop(struct sock *sk, struct sk_buff *skb,
  1029. enum skb_drop_reason reason)
  1030. {
  1031. if (unlikely(!skb_unref(skb)))
  1032. return false;
  1033. DEBUG_NET_WARN_ON_ONCE(reason == SKB_NOT_DROPPED_YET ||
  1034. u32_get_bits(reason,
  1035. SKB_DROP_REASON_SUBSYS_MASK) >=
  1036. SKB_DROP_REASON_SUBSYS_NUM);
  1037. if (reason == SKB_CONSUMED)
  1038. trace_consume_skb(skb, __builtin_return_address(0));
  1039. else
  1040. trace_kfree_skb(skb, __builtin_return_address(0), reason, sk);
  1041. return true;
  1042. }
  1043. /**
  1044. * sk_skb_reason_drop - free an sk_buff with special reason
  1045. * @sk: the socket to receive @skb, or NULL if not applicable
  1046. * @skb: buffer to free
  1047. * @reason: reason why this skb is dropped
  1048. *
  1049. * Drop a reference to the buffer and free it if the usage count has hit
  1050. * zero. Meanwhile, pass the receiving socket and drop reason to
  1051. * 'kfree_skb' tracepoint.
  1052. */
  1053. void __fix_address
  1054. sk_skb_reason_drop(struct sock *sk, struct sk_buff *skb, enum skb_drop_reason reason)
  1055. {
  1056. if (__sk_skb_reason_drop(sk, skb, reason))
  1057. __kfree_skb(skb);
  1058. }
  1059. EXPORT_SYMBOL(sk_skb_reason_drop);
  1060. #define KFREE_SKB_BULK_SIZE 16
  1061. struct skb_free_array {
  1062. unsigned int skb_count;
  1063. void *skb_array[KFREE_SKB_BULK_SIZE];
  1064. };
  1065. static void kfree_skb_add_bulk(struct sk_buff *skb,
  1066. struct skb_free_array *sa,
  1067. enum skb_drop_reason reason)
  1068. {
  1069. /* if SKB is a clone, don't handle this case */
  1070. if (unlikely(skb->fclone != SKB_FCLONE_UNAVAILABLE)) {
  1071. __kfree_skb(skb);
  1072. return;
  1073. }
  1074. skb_release_all(skb, reason);
  1075. sa->skb_array[sa->skb_count++] = skb;
  1076. if (unlikely(sa->skb_count == KFREE_SKB_BULK_SIZE)) {
  1077. kmem_cache_free_bulk(net_hotdata.skbuff_cache, KFREE_SKB_BULK_SIZE,
  1078. sa->skb_array);
  1079. sa->skb_count = 0;
  1080. }
  1081. }
  1082. void __fix_address
  1083. kfree_skb_list_reason(struct sk_buff *segs, enum skb_drop_reason reason)
  1084. {
  1085. struct skb_free_array sa;
  1086. sa.skb_count = 0;
  1087. while (segs) {
  1088. struct sk_buff *next = segs->next;
  1089. if (__sk_skb_reason_drop(NULL, segs, reason)) {
  1090. skb_poison_list(segs);
  1091. kfree_skb_add_bulk(segs, &sa, reason);
  1092. }
  1093. segs = next;
  1094. }
  1095. if (sa.skb_count)
  1096. kmem_cache_free_bulk(net_hotdata.skbuff_cache, sa.skb_count, sa.skb_array);
  1097. }
  1098. EXPORT_SYMBOL(kfree_skb_list_reason);
  1099. /* Dump skb information and contents.
  1100. *
  1101. * Must only be called from net_ratelimit()-ed paths.
  1102. *
  1103. * Dumps whole packets if full_pkt, only headers otherwise.
  1104. */
  1105. void skb_dump(const char *level, const struct sk_buff *skb, bool full_pkt)
  1106. {
  1107. struct skb_shared_info *sh = skb_shinfo(skb);
  1108. struct net_device *dev = skb->dev;
  1109. struct sock *sk = skb->sk;
  1110. struct sk_buff *list_skb;
  1111. bool has_mac, has_trans;
  1112. int headroom, tailroom;
  1113. int i, len, seg_len;
  1114. if (full_pkt)
  1115. len = skb->len;
  1116. else
  1117. len = min_t(int, skb->len, MAX_HEADER + 128);
  1118. headroom = skb_headroom(skb);
  1119. tailroom = skb_tailroom(skb);
  1120. has_mac = skb_mac_header_was_set(skb);
  1121. has_trans = skb_transport_header_was_set(skb);
  1122. printk("%sskb len=%u headroom=%u headlen=%u tailroom=%u\n"
  1123. "mac=(%d,%d) mac_len=%u net=(%d,%d) trans=%d\n"
  1124. "shinfo(txflags=%u nr_frags=%u gso(size=%hu type=%u segs=%hu))\n"
  1125. "csum(0x%x start=%u offset=%u ip_summed=%u complete_sw=%u valid=%u level=%u)\n"
  1126. "hash(0x%x sw=%u l4=%u) proto=0x%04x pkttype=%u iif=%d\n"
  1127. "priority=0x%x mark=0x%x alloc_cpu=%u vlan_all=0x%x\n"
  1128. "encapsulation=%d inner(proto=0x%04x, mac=%u, net=%u, trans=%u)\n",
  1129. level, skb->len, headroom, skb_headlen(skb), tailroom,
  1130. has_mac ? skb->mac_header : -1,
  1131. has_mac ? skb_mac_header_len(skb) : -1,
  1132. skb->mac_len,
  1133. skb->network_header,
  1134. has_trans ? skb_network_header_len(skb) : -1,
  1135. has_trans ? skb->transport_header : -1,
  1136. sh->tx_flags, sh->nr_frags,
  1137. sh->gso_size, sh->gso_type, sh->gso_segs,
  1138. skb->csum, skb->csum_start, skb->csum_offset, skb->ip_summed,
  1139. skb->csum_complete_sw, skb->csum_valid, skb->csum_level,
  1140. skb->hash, skb->sw_hash, skb->l4_hash,
  1141. ntohs(skb->protocol), skb->pkt_type, skb->skb_iif,
  1142. skb->priority, skb->mark, skb->alloc_cpu, skb->vlan_all,
  1143. skb->encapsulation, skb->inner_protocol, skb->inner_mac_header,
  1144. skb->inner_network_header, skb->inner_transport_header);
  1145. if (dev)
  1146. printk("%sdev name=%s feat=%pNF\n",
  1147. level, dev->name, &dev->features);
  1148. if (sk)
  1149. printk("%ssk family=%hu type=%u proto=%u\n",
  1150. level, sk->sk_family, sk->sk_type, sk->sk_protocol);
  1151. if (full_pkt && headroom)
  1152. print_hex_dump(level, "skb headroom: ", DUMP_PREFIX_OFFSET,
  1153. 16, 1, skb->head, headroom, false);
  1154. seg_len = min_t(int, skb_headlen(skb), len);
  1155. if (seg_len)
  1156. print_hex_dump(level, "skb linear: ", DUMP_PREFIX_OFFSET,
  1157. 16, 1, skb->data, seg_len, false);
  1158. len -= seg_len;
  1159. if (full_pkt && tailroom)
  1160. print_hex_dump(level, "skb tailroom: ", DUMP_PREFIX_OFFSET,
  1161. 16, 1, skb_tail_pointer(skb), tailroom, false);
  1162. for (i = 0; len && i < skb_shinfo(skb)->nr_frags; i++) {
  1163. skb_frag_t *frag = &skb_shinfo(skb)->frags[i];
  1164. u32 p_off, p_len, copied;
  1165. struct page *p;
  1166. u8 *vaddr;
  1167. if (skb_frag_is_net_iov(frag)) {
  1168. printk("%sskb frag %d: not readable\n", level, i);
  1169. len -= skb_frag_size(frag);
  1170. if (!len)
  1171. break;
  1172. continue;
  1173. }
  1174. skb_frag_foreach_page(frag, skb_frag_off(frag),
  1175. skb_frag_size(frag), p, p_off, p_len,
  1176. copied) {
  1177. seg_len = min_t(int, p_len, len);
  1178. vaddr = kmap_atomic(p);
  1179. print_hex_dump(level, "skb frag: ",
  1180. DUMP_PREFIX_OFFSET,
  1181. 16, 1, vaddr + p_off, seg_len, false);
  1182. kunmap_atomic(vaddr);
  1183. len -= seg_len;
  1184. if (!len)
  1185. break;
  1186. }
  1187. }
  1188. if (full_pkt && skb_has_frag_list(skb)) {
  1189. printk("skb fraglist:\n");
  1190. skb_walk_frags(skb, list_skb)
  1191. skb_dump(level, list_skb, true);
  1192. }
  1193. }
  1194. EXPORT_SYMBOL(skb_dump);
  1195. /**
  1196. * skb_tx_error - report an sk_buff xmit error
  1197. * @skb: buffer that triggered an error
  1198. *
  1199. * Report xmit error if a device callback is tracking this skb.
  1200. * skb must be freed afterwards.
  1201. */
  1202. void skb_tx_error(struct sk_buff *skb)
  1203. {
  1204. if (skb) {
  1205. skb_zcopy_downgrade_managed(skb);
  1206. skb_zcopy_clear(skb, true);
  1207. }
  1208. }
  1209. EXPORT_SYMBOL(skb_tx_error);
  1210. #ifdef CONFIG_TRACEPOINTS
  1211. /**
  1212. * consume_skb - free an skbuff
  1213. * @skb: buffer to free
  1214. *
  1215. * Drop a ref to the buffer and free it if the usage count has hit zero
  1216. * Functions identically to kfree_skb, but kfree_skb assumes that the frame
  1217. * is being dropped after a failure and notes that
  1218. */
  1219. void consume_skb(struct sk_buff *skb)
  1220. {
  1221. if (!skb_unref(skb))
  1222. return;
  1223. trace_consume_skb(skb, __builtin_return_address(0));
  1224. __kfree_skb(skb);
  1225. }
  1226. EXPORT_SYMBOL(consume_skb);
  1227. #endif
  1228. /**
  1229. * __consume_stateless_skb - free an skbuff, assuming it is stateless
  1230. * @skb: buffer to free
  1231. *
  1232. * Alike consume_skb(), but this variant assumes that this is the last
  1233. * skb reference and all the head states have been already dropped
  1234. */
  1235. void __consume_stateless_skb(struct sk_buff *skb)
  1236. {
  1237. trace_consume_skb(skb, __builtin_return_address(0));
  1238. skb_release_data(skb, SKB_CONSUMED);
  1239. kfree_skbmem(skb);
  1240. }
  1241. static void napi_skb_cache_put(struct sk_buff *skb)
  1242. {
  1243. struct napi_alloc_cache *nc = this_cpu_ptr(&napi_alloc_cache);
  1244. u32 i;
  1245. if (!kasan_mempool_poison_object(skb))
  1246. return;
  1247. local_lock_nested_bh(&napi_alloc_cache.bh_lock);
  1248. nc->skb_cache[nc->skb_count++] = skb;
  1249. if (unlikely(nc->skb_count == NAPI_SKB_CACHE_SIZE)) {
  1250. for (i = NAPI_SKB_CACHE_HALF; i < NAPI_SKB_CACHE_SIZE; i++)
  1251. kasan_mempool_unpoison_object(nc->skb_cache[i],
  1252. kmem_cache_size(net_hotdata.skbuff_cache));
  1253. kmem_cache_free_bulk(net_hotdata.skbuff_cache, NAPI_SKB_CACHE_HALF,
  1254. nc->skb_cache + NAPI_SKB_CACHE_HALF);
  1255. nc->skb_count = NAPI_SKB_CACHE_HALF;
  1256. }
  1257. local_unlock_nested_bh(&napi_alloc_cache.bh_lock);
  1258. }
  1259. void __napi_kfree_skb(struct sk_buff *skb, enum skb_drop_reason reason)
  1260. {
  1261. skb_release_all(skb, reason);
  1262. napi_skb_cache_put(skb);
  1263. }
  1264. void napi_skb_free_stolen_head(struct sk_buff *skb)
  1265. {
  1266. if (unlikely(skb->slow_gro)) {
  1267. nf_reset_ct(skb);
  1268. skb_dst_drop(skb);
  1269. skb_ext_put(skb);
  1270. skb_orphan(skb);
  1271. skb->slow_gro = 0;
  1272. }
  1273. napi_skb_cache_put(skb);
  1274. }
  1275. void napi_consume_skb(struct sk_buff *skb, int budget)
  1276. {
  1277. /* Zero budget indicate non-NAPI context called us, like netpoll */
  1278. if (unlikely(!budget)) {
  1279. dev_consume_skb_any(skb);
  1280. return;
  1281. }
  1282. DEBUG_NET_WARN_ON_ONCE(!in_softirq());
  1283. if (!skb_unref(skb))
  1284. return;
  1285. /* if reaching here SKB is ready to free */
  1286. trace_consume_skb(skb, __builtin_return_address(0));
  1287. /* if SKB is a clone, don't handle this case */
  1288. if (skb->fclone != SKB_FCLONE_UNAVAILABLE) {
  1289. __kfree_skb(skb);
  1290. return;
  1291. }
  1292. skb_release_all(skb, SKB_CONSUMED);
  1293. napi_skb_cache_put(skb);
  1294. }
  1295. EXPORT_SYMBOL(napi_consume_skb);
  1296. /* Make sure a field is contained by headers group */
  1297. #define CHECK_SKB_FIELD(field) \
  1298. BUILD_BUG_ON(offsetof(struct sk_buff, field) != \
  1299. offsetof(struct sk_buff, headers.field)); \
  1300. static void __copy_skb_header(struct sk_buff *new, const struct sk_buff *old)
  1301. {
  1302. new->tstamp = old->tstamp;
  1303. /* We do not copy old->sk */
  1304. new->dev = old->dev;
  1305. memcpy(new->cb, old->cb, sizeof(old->cb));
  1306. skb_dst_copy(new, old);
  1307. __skb_ext_copy(new, old);
  1308. __nf_copy(new, old, false);
  1309. /* Note : this field could be in the headers group.
  1310. * It is not yet because we do not want to have a 16 bit hole
  1311. */
  1312. new->queue_mapping = old->queue_mapping;
  1313. memcpy(&new->headers, &old->headers, sizeof(new->headers));
  1314. CHECK_SKB_FIELD(protocol);
  1315. CHECK_SKB_FIELD(csum);
  1316. CHECK_SKB_FIELD(hash);
  1317. CHECK_SKB_FIELD(priority);
  1318. CHECK_SKB_FIELD(skb_iif);
  1319. CHECK_SKB_FIELD(vlan_proto);
  1320. CHECK_SKB_FIELD(vlan_tci);
  1321. CHECK_SKB_FIELD(transport_header);
  1322. CHECK_SKB_FIELD(network_header);
  1323. CHECK_SKB_FIELD(mac_header);
  1324. CHECK_SKB_FIELD(inner_protocol);
  1325. CHECK_SKB_FIELD(inner_transport_header);
  1326. CHECK_SKB_FIELD(inner_network_header);
  1327. CHECK_SKB_FIELD(inner_mac_header);
  1328. CHECK_SKB_FIELD(mark);
  1329. #ifdef CONFIG_NETWORK_SECMARK
  1330. CHECK_SKB_FIELD(secmark);
  1331. #endif
  1332. #ifdef CONFIG_NET_RX_BUSY_POLL
  1333. CHECK_SKB_FIELD(napi_id);
  1334. #endif
  1335. CHECK_SKB_FIELD(alloc_cpu);
  1336. #ifdef CONFIG_XPS
  1337. CHECK_SKB_FIELD(sender_cpu);
  1338. #endif
  1339. #ifdef CONFIG_NET_SCHED
  1340. CHECK_SKB_FIELD(tc_index);
  1341. #endif
  1342. }
  1343. /*
  1344. * You should not add any new code to this function. Add it to
  1345. * __copy_skb_header above instead.
  1346. */
  1347. static struct sk_buff *__skb_clone(struct sk_buff *n, struct sk_buff *skb)
  1348. {
  1349. #define C(x) n->x = skb->x
  1350. n->next = n->prev = NULL;
  1351. n->sk = NULL;
  1352. __copy_skb_header(n, skb);
  1353. C(len);
  1354. C(data_len);
  1355. C(mac_len);
  1356. n->hdr_len = skb->nohdr ? skb_headroom(skb) : skb->hdr_len;
  1357. n->cloned = 1;
  1358. n->nohdr = 0;
  1359. n->peeked = 0;
  1360. C(pfmemalloc);
  1361. C(pp_recycle);
  1362. n->destructor = NULL;
  1363. C(tail);
  1364. C(end);
  1365. C(head);
  1366. C(head_frag);
  1367. C(data);
  1368. C(truesize);
  1369. refcount_set(&n->users, 1);
  1370. atomic_inc(&(skb_shinfo(skb)->dataref));
  1371. skb->cloned = 1;
  1372. return n;
  1373. #undef C
  1374. }
  1375. /**
  1376. * alloc_skb_for_msg() - allocate sk_buff to wrap frag list forming a msg
  1377. * @first: first sk_buff of the msg
  1378. */
  1379. struct sk_buff *alloc_skb_for_msg(struct sk_buff *first)
  1380. {
  1381. struct sk_buff *n;
  1382. n = alloc_skb(0, GFP_ATOMIC);
  1383. if (!n)
  1384. return NULL;
  1385. n->len = first->len;
  1386. n->data_len = first->len;
  1387. n->truesize = first->truesize;
  1388. skb_shinfo(n)->frag_list = first;
  1389. __copy_skb_header(n, first);
  1390. n->destructor = NULL;
  1391. return n;
  1392. }
  1393. EXPORT_SYMBOL_GPL(alloc_skb_for_msg);
  1394. /**
  1395. * skb_morph - morph one skb into another
  1396. * @dst: the skb to receive the contents
  1397. * @src: the skb to supply the contents
  1398. *
  1399. * This is identical to skb_clone except that the target skb is
  1400. * supplied by the user.
  1401. *
  1402. * The target skb is returned upon exit.
  1403. */
  1404. struct sk_buff *skb_morph(struct sk_buff *dst, struct sk_buff *src)
  1405. {
  1406. skb_release_all(dst, SKB_CONSUMED);
  1407. return __skb_clone(dst, src);
  1408. }
  1409. EXPORT_SYMBOL_GPL(skb_morph);
  1410. int mm_account_pinned_pages(struct mmpin *mmp, size_t size)
  1411. {
  1412. unsigned long max_pg, num_pg, new_pg, old_pg, rlim;
  1413. struct user_struct *user;
  1414. if (capable(CAP_IPC_LOCK) || !size)
  1415. return 0;
  1416. rlim = rlimit(RLIMIT_MEMLOCK);
  1417. if (rlim == RLIM_INFINITY)
  1418. return 0;
  1419. num_pg = (size >> PAGE_SHIFT) + 2; /* worst case */
  1420. max_pg = rlim >> PAGE_SHIFT;
  1421. user = mmp->user ? : current_user();
  1422. old_pg = atomic_long_read(&user->locked_vm);
  1423. do {
  1424. new_pg = old_pg + num_pg;
  1425. if (new_pg > max_pg)
  1426. return -ENOBUFS;
  1427. } while (!atomic_long_try_cmpxchg(&user->locked_vm, &old_pg, new_pg));
  1428. if (!mmp->user) {
  1429. mmp->user = get_uid(user);
  1430. mmp->num_pg = num_pg;
  1431. } else {
  1432. mmp->num_pg += num_pg;
  1433. }
  1434. return 0;
  1435. }
  1436. EXPORT_SYMBOL_GPL(mm_account_pinned_pages);
  1437. void mm_unaccount_pinned_pages(struct mmpin *mmp)
  1438. {
  1439. if (mmp->user) {
  1440. atomic_long_sub(mmp->num_pg, &mmp->user->locked_vm);
  1441. free_uid(mmp->user);
  1442. }
  1443. }
  1444. EXPORT_SYMBOL_GPL(mm_unaccount_pinned_pages);
  1445. static struct ubuf_info *msg_zerocopy_alloc(struct sock *sk, size_t size)
  1446. {
  1447. struct ubuf_info_msgzc *uarg;
  1448. struct sk_buff *skb;
  1449. WARN_ON_ONCE(!in_task());
  1450. skb = sock_omalloc(sk, 0, GFP_KERNEL);
  1451. if (!skb)
  1452. return NULL;
  1453. BUILD_BUG_ON(sizeof(*uarg) > sizeof(skb->cb));
  1454. uarg = (void *)skb->cb;
  1455. uarg->mmp.user = NULL;
  1456. if (mm_account_pinned_pages(&uarg->mmp, size)) {
  1457. kfree_skb(skb);
  1458. return NULL;
  1459. }
  1460. uarg->ubuf.ops = &msg_zerocopy_ubuf_ops;
  1461. uarg->id = ((u32)atomic_inc_return(&sk->sk_zckey)) - 1;
  1462. uarg->len = 1;
  1463. uarg->bytelen = size;
  1464. uarg->zerocopy = 1;
  1465. uarg->ubuf.flags = SKBFL_ZEROCOPY_FRAG | SKBFL_DONT_ORPHAN;
  1466. refcount_set(&uarg->ubuf.refcnt, 1);
  1467. sock_hold(sk);
  1468. return &uarg->ubuf;
  1469. }
  1470. static inline struct sk_buff *skb_from_uarg(struct ubuf_info_msgzc *uarg)
  1471. {
  1472. return container_of((void *)uarg, struct sk_buff, cb);
  1473. }
  1474. struct ubuf_info *msg_zerocopy_realloc(struct sock *sk, size_t size,
  1475. struct ubuf_info *uarg)
  1476. {
  1477. if (uarg) {
  1478. struct ubuf_info_msgzc *uarg_zc;
  1479. const u32 byte_limit = 1 << 19; /* limit to a few TSO */
  1480. u32 bytelen, next;
  1481. /* there might be non MSG_ZEROCOPY users */
  1482. if (uarg->ops != &msg_zerocopy_ubuf_ops)
  1483. return NULL;
  1484. /* realloc only when socket is locked (TCP, UDP cork),
  1485. * so uarg->len and sk_zckey access is serialized
  1486. */
  1487. if (!sock_owned_by_user(sk)) {
  1488. WARN_ON_ONCE(1);
  1489. return NULL;
  1490. }
  1491. uarg_zc = uarg_to_msgzc(uarg);
  1492. bytelen = uarg_zc->bytelen + size;
  1493. if (uarg_zc->len == USHRT_MAX - 1 || bytelen > byte_limit) {
  1494. /* TCP can create new skb to attach new uarg */
  1495. if (sk->sk_type == SOCK_STREAM)
  1496. goto new_alloc;
  1497. return NULL;
  1498. }
  1499. next = (u32)atomic_read(&sk->sk_zckey);
  1500. if ((u32)(uarg_zc->id + uarg_zc->len) == next) {
  1501. if (mm_account_pinned_pages(&uarg_zc->mmp, size))
  1502. return NULL;
  1503. uarg_zc->len++;
  1504. uarg_zc->bytelen = bytelen;
  1505. atomic_set(&sk->sk_zckey, ++next);
  1506. /* no extra ref when appending to datagram (MSG_MORE) */
  1507. if (sk->sk_type == SOCK_STREAM)
  1508. net_zcopy_get(uarg);
  1509. return uarg;
  1510. }
  1511. }
  1512. new_alloc:
  1513. return msg_zerocopy_alloc(sk, size);
  1514. }
  1515. EXPORT_SYMBOL_GPL(msg_zerocopy_realloc);
  1516. static bool skb_zerocopy_notify_extend(struct sk_buff *skb, u32 lo, u16 len)
  1517. {
  1518. struct sock_exterr_skb *serr = SKB_EXT_ERR(skb);
  1519. u32 old_lo, old_hi;
  1520. u64 sum_len;
  1521. old_lo = serr->ee.ee_info;
  1522. old_hi = serr->ee.ee_data;
  1523. sum_len = old_hi - old_lo + 1ULL + len;
  1524. if (sum_len >= (1ULL << 32))
  1525. return false;
  1526. if (lo != old_hi + 1)
  1527. return false;
  1528. serr->ee.ee_data += len;
  1529. return true;
  1530. }
  1531. static void __msg_zerocopy_callback(struct ubuf_info_msgzc *uarg)
  1532. {
  1533. struct sk_buff *tail, *skb = skb_from_uarg(uarg);
  1534. struct sock_exterr_skb *serr;
  1535. struct sock *sk = skb->sk;
  1536. struct sk_buff_head *q;
  1537. unsigned long flags;
  1538. bool is_zerocopy;
  1539. u32 lo, hi;
  1540. u16 len;
  1541. mm_unaccount_pinned_pages(&uarg->mmp);
  1542. /* if !len, there was only 1 call, and it was aborted
  1543. * so do not queue a completion notification
  1544. */
  1545. if (!uarg->len || sock_flag(sk, SOCK_DEAD))
  1546. goto release;
  1547. len = uarg->len;
  1548. lo = uarg->id;
  1549. hi = uarg->id + len - 1;
  1550. is_zerocopy = uarg->zerocopy;
  1551. serr = SKB_EXT_ERR(skb);
  1552. memset(serr, 0, sizeof(*serr));
  1553. serr->ee.ee_errno = 0;
  1554. serr->ee.ee_origin = SO_EE_ORIGIN_ZEROCOPY;
  1555. serr->ee.ee_data = hi;
  1556. serr->ee.ee_info = lo;
  1557. if (!is_zerocopy)
  1558. serr->ee.ee_code |= SO_EE_CODE_ZEROCOPY_COPIED;
  1559. q = &sk->sk_error_queue;
  1560. spin_lock_irqsave(&q->lock, flags);
  1561. tail = skb_peek_tail(q);
  1562. if (!tail || SKB_EXT_ERR(tail)->ee.ee_origin != SO_EE_ORIGIN_ZEROCOPY ||
  1563. !skb_zerocopy_notify_extend(tail, lo, len)) {
  1564. __skb_queue_tail(q, skb);
  1565. skb = NULL;
  1566. }
  1567. spin_unlock_irqrestore(&q->lock, flags);
  1568. sk_error_report(sk);
  1569. release:
  1570. consume_skb(skb);
  1571. sock_put(sk);
  1572. }
  1573. static void msg_zerocopy_complete(struct sk_buff *skb, struct ubuf_info *uarg,
  1574. bool success)
  1575. {
  1576. struct ubuf_info_msgzc *uarg_zc = uarg_to_msgzc(uarg);
  1577. uarg_zc->zerocopy = uarg_zc->zerocopy & success;
  1578. if (refcount_dec_and_test(&uarg->refcnt))
  1579. __msg_zerocopy_callback(uarg_zc);
  1580. }
  1581. void msg_zerocopy_put_abort(struct ubuf_info *uarg, bool have_uref)
  1582. {
  1583. struct sock *sk = skb_from_uarg(uarg_to_msgzc(uarg))->sk;
  1584. atomic_dec(&sk->sk_zckey);
  1585. uarg_to_msgzc(uarg)->len--;
  1586. if (have_uref)
  1587. msg_zerocopy_complete(NULL, uarg, true);
  1588. }
  1589. EXPORT_SYMBOL_GPL(msg_zerocopy_put_abort);
  1590. const struct ubuf_info_ops msg_zerocopy_ubuf_ops = {
  1591. .complete = msg_zerocopy_complete,
  1592. };
  1593. EXPORT_SYMBOL_GPL(msg_zerocopy_ubuf_ops);
  1594. int skb_zerocopy_iter_stream(struct sock *sk, struct sk_buff *skb,
  1595. struct msghdr *msg, int len,
  1596. struct ubuf_info *uarg)
  1597. {
  1598. int err, orig_len = skb->len;
  1599. if (uarg->ops->link_skb) {
  1600. err = uarg->ops->link_skb(skb, uarg);
  1601. if (err)
  1602. return err;
  1603. } else {
  1604. struct ubuf_info *orig_uarg = skb_zcopy(skb);
  1605. /* An skb can only point to one uarg. This edge case happens
  1606. * when TCP appends to an skb, but zerocopy_realloc triggered
  1607. * a new alloc.
  1608. */
  1609. if (orig_uarg && uarg != orig_uarg)
  1610. return -EEXIST;
  1611. }
  1612. err = __zerocopy_sg_from_iter(msg, sk, skb, &msg->msg_iter, len);
  1613. if (err == -EFAULT || (err == -EMSGSIZE && skb->len == orig_len)) {
  1614. struct sock *save_sk = skb->sk;
  1615. /* Streams do not free skb on error. Reset to prev state. */
  1616. iov_iter_revert(&msg->msg_iter, skb->len - orig_len);
  1617. skb->sk = sk;
  1618. ___pskb_trim(skb, orig_len);
  1619. skb->sk = save_sk;
  1620. return err;
  1621. }
  1622. skb_zcopy_set(skb, uarg, NULL);
  1623. return skb->len - orig_len;
  1624. }
  1625. EXPORT_SYMBOL_GPL(skb_zerocopy_iter_stream);
  1626. void __skb_zcopy_downgrade_managed(struct sk_buff *skb)
  1627. {
  1628. int i;
  1629. skb_shinfo(skb)->flags &= ~SKBFL_MANAGED_FRAG_REFS;
  1630. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++)
  1631. skb_frag_ref(skb, i);
  1632. }
  1633. EXPORT_SYMBOL_GPL(__skb_zcopy_downgrade_managed);
  1634. static int skb_zerocopy_clone(struct sk_buff *nskb, struct sk_buff *orig,
  1635. gfp_t gfp_mask)
  1636. {
  1637. if (skb_zcopy(orig)) {
  1638. if (skb_zcopy(nskb)) {
  1639. /* !gfp_mask callers are verified to !skb_zcopy(nskb) */
  1640. if (!gfp_mask) {
  1641. WARN_ON_ONCE(1);
  1642. return -ENOMEM;
  1643. }
  1644. if (skb_uarg(nskb) == skb_uarg(orig))
  1645. return 0;
  1646. if (skb_copy_ubufs(nskb, GFP_ATOMIC))
  1647. return -EIO;
  1648. }
  1649. skb_zcopy_set(nskb, skb_uarg(orig), NULL);
  1650. }
  1651. return 0;
  1652. }
  1653. /**
  1654. * skb_copy_ubufs - copy userspace skb frags buffers to kernel
  1655. * @skb: the skb to modify
  1656. * @gfp_mask: allocation priority
  1657. *
  1658. * This must be called on skb with SKBFL_ZEROCOPY_ENABLE.
  1659. * It will copy all frags into kernel and drop the reference
  1660. * to userspace pages.
  1661. *
  1662. * If this function is called from an interrupt gfp_mask() must be
  1663. * %GFP_ATOMIC.
  1664. *
  1665. * Returns 0 on success or a negative error code on failure
  1666. * to allocate kernel memory to copy to.
  1667. */
  1668. int skb_copy_ubufs(struct sk_buff *skb, gfp_t gfp_mask)
  1669. {
  1670. int num_frags = skb_shinfo(skb)->nr_frags;
  1671. struct page *page, *head = NULL;
  1672. int i, order, psize, new_frags;
  1673. u32 d_off;
  1674. if (skb_shared(skb) || skb_unclone(skb, gfp_mask))
  1675. return -EINVAL;
  1676. if (!skb_frags_readable(skb))
  1677. return -EFAULT;
  1678. if (!num_frags)
  1679. goto release;
  1680. /* We might have to allocate high order pages, so compute what minimum
  1681. * page order is needed.
  1682. */
  1683. order = 0;
  1684. while ((PAGE_SIZE << order) * MAX_SKB_FRAGS < __skb_pagelen(skb))
  1685. order++;
  1686. psize = (PAGE_SIZE << order);
  1687. new_frags = (__skb_pagelen(skb) + psize - 1) >> (PAGE_SHIFT + order);
  1688. for (i = 0; i < new_frags; i++) {
  1689. page = alloc_pages(gfp_mask | __GFP_COMP, order);
  1690. if (!page) {
  1691. while (head) {
  1692. struct page *next = (struct page *)page_private(head);
  1693. put_page(head);
  1694. head = next;
  1695. }
  1696. return -ENOMEM;
  1697. }
  1698. set_page_private(page, (unsigned long)head);
  1699. head = page;
  1700. }
  1701. page = head;
  1702. d_off = 0;
  1703. for (i = 0; i < num_frags; i++) {
  1704. skb_frag_t *f = &skb_shinfo(skb)->frags[i];
  1705. u32 p_off, p_len, copied;
  1706. struct page *p;
  1707. u8 *vaddr;
  1708. skb_frag_foreach_page(f, skb_frag_off(f), skb_frag_size(f),
  1709. p, p_off, p_len, copied) {
  1710. u32 copy, done = 0;
  1711. vaddr = kmap_atomic(p);
  1712. while (done < p_len) {
  1713. if (d_off == psize) {
  1714. d_off = 0;
  1715. page = (struct page *)page_private(page);
  1716. }
  1717. copy = min_t(u32, psize - d_off, p_len - done);
  1718. memcpy(page_address(page) + d_off,
  1719. vaddr + p_off + done, copy);
  1720. done += copy;
  1721. d_off += copy;
  1722. }
  1723. kunmap_atomic(vaddr);
  1724. }
  1725. }
  1726. /* skb frags release userspace buffers */
  1727. for (i = 0; i < num_frags; i++)
  1728. skb_frag_unref(skb, i);
  1729. /* skb frags point to kernel buffers */
  1730. for (i = 0; i < new_frags - 1; i++) {
  1731. __skb_fill_netmem_desc(skb, i, page_to_netmem(head), 0, psize);
  1732. head = (struct page *)page_private(head);
  1733. }
  1734. __skb_fill_netmem_desc(skb, new_frags - 1, page_to_netmem(head), 0,
  1735. d_off);
  1736. skb_shinfo(skb)->nr_frags = new_frags;
  1737. release:
  1738. skb_zcopy_clear(skb, false);
  1739. return 0;
  1740. }
  1741. EXPORT_SYMBOL_GPL(skb_copy_ubufs);
  1742. /**
  1743. * skb_clone - duplicate an sk_buff
  1744. * @skb: buffer to clone
  1745. * @gfp_mask: allocation priority
  1746. *
  1747. * Duplicate an &sk_buff. The new one is not owned by a socket. Both
  1748. * copies share the same packet data but not structure. The new
  1749. * buffer has a reference count of 1. If the allocation fails the
  1750. * function returns %NULL otherwise the new buffer is returned.
  1751. *
  1752. * If this function is called from an interrupt gfp_mask() must be
  1753. * %GFP_ATOMIC.
  1754. */
  1755. struct sk_buff *skb_clone(struct sk_buff *skb, gfp_t gfp_mask)
  1756. {
  1757. struct sk_buff_fclones *fclones = container_of(skb,
  1758. struct sk_buff_fclones,
  1759. skb1);
  1760. struct sk_buff *n;
  1761. if (skb_orphan_frags(skb, gfp_mask))
  1762. return NULL;
  1763. if (skb->fclone == SKB_FCLONE_ORIG &&
  1764. refcount_read(&fclones->fclone_ref) == 1) {
  1765. n = &fclones->skb2;
  1766. refcount_set(&fclones->fclone_ref, 2);
  1767. n->fclone = SKB_FCLONE_CLONE;
  1768. } else {
  1769. if (skb_pfmemalloc(skb))
  1770. gfp_mask |= __GFP_MEMALLOC;
  1771. n = kmem_cache_alloc(net_hotdata.skbuff_cache, gfp_mask);
  1772. if (!n)
  1773. return NULL;
  1774. n->fclone = SKB_FCLONE_UNAVAILABLE;
  1775. }
  1776. return __skb_clone(n, skb);
  1777. }
  1778. EXPORT_SYMBOL(skb_clone);
  1779. void skb_headers_offset_update(struct sk_buff *skb, int off)
  1780. {
  1781. /* Only adjust this if it actually is csum_start rather than csum */
  1782. if (skb->ip_summed == CHECKSUM_PARTIAL)
  1783. skb->csum_start += off;
  1784. /* {transport,network,mac}_header and tail are relative to skb->head */
  1785. skb->transport_header += off;
  1786. skb->network_header += off;
  1787. if (skb_mac_header_was_set(skb))
  1788. skb->mac_header += off;
  1789. skb->inner_transport_header += off;
  1790. skb->inner_network_header += off;
  1791. skb->inner_mac_header += off;
  1792. }
  1793. EXPORT_SYMBOL(skb_headers_offset_update);
  1794. void skb_copy_header(struct sk_buff *new, const struct sk_buff *old)
  1795. {
  1796. __copy_skb_header(new, old);
  1797. skb_shinfo(new)->gso_size = skb_shinfo(old)->gso_size;
  1798. skb_shinfo(new)->gso_segs = skb_shinfo(old)->gso_segs;
  1799. skb_shinfo(new)->gso_type = skb_shinfo(old)->gso_type;
  1800. }
  1801. EXPORT_SYMBOL(skb_copy_header);
  1802. static inline int skb_alloc_rx_flag(const struct sk_buff *skb)
  1803. {
  1804. if (skb_pfmemalloc(skb))
  1805. return SKB_ALLOC_RX;
  1806. return 0;
  1807. }
  1808. /**
  1809. * skb_copy - create private copy of an sk_buff
  1810. * @skb: buffer to copy
  1811. * @gfp_mask: allocation priority
  1812. *
  1813. * Make a copy of both an &sk_buff and its data. This is used when the
  1814. * caller wishes to modify the data and needs a private copy of the
  1815. * data to alter. Returns %NULL on failure or the pointer to the buffer
  1816. * on success. The returned buffer has a reference count of 1.
  1817. *
  1818. * As by-product this function converts non-linear &sk_buff to linear
  1819. * one, so that &sk_buff becomes completely private and caller is allowed
  1820. * to modify all the data of returned buffer. This means that this
  1821. * function is not recommended for use in circumstances when only
  1822. * header is going to be modified. Use pskb_copy() instead.
  1823. */
  1824. struct sk_buff *skb_copy(const struct sk_buff *skb, gfp_t gfp_mask)
  1825. {
  1826. struct sk_buff *n;
  1827. unsigned int size;
  1828. int headerlen;
  1829. if (!skb_frags_readable(skb))
  1830. return NULL;
  1831. if (WARN_ON_ONCE(skb_shinfo(skb)->gso_type & SKB_GSO_FRAGLIST))
  1832. return NULL;
  1833. headerlen = skb_headroom(skb);
  1834. size = skb_end_offset(skb) + skb->data_len;
  1835. n = __alloc_skb(size, gfp_mask,
  1836. skb_alloc_rx_flag(skb), NUMA_NO_NODE);
  1837. if (!n)
  1838. return NULL;
  1839. /* Set the data pointer */
  1840. skb_reserve(n, headerlen);
  1841. /* Set the tail pointer and length */
  1842. skb_put(n, skb->len);
  1843. BUG_ON(skb_copy_bits(skb, -headerlen, n->head, headerlen + skb->len));
  1844. skb_copy_header(n, skb);
  1845. return n;
  1846. }
  1847. EXPORT_SYMBOL(skb_copy);
  1848. /**
  1849. * __pskb_copy_fclone - create copy of an sk_buff with private head.
  1850. * @skb: buffer to copy
  1851. * @headroom: headroom of new skb
  1852. * @gfp_mask: allocation priority
  1853. * @fclone: if true allocate the copy of the skb from the fclone
  1854. * cache instead of the head cache; it is recommended to set this
  1855. * to true for the cases where the copy will likely be cloned
  1856. *
  1857. * Make a copy of both an &sk_buff and part of its data, located
  1858. * in header. Fragmented data remain shared. This is used when
  1859. * the caller wishes to modify only header of &sk_buff and needs
  1860. * private copy of the header to alter. Returns %NULL on failure
  1861. * or the pointer to the buffer on success.
  1862. * The returned buffer has a reference count of 1.
  1863. */
  1864. struct sk_buff *__pskb_copy_fclone(struct sk_buff *skb, int headroom,
  1865. gfp_t gfp_mask, bool fclone)
  1866. {
  1867. unsigned int size = skb_headlen(skb) + headroom;
  1868. int flags = skb_alloc_rx_flag(skb) | (fclone ? SKB_ALLOC_FCLONE : 0);
  1869. struct sk_buff *n = __alloc_skb(size, gfp_mask, flags, NUMA_NO_NODE);
  1870. if (!n)
  1871. goto out;
  1872. /* Set the data pointer */
  1873. skb_reserve(n, headroom);
  1874. /* Set the tail pointer and length */
  1875. skb_put(n, skb_headlen(skb));
  1876. /* Copy the bytes */
  1877. skb_copy_from_linear_data(skb, n->data, n->len);
  1878. n->truesize += skb->data_len;
  1879. n->data_len = skb->data_len;
  1880. n->len = skb->len;
  1881. if (skb_shinfo(skb)->nr_frags) {
  1882. int i;
  1883. if (skb_orphan_frags(skb, gfp_mask) ||
  1884. skb_zerocopy_clone(n, skb, gfp_mask)) {
  1885. kfree_skb(n);
  1886. n = NULL;
  1887. goto out;
  1888. }
  1889. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  1890. skb_shinfo(n)->frags[i] = skb_shinfo(skb)->frags[i];
  1891. skb_frag_ref(skb, i);
  1892. }
  1893. skb_shinfo(n)->nr_frags = i;
  1894. }
  1895. if (skb_has_frag_list(skb)) {
  1896. skb_shinfo(n)->frag_list = skb_shinfo(skb)->frag_list;
  1897. skb_clone_fraglist(n);
  1898. }
  1899. skb_copy_header(n, skb);
  1900. out:
  1901. return n;
  1902. }
  1903. EXPORT_SYMBOL(__pskb_copy_fclone);
  1904. /**
  1905. * pskb_expand_head - reallocate header of &sk_buff
  1906. * @skb: buffer to reallocate
  1907. * @nhead: room to add at head
  1908. * @ntail: room to add at tail
  1909. * @gfp_mask: allocation priority
  1910. *
  1911. * Expands (or creates identical copy, if @nhead and @ntail are zero)
  1912. * header of @skb. &sk_buff itself is not changed. &sk_buff MUST have
  1913. * reference count of 1. Returns zero in the case of success or error,
  1914. * if expansion failed. In the last case, &sk_buff is not changed.
  1915. *
  1916. * All the pointers pointing into skb header may change and must be
  1917. * reloaded after call to this function.
  1918. */
  1919. int pskb_expand_head(struct sk_buff *skb, int nhead, int ntail,
  1920. gfp_t gfp_mask)
  1921. {
  1922. unsigned int osize = skb_end_offset(skb);
  1923. unsigned int size = osize + nhead + ntail;
  1924. long off;
  1925. u8 *data;
  1926. int i;
  1927. BUG_ON(nhead < 0);
  1928. BUG_ON(skb_shared(skb));
  1929. skb_zcopy_downgrade_managed(skb);
  1930. if (skb_pfmemalloc(skb))
  1931. gfp_mask |= __GFP_MEMALLOC;
  1932. data = kmalloc_reserve(&size, gfp_mask, NUMA_NO_NODE, NULL);
  1933. if (!data)
  1934. goto nodata;
  1935. size = SKB_WITH_OVERHEAD(size);
  1936. /* Copy only real data... and, alas, header. This should be
  1937. * optimized for the cases when header is void.
  1938. */
  1939. memcpy(data + nhead, skb->head, skb_tail_pointer(skb) - skb->head);
  1940. memcpy((struct skb_shared_info *)(data + size),
  1941. skb_shinfo(skb),
  1942. offsetof(struct skb_shared_info, frags[skb_shinfo(skb)->nr_frags]));
  1943. /*
  1944. * if shinfo is shared we must drop the old head gracefully, but if it
  1945. * is not we can just drop the old head and let the existing refcount
  1946. * be since all we did is relocate the values
  1947. */
  1948. if (skb_cloned(skb)) {
  1949. if (skb_orphan_frags(skb, gfp_mask))
  1950. goto nofrags;
  1951. if (skb_zcopy(skb))
  1952. refcount_inc(&skb_uarg(skb)->refcnt);
  1953. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++)
  1954. skb_frag_ref(skb, i);
  1955. if (skb_has_frag_list(skb))
  1956. skb_clone_fraglist(skb);
  1957. skb_release_data(skb, SKB_CONSUMED);
  1958. } else {
  1959. skb_free_head(skb);
  1960. }
  1961. off = (data + nhead) - skb->head;
  1962. skb->head = data;
  1963. skb->head_frag = 0;
  1964. skb->data += off;
  1965. skb_set_end_offset(skb, size);
  1966. #ifdef NET_SKBUFF_DATA_USES_OFFSET
  1967. off = nhead;
  1968. #endif
  1969. skb->tail += off;
  1970. skb_headers_offset_update(skb, nhead);
  1971. skb->cloned = 0;
  1972. skb->hdr_len = 0;
  1973. skb->nohdr = 0;
  1974. atomic_set(&skb_shinfo(skb)->dataref, 1);
  1975. skb_metadata_clear(skb);
  1976. /* It is not generally safe to change skb->truesize.
  1977. * For the moment, we really care of rx path, or
  1978. * when skb is orphaned (not attached to a socket).
  1979. */
  1980. if (!skb->sk || skb->destructor == sock_edemux)
  1981. skb->truesize += size - osize;
  1982. return 0;
  1983. nofrags:
  1984. skb_kfree_head(data, size);
  1985. nodata:
  1986. return -ENOMEM;
  1987. }
  1988. EXPORT_SYMBOL(pskb_expand_head);
  1989. /* Make private copy of skb with writable head and some headroom */
  1990. struct sk_buff *skb_realloc_headroom(struct sk_buff *skb, unsigned int headroom)
  1991. {
  1992. struct sk_buff *skb2;
  1993. int delta = headroom - skb_headroom(skb);
  1994. if (delta <= 0)
  1995. skb2 = pskb_copy(skb, GFP_ATOMIC);
  1996. else {
  1997. skb2 = skb_clone(skb, GFP_ATOMIC);
  1998. if (skb2 && pskb_expand_head(skb2, SKB_DATA_ALIGN(delta), 0,
  1999. GFP_ATOMIC)) {
  2000. kfree_skb(skb2);
  2001. skb2 = NULL;
  2002. }
  2003. }
  2004. return skb2;
  2005. }
  2006. EXPORT_SYMBOL(skb_realloc_headroom);
  2007. /* Note: We plan to rework this in linux-6.4 */
  2008. int __skb_unclone_keeptruesize(struct sk_buff *skb, gfp_t pri)
  2009. {
  2010. unsigned int saved_end_offset, saved_truesize;
  2011. struct skb_shared_info *shinfo;
  2012. int res;
  2013. saved_end_offset = skb_end_offset(skb);
  2014. saved_truesize = skb->truesize;
  2015. res = pskb_expand_head(skb, 0, 0, pri);
  2016. if (res)
  2017. return res;
  2018. skb->truesize = saved_truesize;
  2019. if (likely(skb_end_offset(skb) == saved_end_offset))
  2020. return 0;
  2021. /* We can not change skb->end if the original or new value
  2022. * is SKB_SMALL_HEAD_HEADROOM, as it might break skb_kfree_head().
  2023. */
  2024. if (saved_end_offset == SKB_SMALL_HEAD_HEADROOM ||
  2025. skb_end_offset(skb) == SKB_SMALL_HEAD_HEADROOM) {
  2026. /* We think this path should not be taken.
  2027. * Add a temporary trace to warn us just in case.
  2028. */
  2029. pr_err_once("__skb_unclone_keeptruesize() skb_end_offset() %u -> %u\n",
  2030. saved_end_offset, skb_end_offset(skb));
  2031. WARN_ON_ONCE(1);
  2032. return 0;
  2033. }
  2034. shinfo = skb_shinfo(skb);
  2035. /* We are about to change back skb->end,
  2036. * we need to move skb_shinfo() to its new location.
  2037. */
  2038. memmove(skb->head + saved_end_offset,
  2039. shinfo,
  2040. offsetof(struct skb_shared_info, frags[shinfo->nr_frags]));
  2041. skb_set_end_offset(skb, saved_end_offset);
  2042. return 0;
  2043. }
  2044. /**
  2045. * skb_expand_head - reallocate header of &sk_buff
  2046. * @skb: buffer to reallocate
  2047. * @headroom: needed headroom
  2048. *
  2049. * Unlike skb_realloc_headroom, this one does not allocate a new skb
  2050. * if possible; copies skb->sk to new skb as needed
  2051. * and frees original skb in case of failures.
  2052. *
  2053. * It expect increased headroom and generates warning otherwise.
  2054. */
  2055. struct sk_buff *skb_expand_head(struct sk_buff *skb, unsigned int headroom)
  2056. {
  2057. int delta = headroom - skb_headroom(skb);
  2058. int osize = skb_end_offset(skb);
  2059. struct sock *sk = skb->sk;
  2060. if (WARN_ONCE(delta <= 0,
  2061. "%s is expecting an increase in the headroom", __func__))
  2062. return skb;
  2063. delta = SKB_DATA_ALIGN(delta);
  2064. /* pskb_expand_head() might crash, if skb is shared. */
  2065. if (skb_shared(skb) || !is_skb_wmem(skb)) {
  2066. struct sk_buff *nskb = skb_clone(skb, GFP_ATOMIC);
  2067. if (unlikely(!nskb))
  2068. goto fail;
  2069. if (sk)
  2070. skb_set_owner_w(nskb, sk);
  2071. consume_skb(skb);
  2072. skb = nskb;
  2073. }
  2074. if (pskb_expand_head(skb, delta, 0, GFP_ATOMIC))
  2075. goto fail;
  2076. if (sk && is_skb_wmem(skb)) {
  2077. delta = skb_end_offset(skb) - osize;
  2078. refcount_add(delta, &sk->sk_wmem_alloc);
  2079. skb->truesize += delta;
  2080. }
  2081. return skb;
  2082. fail:
  2083. kfree_skb(skb);
  2084. return NULL;
  2085. }
  2086. EXPORT_SYMBOL(skb_expand_head);
  2087. /**
  2088. * skb_copy_expand - copy and expand sk_buff
  2089. * @skb: buffer to copy
  2090. * @newheadroom: new free bytes at head
  2091. * @newtailroom: new free bytes at tail
  2092. * @gfp_mask: allocation priority
  2093. *
  2094. * Make a copy of both an &sk_buff and its data and while doing so
  2095. * allocate additional space.
  2096. *
  2097. * This is used when the caller wishes to modify the data and needs a
  2098. * private copy of the data to alter as well as more space for new fields.
  2099. * Returns %NULL on failure or the pointer to the buffer
  2100. * on success. The returned buffer has a reference count of 1.
  2101. *
  2102. * You must pass %GFP_ATOMIC as the allocation priority if this function
  2103. * is called from an interrupt.
  2104. */
  2105. struct sk_buff *skb_copy_expand(const struct sk_buff *skb,
  2106. int newheadroom, int newtailroom,
  2107. gfp_t gfp_mask)
  2108. {
  2109. /*
  2110. * Allocate the copy buffer
  2111. */
  2112. int head_copy_len, head_copy_off;
  2113. struct sk_buff *n;
  2114. int oldheadroom;
  2115. if (!skb_frags_readable(skb))
  2116. return NULL;
  2117. if (WARN_ON_ONCE(skb_shinfo(skb)->gso_type & SKB_GSO_FRAGLIST))
  2118. return NULL;
  2119. oldheadroom = skb_headroom(skb);
  2120. n = __alloc_skb(newheadroom + skb->len + newtailroom,
  2121. gfp_mask, skb_alloc_rx_flag(skb),
  2122. NUMA_NO_NODE);
  2123. if (!n)
  2124. return NULL;
  2125. skb_reserve(n, newheadroom);
  2126. /* Set the tail pointer and length */
  2127. skb_put(n, skb->len);
  2128. head_copy_len = oldheadroom;
  2129. head_copy_off = 0;
  2130. if (newheadroom <= head_copy_len)
  2131. head_copy_len = newheadroom;
  2132. else
  2133. head_copy_off = newheadroom - head_copy_len;
  2134. /* Copy the linear header and data. */
  2135. BUG_ON(skb_copy_bits(skb, -head_copy_len, n->head + head_copy_off,
  2136. skb->len + head_copy_len));
  2137. skb_copy_header(n, skb);
  2138. skb_headers_offset_update(n, newheadroom - oldheadroom);
  2139. return n;
  2140. }
  2141. EXPORT_SYMBOL(skb_copy_expand);
  2142. /**
  2143. * __skb_pad - zero pad the tail of an skb
  2144. * @skb: buffer to pad
  2145. * @pad: space to pad
  2146. * @free_on_error: free buffer on error
  2147. *
  2148. * Ensure that a buffer is followed by a padding area that is zero
  2149. * filled. Used by network drivers which may DMA or transfer data
  2150. * beyond the buffer end onto the wire.
  2151. *
  2152. * May return error in out of memory cases. The skb is freed on error
  2153. * if @free_on_error is true.
  2154. */
  2155. int __skb_pad(struct sk_buff *skb, int pad, bool free_on_error)
  2156. {
  2157. int err;
  2158. int ntail;
  2159. /* If the skbuff is non linear tailroom is always zero.. */
  2160. if (!skb_cloned(skb) && skb_tailroom(skb) >= pad) {
  2161. memset(skb->data+skb->len, 0, pad);
  2162. return 0;
  2163. }
  2164. ntail = skb->data_len + pad - (skb->end - skb->tail);
  2165. if (likely(skb_cloned(skb) || ntail > 0)) {
  2166. err = pskb_expand_head(skb, 0, ntail, GFP_ATOMIC);
  2167. if (unlikely(err))
  2168. goto free_skb;
  2169. }
  2170. /* FIXME: The use of this function with non-linear skb's really needs
  2171. * to be audited.
  2172. */
  2173. err = skb_linearize(skb);
  2174. if (unlikely(err))
  2175. goto free_skb;
  2176. memset(skb->data + skb->len, 0, pad);
  2177. return 0;
  2178. free_skb:
  2179. if (free_on_error)
  2180. kfree_skb(skb);
  2181. return err;
  2182. }
  2183. EXPORT_SYMBOL(__skb_pad);
  2184. /**
  2185. * pskb_put - add data to the tail of a potentially fragmented buffer
  2186. * @skb: start of the buffer to use
  2187. * @tail: tail fragment of the buffer to use
  2188. * @len: amount of data to add
  2189. *
  2190. * This function extends the used data area of the potentially
  2191. * fragmented buffer. @tail must be the last fragment of @skb -- or
  2192. * @skb itself. If this would exceed the total buffer size the kernel
  2193. * will panic. A pointer to the first byte of the extra data is
  2194. * returned.
  2195. */
  2196. void *pskb_put(struct sk_buff *skb, struct sk_buff *tail, int len)
  2197. {
  2198. if (tail != skb) {
  2199. skb->data_len += len;
  2200. skb->len += len;
  2201. }
  2202. return skb_put(tail, len);
  2203. }
  2204. EXPORT_SYMBOL_GPL(pskb_put);
  2205. /**
  2206. * skb_put - add data to a buffer
  2207. * @skb: buffer to use
  2208. * @len: amount of data to add
  2209. *
  2210. * This function extends the used data area of the buffer. If this would
  2211. * exceed the total buffer size the kernel will panic. A pointer to the
  2212. * first byte of the extra data is returned.
  2213. */
  2214. void *skb_put(struct sk_buff *skb, unsigned int len)
  2215. {
  2216. void *tmp = skb_tail_pointer(skb);
  2217. SKB_LINEAR_ASSERT(skb);
  2218. skb->tail += len;
  2219. skb->len += len;
  2220. if (unlikely(skb->tail > skb->end))
  2221. skb_over_panic(skb, len, __builtin_return_address(0));
  2222. return tmp;
  2223. }
  2224. EXPORT_SYMBOL(skb_put);
  2225. /**
  2226. * skb_push - add data to the start of a buffer
  2227. * @skb: buffer to use
  2228. * @len: amount of data to add
  2229. *
  2230. * This function extends the used data area of the buffer at the buffer
  2231. * start. If this would exceed the total buffer headroom the kernel will
  2232. * panic. A pointer to the first byte of the extra data is returned.
  2233. */
  2234. void *skb_push(struct sk_buff *skb, unsigned int len)
  2235. {
  2236. skb->data -= len;
  2237. skb->len += len;
  2238. if (unlikely(skb->data < skb->head))
  2239. skb_under_panic(skb, len, __builtin_return_address(0));
  2240. return skb->data;
  2241. }
  2242. EXPORT_SYMBOL(skb_push);
  2243. /**
  2244. * skb_pull - remove data from the start of a buffer
  2245. * @skb: buffer to use
  2246. * @len: amount of data to remove
  2247. *
  2248. * This function removes data from the start of a buffer, returning
  2249. * the memory to the headroom. A pointer to the next data in the buffer
  2250. * is returned. Once the data has been pulled future pushes will overwrite
  2251. * the old data.
  2252. */
  2253. void *skb_pull(struct sk_buff *skb, unsigned int len)
  2254. {
  2255. return skb_pull_inline(skb, len);
  2256. }
  2257. EXPORT_SYMBOL(skb_pull);
  2258. /**
  2259. * skb_pull_data - remove data from the start of a buffer returning its
  2260. * original position.
  2261. * @skb: buffer to use
  2262. * @len: amount of data to remove
  2263. *
  2264. * This function removes data from the start of a buffer, returning
  2265. * the memory to the headroom. A pointer to the original data in the buffer
  2266. * is returned after checking if there is enough data to pull. Once the
  2267. * data has been pulled future pushes will overwrite the old data.
  2268. */
  2269. void *skb_pull_data(struct sk_buff *skb, size_t len)
  2270. {
  2271. void *data = skb->data;
  2272. if (skb->len < len)
  2273. return NULL;
  2274. skb_pull(skb, len);
  2275. return data;
  2276. }
  2277. EXPORT_SYMBOL(skb_pull_data);
  2278. /**
  2279. * skb_trim - remove end from a buffer
  2280. * @skb: buffer to alter
  2281. * @len: new length
  2282. *
  2283. * Cut the length of a buffer down by removing data from the tail. If
  2284. * the buffer is already under the length specified it is not modified.
  2285. * The skb must be linear.
  2286. */
  2287. void skb_trim(struct sk_buff *skb, unsigned int len)
  2288. {
  2289. if (skb->len > len)
  2290. __skb_trim(skb, len);
  2291. }
  2292. EXPORT_SYMBOL(skb_trim);
  2293. /* Trims skb to length len. It can change skb pointers.
  2294. */
  2295. int ___pskb_trim(struct sk_buff *skb, unsigned int len)
  2296. {
  2297. struct sk_buff **fragp;
  2298. struct sk_buff *frag;
  2299. int offset = skb_headlen(skb);
  2300. int nfrags = skb_shinfo(skb)->nr_frags;
  2301. int i;
  2302. int err;
  2303. if (skb_cloned(skb) &&
  2304. unlikely((err = pskb_expand_head(skb, 0, 0, GFP_ATOMIC))))
  2305. return err;
  2306. i = 0;
  2307. if (offset >= len)
  2308. goto drop_pages;
  2309. for (; i < nfrags; i++) {
  2310. int end = offset + skb_frag_size(&skb_shinfo(skb)->frags[i]);
  2311. if (end < len) {
  2312. offset = end;
  2313. continue;
  2314. }
  2315. skb_frag_size_set(&skb_shinfo(skb)->frags[i++], len - offset);
  2316. drop_pages:
  2317. skb_shinfo(skb)->nr_frags = i;
  2318. for (; i < nfrags; i++)
  2319. skb_frag_unref(skb, i);
  2320. if (skb_has_frag_list(skb))
  2321. skb_drop_fraglist(skb);
  2322. goto done;
  2323. }
  2324. for (fragp = &skb_shinfo(skb)->frag_list; (frag = *fragp);
  2325. fragp = &frag->next) {
  2326. int end = offset + frag->len;
  2327. if (skb_shared(frag)) {
  2328. struct sk_buff *nfrag;
  2329. nfrag = skb_clone(frag, GFP_ATOMIC);
  2330. if (unlikely(!nfrag))
  2331. return -ENOMEM;
  2332. nfrag->next = frag->next;
  2333. consume_skb(frag);
  2334. frag = nfrag;
  2335. *fragp = frag;
  2336. }
  2337. if (end < len) {
  2338. offset = end;
  2339. continue;
  2340. }
  2341. if (end > len &&
  2342. unlikely((err = pskb_trim(frag, len - offset))))
  2343. return err;
  2344. if (frag->next)
  2345. skb_drop_list(&frag->next);
  2346. break;
  2347. }
  2348. done:
  2349. if (len > skb_headlen(skb)) {
  2350. skb->data_len -= skb->len - len;
  2351. skb->len = len;
  2352. } else {
  2353. skb->len = len;
  2354. skb->data_len = 0;
  2355. skb_set_tail_pointer(skb, len);
  2356. }
  2357. if (!skb->sk || skb->destructor == sock_edemux)
  2358. skb_condense(skb);
  2359. return 0;
  2360. }
  2361. EXPORT_SYMBOL(___pskb_trim);
  2362. /* Note : use pskb_trim_rcsum() instead of calling this directly
  2363. */
  2364. int pskb_trim_rcsum_slow(struct sk_buff *skb, unsigned int len)
  2365. {
  2366. if (skb->ip_summed == CHECKSUM_COMPLETE) {
  2367. int delta = skb->len - len;
  2368. skb->csum = csum_block_sub(skb->csum,
  2369. skb_checksum(skb, len, delta, 0),
  2370. len);
  2371. } else if (skb->ip_summed == CHECKSUM_PARTIAL) {
  2372. int hdlen = (len > skb_headlen(skb)) ? skb_headlen(skb) : len;
  2373. int offset = skb_checksum_start_offset(skb) + skb->csum_offset;
  2374. if (offset + sizeof(__sum16) > hdlen)
  2375. return -EINVAL;
  2376. }
  2377. return __pskb_trim(skb, len);
  2378. }
  2379. EXPORT_SYMBOL(pskb_trim_rcsum_slow);
  2380. /**
  2381. * __pskb_pull_tail - advance tail of skb header
  2382. * @skb: buffer to reallocate
  2383. * @delta: number of bytes to advance tail
  2384. *
  2385. * The function makes a sense only on a fragmented &sk_buff,
  2386. * it expands header moving its tail forward and copying necessary
  2387. * data from fragmented part.
  2388. *
  2389. * &sk_buff MUST have reference count of 1.
  2390. *
  2391. * Returns %NULL (and &sk_buff does not change) if pull failed
  2392. * or value of new tail of skb in the case of success.
  2393. *
  2394. * All the pointers pointing into skb header may change and must be
  2395. * reloaded after call to this function.
  2396. */
  2397. /* Moves tail of skb head forward, copying data from fragmented part,
  2398. * when it is necessary.
  2399. * 1. It may fail due to malloc failure.
  2400. * 2. It may change skb pointers.
  2401. *
  2402. * It is pretty complicated. Luckily, it is called only in exceptional cases.
  2403. */
  2404. void *__pskb_pull_tail(struct sk_buff *skb, int delta)
  2405. {
  2406. /* If skb has not enough free space at tail, get new one
  2407. * plus 128 bytes for future expansions. If we have enough
  2408. * room at tail, reallocate without expansion only if skb is cloned.
  2409. */
  2410. int i, k, eat = (skb->tail + delta) - skb->end;
  2411. if (!skb_frags_readable(skb))
  2412. return NULL;
  2413. if (eat > 0 || skb_cloned(skb)) {
  2414. if (pskb_expand_head(skb, 0, eat > 0 ? eat + 128 : 0,
  2415. GFP_ATOMIC))
  2416. return NULL;
  2417. }
  2418. BUG_ON(skb_copy_bits(skb, skb_headlen(skb),
  2419. skb_tail_pointer(skb), delta));
  2420. /* Optimization: no fragments, no reasons to preestimate
  2421. * size of pulled pages. Superb.
  2422. */
  2423. if (!skb_has_frag_list(skb))
  2424. goto pull_pages;
  2425. /* Estimate size of pulled pages. */
  2426. eat = delta;
  2427. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  2428. int size = skb_frag_size(&skb_shinfo(skb)->frags[i]);
  2429. if (size >= eat)
  2430. goto pull_pages;
  2431. eat -= size;
  2432. }
  2433. /* If we need update frag list, we are in troubles.
  2434. * Certainly, it is possible to add an offset to skb data,
  2435. * but taking into account that pulling is expected to
  2436. * be very rare operation, it is worth to fight against
  2437. * further bloating skb head and crucify ourselves here instead.
  2438. * Pure masohism, indeed. 8)8)
  2439. */
  2440. if (eat) {
  2441. struct sk_buff *list = skb_shinfo(skb)->frag_list;
  2442. struct sk_buff *clone = NULL;
  2443. struct sk_buff *insp = NULL;
  2444. do {
  2445. if (list->len <= eat) {
  2446. /* Eaten as whole. */
  2447. eat -= list->len;
  2448. list = list->next;
  2449. insp = list;
  2450. } else {
  2451. /* Eaten partially. */
  2452. if (skb_is_gso(skb) && !list->head_frag &&
  2453. skb_headlen(list))
  2454. skb_shinfo(skb)->gso_type |= SKB_GSO_DODGY;
  2455. if (skb_shared(list)) {
  2456. /* Sucks! We need to fork list. :-( */
  2457. clone = skb_clone(list, GFP_ATOMIC);
  2458. if (!clone)
  2459. return NULL;
  2460. insp = list->next;
  2461. list = clone;
  2462. } else {
  2463. /* This may be pulled without
  2464. * problems. */
  2465. insp = list;
  2466. }
  2467. if (!pskb_pull(list, eat)) {
  2468. kfree_skb(clone);
  2469. return NULL;
  2470. }
  2471. break;
  2472. }
  2473. } while (eat);
  2474. /* Free pulled out fragments. */
  2475. while ((list = skb_shinfo(skb)->frag_list) != insp) {
  2476. skb_shinfo(skb)->frag_list = list->next;
  2477. consume_skb(list);
  2478. }
  2479. /* And insert new clone at head. */
  2480. if (clone) {
  2481. clone->next = list;
  2482. skb_shinfo(skb)->frag_list = clone;
  2483. }
  2484. }
  2485. /* Success! Now we may commit changes to skb data. */
  2486. pull_pages:
  2487. eat = delta;
  2488. k = 0;
  2489. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  2490. int size = skb_frag_size(&skb_shinfo(skb)->frags[i]);
  2491. if (size <= eat) {
  2492. skb_frag_unref(skb, i);
  2493. eat -= size;
  2494. } else {
  2495. skb_frag_t *frag = &skb_shinfo(skb)->frags[k];
  2496. *frag = skb_shinfo(skb)->frags[i];
  2497. if (eat) {
  2498. skb_frag_off_add(frag, eat);
  2499. skb_frag_size_sub(frag, eat);
  2500. if (!i)
  2501. goto end;
  2502. eat = 0;
  2503. }
  2504. k++;
  2505. }
  2506. }
  2507. skb_shinfo(skb)->nr_frags = k;
  2508. end:
  2509. skb->tail += delta;
  2510. skb->data_len -= delta;
  2511. if (!skb->data_len)
  2512. skb_zcopy_clear(skb, false);
  2513. return skb_tail_pointer(skb);
  2514. }
  2515. EXPORT_SYMBOL(__pskb_pull_tail);
  2516. /**
  2517. * skb_copy_bits - copy bits from skb to kernel buffer
  2518. * @skb: source skb
  2519. * @offset: offset in source
  2520. * @to: destination buffer
  2521. * @len: number of bytes to copy
  2522. *
  2523. * Copy the specified number of bytes from the source skb to the
  2524. * destination buffer.
  2525. *
  2526. * CAUTION ! :
  2527. * If its prototype is ever changed,
  2528. * check arch/{*}/net/{*}.S files,
  2529. * since it is called from BPF assembly code.
  2530. */
  2531. int skb_copy_bits(const struct sk_buff *skb, int offset, void *to, int len)
  2532. {
  2533. int start = skb_headlen(skb);
  2534. struct sk_buff *frag_iter;
  2535. int i, copy;
  2536. if (offset > (int)skb->len - len)
  2537. goto fault;
  2538. /* Copy header. */
  2539. if ((copy = start - offset) > 0) {
  2540. if (copy > len)
  2541. copy = len;
  2542. skb_copy_from_linear_data_offset(skb, offset, to, copy);
  2543. if ((len -= copy) == 0)
  2544. return 0;
  2545. offset += copy;
  2546. to += copy;
  2547. }
  2548. if (!skb_frags_readable(skb))
  2549. goto fault;
  2550. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  2551. int end;
  2552. skb_frag_t *f = &skb_shinfo(skb)->frags[i];
  2553. WARN_ON(start > offset + len);
  2554. end = start + skb_frag_size(f);
  2555. if ((copy = end - offset) > 0) {
  2556. u32 p_off, p_len, copied;
  2557. struct page *p;
  2558. u8 *vaddr;
  2559. if (copy > len)
  2560. copy = len;
  2561. skb_frag_foreach_page(f,
  2562. skb_frag_off(f) + offset - start,
  2563. copy, p, p_off, p_len, copied) {
  2564. vaddr = kmap_atomic(p);
  2565. memcpy(to + copied, vaddr + p_off, p_len);
  2566. kunmap_atomic(vaddr);
  2567. }
  2568. if ((len -= copy) == 0)
  2569. return 0;
  2570. offset += copy;
  2571. to += copy;
  2572. }
  2573. start = end;
  2574. }
  2575. skb_walk_frags(skb, frag_iter) {
  2576. int end;
  2577. WARN_ON(start > offset + len);
  2578. end = start + frag_iter->len;
  2579. if ((copy = end - offset) > 0) {
  2580. if (copy > len)
  2581. copy = len;
  2582. if (skb_copy_bits(frag_iter, offset - start, to, copy))
  2583. goto fault;
  2584. if ((len -= copy) == 0)
  2585. return 0;
  2586. offset += copy;
  2587. to += copy;
  2588. }
  2589. start = end;
  2590. }
  2591. if (!len)
  2592. return 0;
  2593. fault:
  2594. return -EFAULT;
  2595. }
  2596. EXPORT_SYMBOL(skb_copy_bits);
  2597. /*
  2598. * Callback from splice_to_pipe(), if we need to release some pages
  2599. * at the end of the spd in case we error'ed out in filling the pipe.
  2600. */
  2601. static void sock_spd_release(struct splice_pipe_desc *spd, unsigned int i)
  2602. {
  2603. put_page(spd->pages[i]);
  2604. }
  2605. static struct page *linear_to_page(struct page *page, unsigned int *len,
  2606. unsigned int *offset,
  2607. struct sock *sk)
  2608. {
  2609. struct page_frag *pfrag = sk_page_frag(sk);
  2610. if (!sk_page_frag_refill(sk, pfrag))
  2611. return NULL;
  2612. *len = min_t(unsigned int, *len, pfrag->size - pfrag->offset);
  2613. memcpy(page_address(pfrag->page) + pfrag->offset,
  2614. page_address(page) + *offset, *len);
  2615. *offset = pfrag->offset;
  2616. pfrag->offset += *len;
  2617. return pfrag->page;
  2618. }
  2619. static bool spd_can_coalesce(const struct splice_pipe_desc *spd,
  2620. struct page *page,
  2621. unsigned int offset)
  2622. {
  2623. return spd->nr_pages &&
  2624. spd->pages[spd->nr_pages - 1] == page &&
  2625. (spd->partial[spd->nr_pages - 1].offset +
  2626. spd->partial[spd->nr_pages - 1].len == offset);
  2627. }
  2628. /*
  2629. * Fill page/offset/length into spd, if it can hold more pages.
  2630. */
  2631. static bool spd_fill_page(struct splice_pipe_desc *spd,
  2632. struct pipe_inode_info *pipe, struct page *page,
  2633. unsigned int *len, unsigned int offset,
  2634. bool linear,
  2635. struct sock *sk)
  2636. {
  2637. if (unlikely(spd->nr_pages == MAX_SKB_FRAGS))
  2638. return true;
  2639. if (linear) {
  2640. page = linear_to_page(page, len, &offset, sk);
  2641. if (!page)
  2642. return true;
  2643. }
  2644. if (spd_can_coalesce(spd, page, offset)) {
  2645. spd->partial[spd->nr_pages - 1].len += *len;
  2646. return false;
  2647. }
  2648. get_page(page);
  2649. spd->pages[spd->nr_pages] = page;
  2650. spd->partial[spd->nr_pages].len = *len;
  2651. spd->partial[spd->nr_pages].offset = offset;
  2652. spd->nr_pages++;
  2653. return false;
  2654. }
  2655. static bool __splice_segment(struct page *page, unsigned int poff,
  2656. unsigned int plen, unsigned int *off,
  2657. unsigned int *len,
  2658. struct splice_pipe_desc *spd, bool linear,
  2659. struct sock *sk,
  2660. struct pipe_inode_info *pipe)
  2661. {
  2662. if (!*len)
  2663. return true;
  2664. /* skip this segment if already processed */
  2665. if (*off >= plen) {
  2666. *off -= plen;
  2667. return false;
  2668. }
  2669. /* ignore any bits we already processed */
  2670. poff += *off;
  2671. plen -= *off;
  2672. *off = 0;
  2673. do {
  2674. unsigned int flen = min(*len, plen);
  2675. if (spd_fill_page(spd, pipe, page, &flen, poff,
  2676. linear, sk))
  2677. return true;
  2678. poff += flen;
  2679. plen -= flen;
  2680. *len -= flen;
  2681. } while (*len && plen);
  2682. return false;
  2683. }
  2684. /*
  2685. * Map linear and fragment data from the skb to spd. It reports true if the
  2686. * pipe is full or if we already spliced the requested length.
  2687. */
  2688. static bool __skb_splice_bits(struct sk_buff *skb, struct pipe_inode_info *pipe,
  2689. unsigned int *offset, unsigned int *len,
  2690. struct splice_pipe_desc *spd, struct sock *sk)
  2691. {
  2692. int seg;
  2693. struct sk_buff *iter;
  2694. /* map the linear part :
  2695. * If skb->head_frag is set, this 'linear' part is backed by a
  2696. * fragment, and if the head is not shared with any clones then
  2697. * we can avoid a copy since we own the head portion of this page.
  2698. */
  2699. if (__splice_segment(virt_to_page(skb->data),
  2700. (unsigned long) skb->data & (PAGE_SIZE - 1),
  2701. skb_headlen(skb),
  2702. offset, len, spd,
  2703. skb_head_is_locked(skb),
  2704. sk, pipe))
  2705. return true;
  2706. /*
  2707. * then map the fragments
  2708. */
  2709. if (!skb_frags_readable(skb))
  2710. return false;
  2711. for (seg = 0; seg < skb_shinfo(skb)->nr_frags; seg++) {
  2712. const skb_frag_t *f = &skb_shinfo(skb)->frags[seg];
  2713. if (WARN_ON_ONCE(!skb_frag_page(f)))
  2714. return false;
  2715. if (__splice_segment(skb_frag_page(f),
  2716. skb_frag_off(f), skb_frag_size(f),
  2717. offset, len, spd, false, sk, pipe))
  2718. return true;
  2719. }
  2720. skb_walk_frags(skb, iter) {
  2721. if (*offset >= iter->len) {
  2722. *offset -= iter->len;
  2723. continue;
  2724. }
  2725. /* __skb_splice_bits() only fails if the output has no room
  2726. * left, so no point in going over the frag_list for the error
  2727. * case.
  2728. */
  2729. if (__skb_splice_bits(iter, pipe, offset, len, spd, sk))
  2730. return true;
  2731. }
  2732. return false;
  2733. }
  2734. /*
  2735. * Map data from the skb to a pipe. Should handle both the linear part,
  2736. * the fragments, and the frag list.
  2737. */
  2738. int skb_splice_bits(struct sk_buff *skb, struct sock *sk, unsigned int offset,
  2739. struct pipe_inode_info *pipe, unsigned int tlen,
  2740. unsigned int flags)
  2741. {
  2742. struct partial_page partial[MAX_SKB_FRAGS];
  2743. struct page *pages[MAX_SKB_FRAGS];
  2744. struct splice_pipe_desc spd = {
  2745. .pages = pages,
  2746. .partial = partial,
  2747. .nr_pages_max = MAX_SKB_FRAGS,
  2748. .ops = &nosteal_pipe_buf_ops,
  2749. .spd_release = sock_spd_release,
  2750. };
  2751. int ret = 0;
  2752. __skb_splice_bits(skb, pipe, &offset, &tlen, &spd, sk);
  2753. if (spd.nr_pages)
  2754. ret = splice_to_pipe(pipe, &spd);
  2755. return ret;
  2756. }
  2757. EXPORT_SYMBOL_GPL(skb_splice_bits);
  2758. static int sendmsg_locked(struct sock *sk, struct msghdr *msg)
  2759. {
  2760. struct socket *sock = sk->sk_socket;
  2761. size_t size = msg_data_left(msg);
  2762. if (!sock)
  2763. return -EINVAL;
  2764. if (!sock->ops->sendmsg_locked)
  2765. return sock_no_sendmsg_locked(sk, msg, size);
  2766. return sock->ops->sendmsg_locked(sk, msg, size);
  2767. }
  2768. static int sendmsg_unlocked(struct sock *sk, struct msghdr *msg)
  2769. {
  2770. struct socket *sock = sk->sk_socket;
  2771. if (!sock)
  2772. return -EINVAL;
  2773. return sock_sendmsg(sock, msg);
  2774. }
  2775. typedef int (*sendmsg_func)(struct sock *sk, struct msghdr *msg);
  2776. static int __skb_send_sock(struct sock *sk, struct sk_buff *skb, int offset,
  2777. int len, sendmsg_func sendmsg)
  2778. {
  2779. unsigned int orig_len = len;
  2780. struct sk_buff *head = skb;
  2781. unsigned short fragidx;
  2782. int slen, ret;
  2783. do_frag_list:
  2784. /* Deal with head data */
  2785. while (offset < skb_headlen(skb) && len) {
  2786. struct kvec kv;
  2787. struct msghdr msg;
  2788. slen = min_t(int, len, skb_headlen(skb) - offset);
  2789. kv.iov_base = skb->data + offset;
  2790. kv.iov_len = slen;
  2791. memset(&msg, 0, sizeof(msg));
  2792. msg.msg_flags = MSG_DONTWAIT;
  2793. iov_iter_kvec(&msg.msg_iter, ITER_SOURCE, &kv, 1, slen);
  2794. ret = INDIRECT_CALL_2(sendmsg, sendmsg_locked,
  2795. sendmsg_unlocked, sk, &msg);
  2796. if (ret <= 0)
  2797. goto error;
  2798. offset += ret;
  2799. len -= ret;
  2800. }
  2801. /* All the data was skb head? */
  2802. if (!len)
  2803. goto out;
  2804. /* Make offset relative to start of frags */
  2805. offset -= skb_headlen(skb);
  2806. /* Find where we are in frag list */
  2807. for (fragidx = 0; fragidx < skb_shinfo(skb)->nr_frags; fragidx++) {
  2808. skb_frag_t *frag = &skb_shinfo(skb)->frags[fragidx];
  2809. if (offset < skb_frag_size(frag))
  2810. break;
  2811. offset -= skb_frag_size(frag);
  2812. }
  2813. for (; len && fragidx < skb_shinfo(skb)->nr_frags; fragidx++) {
  2814. skb_frag_t *frag = &skb_shinfo(skb)->frags[fragidx];
  2815. slen = min_t(size_t, len, skb_frag_size(frag) - offset);
  2816. while (slen) {
  2817. struct bio_vec bvec;
  2818. struct msghdr msg = {
  2819. .msg_flags = MSG_SPLICE_PAGES | MSG_DONTWAIT,
  2820. };
  2821. bvec_set_page(&bvec, skb_frag_page(frag), slen,
  2822. skb_frag_off(frag) + offset);
  2823. iov_iter_bvec(&msg.msg_iter, ITER_SOURCE, &bvec, 1,
  2824. slen);
  2825. ret = INDIRECT_CALL_2(sendmsg, sendmsg_locked,
  2826. sendmsg_unlocked, sk, &msg);
  2827. if (ret <= 0)
  2828. goto error;
  2829. len -= ret;
  2830. offset += ret;
  2831. slen -= ret;
  2832. }
  2833. offset = 0;
  2834. }
  2835. if (len) {
  2836. /* Process any frag lists */
  2837. if (skb == head) {
  2838. if (skb_has_frag_list(skb)) {
  2839. skb = skb_shinfo(skb)->frag_list;
  2840. goto do_frag_list;
  2841. }
  2842. } else if (skb->next) {
  2843. skb = skb->next;
  2844. goto do_frag_list;
  2845. }
  2846. }
  2847. out:
  2848. return orig_len - len;
  2849. error:
  2850. return orig_len == len ? ret : orig_len - len;
  2851. }
  2852. /* Send skb data on a socket. Socket must be locked. */
  2853. int skb_send_sock_locked(struct sock *sk, struct sk_buff *skb, int offset,
  2854. int len)
  2855. {
  2856. return __skb_send_sock(sk, skb, offset, len, sendmsg_locked);
  2857. }
  2858. EXPORT_SYMBOL_GPL(skb_send_sock_locked);
  2859. /* Send skb data on a socket. Socket must be unlocked. */
  2860. int skb_send_sock(struct sock *sk, struct sk_buff *skb, int offset, int len)
  2861. {
  2862. return __skb_send_sock(sk, skb, offset, len, sendmsg_unlocked);
  2863. }
  2864. /**
  2865. * skb_store_bits - store bits from kernel buffer to skb
  2866. * @skb: destination buffer
  2867. * @offset: offset in destination
  2868. * @from: source buffer
  2869. * @len: number of bytes to copy
  2870. *
  2871. * Copy the specified number of bytes from the source buffer to the
  2872. * destination skb. This function handles all the messy bits of
  2873. * traversing fragment lists and such.
  2874. */
  2875. int skb_store_bits(struct sk_buff *skb, int offset, const void *from, int len)
  2876. {
  2877. int start = skb_headlen(skb);
  2878. struct sk_buff *frag_iter;
  2879. int i, copy;
  2880. if (offset > (int)skb->len - len)
  2881. goto fault;
  2882. if ((copy = start - offset) > 0) {
  2883. if (copy > len)
  2884. copy = len;
  2885. skb_copy_to_linear_data_offset(skb, offset, from, copy);
  2886. if ((len -= copy) == 0)
  2887. return 0;
  2888. offset += copy;
  2889. from += copy;
  2890. }
  2891. if (!skb_frags_readable(skb))
  2892. goto fault;
  2893. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  2894. skb_frag_t *frag = &skb_shinfo(skb)->frags[i];
  2895. int end;
  2896. WARN_ON(start > offset + len);
  2897. end = start + skb_frag_size(frag);
  2898. if ((copy = end - offset) > 0) {
  2899. u32 p_off, p_len, copied;
  2900. struct page *p;
  2901. u8 *vaddr;
  2902. if (copy > len)
  2903. copy = len;
  2904. skb_frag_foreach_page(frag,
  2905. skb_frag_off(frag) + offset - start,
  2906. copy, p, p_off, p_len, copied) {
  2907. vaddr = kmap_atomic(p);
  2908. memcpy(vaddr + p_off, from + copied, p_len);
  2909. kunmap_atomic(vaddr);
  2910. }
  2911. if ((len -= copy) == 0)
  2912. return 0;
  2913. offset += copy;
  2914. from += copy;
  2915. }
  2916. start = end;
  2917. }
  2918. skb_walk_frags(skb, frag_iter) {
  2919. int end;
  2920. WARN_ON(start > offset + len);
  2921. end = start + frag_iter->len;
  2922. if ((copy = end - offset) > 0) {
  2923. if (copy > len)
  2924. copy = len;
  2925. if (skb_store_bits(frag_iter, offset - start,
  2926. from, copy))
  2927. goto fault;
  2928. if ((len -= copy) == 0)
  2929. return 0;
  2930. offset += copy;
  2931. from += copy;
  2932. }
  2933. start = end;
  2934. }
  2935. if (!len)
  2936. return 0;
  2937. fault:
  2938. return -EFAULT;
  2939. }
  2940. EXPORT_SYMBOL(skb_store_bits);
  2941. /* Checksum skb data. */
  2942. __wsum __skb_checksum(const struct sk_buff *skb, int offset, int len,
  2943. __wsum csum, const struct skb_checksum_ops *ops)
  2944. {
  2945. int start = skb_headlen(skb);
  2946. int i, copy = start - offset;
  2947. struct sk_buff *frag_iter;
  2948. int pos = 0;
  2949. /* Checksum header. */
  2950. if (copy > 0) {
  2951. if (copy > len)
  2952. copy = len;
  2953. csum = INDIRECT_CALL_1(ops->update, csum_partial_ext,
  2954. skb->data + offset, copy, csum);
  2955. if ((len -= copy) == 0)
  2956. return csum;
  2957. offset += copy;
  2958. pos = copy;
  2959. }
  2960. if (WARN_ON_ONCE(!skb_frags_readable(skb)))
  2961. return 0;
  2962. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  2963. int end;
  2964. skb_frag_t *frag = &skb_shinfo(skb)->frags[i];
  2965. WARN_ON(start > offset + len);
  2966. end = start + skb_frag_size(frag);
  2967. if ((copy = end - offset) > 0) {
  2968. u32 p_off, p_len, copied;
  2969. struct page *p;
  2970. __wsum csum2;
  2971. u8 *vaddr;
  2972. if (copy > len)
  2973. copy = len;
  2974. skb_frag_foreach_page(frag,
  2975. skb_frag_off(frag) + offset - start,
  2976. copy, p, p_off, p_len, copied) {
  2977. vaddr = kmap_atomic(p);
  2978. csum2 = INDIRECT_CALL_1(ops->update,
  2979. csum_partial_ext,
  2980. vaddr + p_off, p_len, 0);
  2981. kunmap_atomic(vaddr);
  2982. csum = INDIRECT_CALL_1(ops->combine,
  2983. csum_block_add_ext, csum,
  2984. csum2, pos, p_len);
  2985. pos += p_len;
  2986. }
  2987. if (!(len -= copy))
  2988. return csum;
  2989. offset += copy;
  2990. }
  2991. start = end;
  2992. }
  2993. skb_walk_frags(skb, frag_iter) {
  2994. int end;
  2995. WARN_ON(start > offset + len);
  2996. end = start + frag_iter->len;
  2997. if ((copy = end - offset) > 0) {
  2998. __wsum csum2;
  2999. if (copy > len)
  3000. copy = len;
  3001. csum2 = __skb_checksum(frag_iter, offset - start,
  3002. copy, 0, ops);
  3003. csum = INDIRECT_CALL_1(ops->combine, csum_block_add_ext,
  3004. csum, csum2, pos, copy);
  3005. if ((len -= copy) == 0)
  3006. return csum;
  3007. offset += copy;
  3008. pos += copy;
  3009. }
  3010. start = end;
  3011. }
  3012. BUG_ON(len);
  3013. return csum;
  3014. }
  3015. EXPORT_SYMBOL(__skb_checksum);
  3016. __wsum skb_checksum(const struct sk_buff *skb, int offset,
  3017. int len, __wsum csum)
  3018. {
  3019. const struct skb_checksum_ops ops = {
  3020. .update = csum_partial_ext,
  3021. .combine = csum_block_add_ext,
  3022. };
  3023. return __skb_checksum(skb, offset, len, csum, &ops);
  3024. }
  3025. EXPORT_SYMBOL(skb_checksum);
  3026. /* Both of above in one bottle. */
  3027. __wsum skb_copy_and_csum_bits(const struct sk_buff *skb, int offset,
  3028. u8 *to, int len)
  3029. {
  3030. int start = skb_headlen(skb);
  3031. int i, copy = start - offset;
  3032. struct sk_buff *frag_iter;
  3033. int pos = 0;
  3034. __wsum csum = 0;
  3035. /* Copy header. */
  3036. if (copy > 0) {
  3037. if (copy > len)
  3038. copy = len;
  3039. csum = csum_partial_copy_nocheck(skb->data + offset, to,
  3040. copy);
  3041. if ((len -= copy) == 0)
  3042. return csum;
  3043. offset += copy;
  3044. to += copy;
  3045. pos = copy;
  3046. }
  3047. if (!skb_frags_readable(skb))
  3048. return 0;
  3049. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  3050. int end;
  3051. WARN_ON(start > offset + len);
  3052. end = start + skb_frag_size(&skb_shinfo(skb)->frags[i]);
  3053. if ((copy = end - offset) > 0) {
  3054. skb_frag_t *frag = &skb_shinfo(skb)->frags[i];
  3055. u32 p_off, p_len, copied;
  3056. struct page *p;
  3057. __wsum csum2;
  3058. u8 *vaddr;
  3059. if (copy > len)
  3060. copy = len;
  3061. skb_frag_foreach_page(frag,
  3062. skb_frag_off(frag) + offset - start,
  3063. copy, p, p_off, p_len, copied) {
  3064. vaddr = kmap_atomic(p);
  3065. csum2 = csum_partial_copy_nocheck(vaddr + p_off,
  3066. to + copied,
  3067. p_len);
  3068. kunmap_atomic(vaddr);
  3069. csum = csum_block_add(csum, csum2, pos);
  3070. pos += p_len;
  3071. }
  3072. if (!(len -= copy))
  3073. return csum;
  3074. offset += copy;
  3075. to += copy;
  3076. }
  3077. start = end;
  3078. }
  3079. skb_walk_frags(skb, frag_iter) {
  3080. __wsum csum2;
  3081. int end;
  3082. WARN_ON(start > offset + len);
  3083. end = start + frag_iter->len;
  3084. if ((copy = end - offset) > 0) {
  3085. if (copy > len)
  3086. copy = len;
  3087. csum2 = skb_copy_and_csum_bits(frag_iter,
  3088. offset - start,
  3089. to, copy);
  3090. csum = csum_block_add(csum, csum2, pos);
  3091. if ((len -= copy) == 0)
  3092. return csum;
  3093. offset += copy;
  3094. to += copy;
  3095. pos += copy;
  3096. }
  3097. start = end;
  3098. }
  3099. BUG_ON(len);
  3100. return csum;
  3101. }
  3102. EXPORT_SYMBOL(skb_copy_and_csum_bits);
  3103. __sum16 __skb_checksum_complete_head(struct sk_buff *skb, int len)
  3104. {
  3105. __sum16 sum;
  3106. sum = csum_fold(skb_checksum(skb, 0, len, skb->csum));
  3107. /* See comments in __skb_checksum_complete(). */
  3108. if (likely(!sum)) {
  3109. if (unlikely(skb->ip_summed == CHECKSUM_COMPLETE) &&
  3110. !skb->csum_complete_sw)
  3111. netdev_rx_csum_fault(skb->dev, skb);
  3112. }
  3113. if (!skb_shared(skb))
  3114. skb->csum_valid = !sum;
  3115. return sum;
  3116. }
  3117. EXPORT_SYMBOL(__skb_checksum_complete_head);
  3118. /* This function assumes skb->csum already holds pseudo header's checksum,
  3119. * which has been changed from the hardware checksum, for example, by
  3120. * __skb_checksum_validate_complete(). And, the original skb->csum must
  3121. * have been validated unsuccessfully for CHECKSUM_COMPLETE case.
  3122. *
  3123. * It returns non-zero if the recomputed checksum is still invalid, otherwise
  3124. * zero. The new checksum is stored back into skb->csum unless the skb is
  3125. * shared.
  3126. */
  3127. __sum16 __skb_checksum_complete(struct sk_buff *skb)
  3128. {
  3129. __wsum csum;
  3130. __sum16 sum;
  3131. csum = skb_checksum(skb, 0, skb->len, 0);
  3132. sum = csum_fold(csum_add(skb->csum, csum));
  3133. /* This check is inverted, because we already knew the hardware
  3134. * checksum is invalid before calling this function. So, if the
  3135. * re-computed checksum is valid instead, then we have a mismatch
  3136. * between the original skb->csum and skb_checksum(). This means either
  3137. * the original hardware checksum is incorrect or we screw up skb->csum
  3138. * when moving skb->data around.
  3139. */
  3140. if (likely(!sum)) {
  3141. if (unlikely(skb->ip_summed == CHECKSUM_COMPLETE) &&
  3142. !skb->csum_complete_sw)
  3143. netdev_rx_csum_fault(skb->dev, skb);
  3144. }
  3145. if (!skb_shared(skb)) {
  3146. /* Save full packet checksum */
  3147. skb->csum = csum;
  3148. skb->ip_summed = CHECKSUM_COMPLETE;
  3149. skb->csum_complete_sw = 1;
  3150. skb->csum_valid = !sum;
  3151. }
  3152. return sum;
  3153. }
  3154. EXPORT_SYMBOL(__skb_checksum_complete);
  3155. static __wsum warn_crc32c_csum_update(const void *buff, int len, __wsum sum)
  3156. {
  3157. net_warn_ratelimited(
  3158. "%s: attempt to compute crc32c without libcrc32c.ko\n",
  3159. __func__);
  3160. return 0;
  3161. }
  3162. static __wsum warn_crc32c_csum_combine(__wsum csum, __wsum csum2,
  3163. int offset, int len)
  3164. {
  3165. net_warn_ratelimited(
  3166. "%s: attempt to compute crc32c without libcrc32c.ko\n",
  3167. __func__);
  3168. return 0;
  3169. }
  3170. static const struct skb_checksum_ops default_crc32c_ops = {
  3171. .update = warn_crc32c_csum_update,
  3172. .combine = warn_crc32c_csum_combine,
  3173. };
  3174. const struct skb_checksum_ops *crc32c_csum_stub __read_mostly =
  3175. &default_crc32c_ops;
  3176. EXPORT_SYMBOL(crc32c_csum_stub);
  3177. /**
  3178. * skb_zerocopy_headlen - Calculate headroom needed for skb_zerocopy()
  3179. * @from: source buffer
  3180. *
  3181. * Calculates the amount of linear headroom needed in the 'to' skb passed
  3182. * into skb_zerocopy().
  3183. */
  3184. unsigned int
  3185. skb_zerocopy_headlen(const struct sk_buff *from)
  3186. {
  3187. unsigned int hlen = 0;
  3188. if (!from->head_frag ||
  3189. skb_headlen(from) < L1_CACHE_BYTES ||
  3190. skb_shinfo(from)->nr_frags >= MAX_SKB_FRAGS) {
  3191. hlen = skb_headlen(from);
  3192. if (!hlen)
  3193. hlen = from->len;
  3194. }
  3195. if (skb_has_frag_list(from))
  3196. hlen = from->len;
  3197. return hlen;
  3198. }
  3199. EXPORT_SYMBOL_GPL(skb_zerocopy_headlen);
  3200. /**
  3201. * skb_zerocopy - Zero copy skb to skb
  3202. * @to: destination buffer
  3203. * @from: source buffer
  3204. * @len: number of bytes to copy from source buffer
  3205. * @hlen: size of linear headroom in destination buffer
  3206. *
  3207. * Copies up to `len` bytes from `from` to `to` by creating references
  3208. * to the frags in the source buffer.
  3209. *
  3210. * The `hlen` as calculated by skb_zerocopy_headlen() specifies the
  3211. * headroom in the `to` buffer.
  3212. *
  3213. * Return value:
  3214. * 0: everything is OK
  3215. * -ENOMEM: couldn't orphan frags of @from due to lack of memory
  3216. * -EFAULT: skb_copy_bits() found some problem with skb geometry
  3217. */
  3218. int
  3219. skb_zerocopy(struct sk_buff *to, struct sk_buff *from, int len, int hlen)
  3220. {
  3221. int i, j = 0;
  3222. int plen = 0; /* length of skb->head fragment */
  3223. int ret;
  3224. struct page *page;
  3225. unsigned int offset;
  3226. BUG_ON(!from->head_frag && !hlen);
  3227. /* dont bother with small payloads */
  3228. if (len <= skb_tailroom(to))
  3229. return skb_copy_bits(from, 0, skb_put(to, len), len);
  3230. if (hlen) {
  3231. ret = skb_copy_bits(from, 0, skb_put(to, hlen), hlen);
  3232. if (unlikely(ret))
  3233. return ret;
  3234. len -= hlen;
  3235. } else {
  3236. plen = min_t(int, skb_headlen(from), len);
  3237. if (plen) {
  3238. page = virt_to_head_page(from->head);
  3239. offset = from->data - (unsigned char *)page_address(page);
  3240. __skb_fill_netmem_desc(to, 0, page_to_netmem(page),
  3241. offset, plen);
  3242. get_page(page);
  3243. j = 1;
  3244. len -= plen;
  3245. }
  3246. }
  3247. skb_len_add(to, len + plen);
  3248. if (unlikely(skb_orphan_frags(from, GFP_ATOMIC))) {
  3249. skb_tx_error(from);
  3250. return -ENOMEM;
  3251. }
  3252. skb_zerocopy_clone(to, from, GFP_ATOMIC);
  3253. for (i = 0; i < skb_shinfo(from)->nr_frags; i++) {
  3254. int size;
  3255. if (!len)
  3256. break;
  3257. skb_shinfo(to)->frags[j] = skb_shinfo(from)->frags[i];
  3258. size = min_t(int, skb_frag_size(&skb_shinfo(to)->frags[j]),
  3259. len);
  3260. skb_frag_size_set(&skb_shinfo(to)->frags[j], size);
  3261. len -= size;
  3262. skb_frag_ref(to, j);
  3263. j++;
  3264. }
  3265. skb_shinfo(to)->nr_frags = j;
  3266. return 0;
  3267. }
  3268. EXPORT_SYMBOL_GPL(skb_zerocopy);
  3269. void skb_copy_and_csum_dev(const struct sk_buff *skb, u8 *to)
  3270. {
  3271. __wsum csum;
  3272. long csstart;
  3273. if (skb->ip_summed == CHECKSUM_PARTIAL)
  3274. csstart = skb_checksum_start_offset(skb);
  3275. else
  3276. csstart = skb_headlen(skb);
  3277. BUG_ON(csstart > skb_headlen(skb));
  3278. skb_copy_from_linear_data(skb, to, csstart);
  3279. csum = 0;
  3280. if (csstart != skb->len)
  3281. csum = skb_copy_and_csum_bits(skb, csstart, to + csstart,
  3282. skb->len - csstart);
  3283. if (skb->ip_summed == CHECKSUM_PARTIAL) {
  3284. long csstuff = csstart + skb->csum_offset;
  3285. *((__sum16 *)(to + csstuff)) = csum_fold(csum);
  3286. }
  3287. }
  3288. EXPORT_SYMBOL(skb_copy_and_csum_dev);
  3289. /**
  3290. * skb_dequeue - remove from the head of the queue
  3291. * @list: list to dequeue from
  3292. *
  3293. * Remove the head of the list. The list lock is taken so the function
  3294. * may be used safely with other locking list functions. The head item is
  3295. * returned or %NULL if the list is empty.
  3296. */
  3297. struct sk_buff *skb_dequeue(struct sk_buff_head *list)
  3298. {
  3299. unsigned long flags;
  3300. struct sk_buff *result;
  3301. spin_lock_irqsave(&list->lock, flags);
  3302. result = __skb_dequeue(list);
  3303. spin_unlock_irqrestore(&list->lock, flags);
  3304. return result;
  3305. }
  3306. EXPORT_SYMBOL(skb_dequeue);
  3307. /**
  3308. * skb_dequeue_tail - remove from the tail of the queue
  3309. * @list: list to dequeue from
  3310. *
  3311. * Remove the tail of the list. The list lock is taken so the function
  3312. * may be used safely with other locking list functions. The tail item is
  3313. * returned or %NULL if the list is empty.
  3314. */
  3315. struct sk_buff *skb_dequeue_tail(struct sk_buff_head *list)
  3316. {
  3317. unsigned long flags;
  3318. struct sk_buff *result;
  3319. spin_lock_irqsave(&list->lock, flags);
  3320. result = __skb_dequeue_tail(list);
  3321. spin_unlock_irqrestore(&list->lock, flags);
  3322. return result;
  3323. }
  3324. EXPORT_SYMBOL(skb_dequeue_tail);
  3325. /**
  3326. * skb_queue_purge_reason - empty a list
  3327. * @list: list to empty
  3328. * @reason: drop reason
  3329. *
  3330. * Delete all buffers on an &sk_buff list. Each buffer is removed from
  3331. * the list and one reference dropped. This function takes the list
  3332. * lock and is atomic with respect to other list locking functions.
  3333. */
  3334. void skb_queue_purge_reason(struct sk_buff_head *list,
  3335. enum skb_drop_reason reason)
  3336. {
  3337. struct sk_buff_head tmp;
  3338. unsigned long flags;
  3339. if (skb_queue_empty_lockless(list))
  3340. return;
  3341. __skb_queue_head_init(&tmp);
  3342. spin_lock_irqsave(&list->lock, flags);
  3343. skb_queue_splice_init(list, &tmp);
  3344. spin_unlock_irqrestore(&list->lock, flags);
  3345. __skb_queue_purge_reason(&tmp, reason);
  3346. }
  3347. EXPORT_SYMBOL(skb_queue_purge_reason);
  3348. /**
  3349. * skb_rbtree_purge - empty a skb rbtree
  3350. * @root: root of the rbtree to empty
  3351. * Return value: the sum of truesizes of all purged skbs.
  3352. *
  3353. * Delete all buffers on an &sk_buff rbtree. Each buffer is removed from
  3354. * the list and one reference dropped. This function does not take
  3355. * any lock. Synchronization should be handled by the caller (e.g., TCP
  3356. * out-of-order queue is protected by the socket lock).
  3357. */
  3358. unsigned int skb_rbtree_purge(struct rb_root *root)
  3359. {
  3360. struct rb_node *p = rb_first(root);
  3361. unsigned int sum = 0;
  3362. while (p) {
  3363. struct sk_buff *skb = rb_entry(p, struct sk_buff, rbnode);
  3364. p = rb_next(p);
  3365. rb_erase(&skb->rbnode, root);
  3366. sum += skb->truesize;
  3367. kfree_skb(skb);
  3368. }
  3369. return sum;
  3370. }
  3371. void skb_errqueue_purge(struct sk_buff_head *list)
  3372. {
  3373. struct sk_buff *skb, *next;
  3374. struct sk_buff_head kill;
  3375. unsigned long flags;
  3376. __skb_queue_head_init(&kill);
  3377. spin_lock_irqsave(&list->lock, flags);
  3378. skb_queue_walk_safe(list, skb, next) {
  3379. if (SKB_EXT_ERR(skb)->ee.ee_origin == SO_EE_ORIGIN_ZEROCOPY ||
  3380. SKB_EXT_ERR(skb)->ee.ee_origin == SO_EE_ORIGIN_TIMESTAMPING)
  3381. continue;
  3382. __skb_unlink(skb, list);
  3383. __skb_queue_tail(&kill, skb);
  3384. }
  3385. spin_unlock_irqrestore(&list->lock, flags);
  3386. __skb_queue_purge(&kill);
  3387. }
  3388. EXPORT_SYMBOL(skb_errqueue_purge);
  3389. /**
  3390. * skb_queue_head - queue a buffer at the list head
  3391. * @list: list to use
  3392. * @newsk: buffer to queue
  3393. *
  3394. * Queue a buffer at the start of the list. This function takes the
  3395. * list lock and can be used safely with other locking &sk_buff functions
  3396. * safely.
  3397. *
  3398. * A buffer cannot be placed on two lists at the same time.
  3399. */
  3400. void skb_queue_head(struct sk_buff_head *list, struct sk_buff *newsk)
  3401. {
  3402. unsigned long flags;
  3403. spin_lock_irqsave(&list->lock, flags);
  3404. __skb_queue_head(list, newsk);
  3405. spin_unlock_irqrestore(&list->lock, flags);
  3406. }
  3407. EXPORT_SYMBOL(skb_queue_head);
  3408. /**
  3409. * skb_queue_tail - queue a buffer at the list tail
  3410. * @list: list to use
  3411. * @newsk: buffer to queue
  3412. *
  3413. * Queue a buffer at the tail of the list. This function takes the
  3414. * list lock and can be used safely with other locking &sk_buff functions
  3415. * safely.
  3416. *
  3417. * A buffer cannot be placed on two lists at the same time.
  3418. */
  3419. void skb_queue_tail(struct sk_buff_head *list, struct sk_buff *newsk)
  3420. {
  3421. unsigned long flags;
  3422. spin_lock_irqsave(&list->lock, flags);
  3423. __skb_queue_tail(list, newsk);
  3424. spin_unlock_irqrestore(&list->lock, flags);
  3425. }
  3426. EXPORT_SYMBOL(skb_queue_tail);
  3427. /**
  3428. * skb_unlink - remove a buffer from a list
  3429. * @skb: buffer to remove
  3430. * @list: list to use
  3431. *
  3432. * Remove a packet from a list. The list locks are taken and this
  3433. * function is atomic with respect to other list locked calls
  3434. *
  3435. * You must know what list the SKB is on.
  3436. */
  3437. void skb_unlink(struct sk_buff *skb, struct sk_buff_head *list)
  3438. {
  3439. unsigned long flags;
  3440. spin_lock_irqsave(&list->lock, flags);
  3441. __skb_unlink(skb, list);
  3442. spin_unlock_irqrestore(&list->lock, flags);
  3443. }
  3444. EXPORT_SYMBOL(skb_unlink);
  3445. /**
  3446. * skb_append - append a buffer
  3447. * @old: buffer to insert after
  3448. * @newsk: buffer to insert
  3449. * @list: list to use
  3450. *
  3451. * Place a packet after a given packet in a list. The list locks are taken
  3452. * and this function is atomic with respect to other list locked calls.
  3453. * A buffer cannot be placed on two lists at the same time.
  3454. */
  3455. void skb_append(struct sk_buff *old, struct sk_buff *newsk, struct sk_buff_head *list)
  3456. {
  3457. unsigned long flags;
  3458. spin_lock_irqsave(&list->lock, flags);
  3459. __skb_queue_after(list, old, newsk);
  3460. spin_unlock_irqrestore(&list->lock, flags);
  3461. }
  3462. EXPORT_SYMBOL(skb_append);
  3463. static inline void skb_split_inside_header(struct sk_buff *skb,
  3464. struct sk_buff* skb1,
  3465. const u32 len, const int pos)
  3466. {
  3467. int i;
  3468. skb_copy_from_linear_data_offset(skb, len, skb_put(skb1, pos - len),
  3469. pos - len);
  3470. /* And move data appendix as is. */
  3471. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++)
  3472. skb_shinfo(skb1)->frags[i] = skb_shinfo(skb)->frags[i];
  3473. skb_shinfo(skb1)->nr_frags = skb_shinfo(skb)->nr_frags;
  3474. skb1->unreadable = skb->unreadable;
  3475. skb_shinfo(skb)->nr_frags = 0;
  3476. skb1->data_len = skb->data_len;
  3477. skb1->len += skb1->data_len;
  3478. skb->data_len = 0;
  3479. skb->len = len;
  3480. skb_set_tail_pointer(skb, len);
  3481. }
  3482. static inline void skb_split_no_header(struct sk_buff *skb,
  3483. struct sk_buff* skb1,
  3484. const u32 len, int pos)
  3485. {
  3486. int i, k = 0;
  3487. const int nfrags = skb_shinfo(skb)->nr_frags;
  3488. skb_shinfo(skb)->nr_frags = 0;
  3489. skb1->len = skb1->data_len = skb->len - len;
  3490. skb->len = len;
  3491. skb->data_len = len - pos;
  3492. for (i = 0; i < nfrags; i++) {
  3493. int size = skb_frag_size(&skb_shinfo(skb)->frags[i]);
  3494. if (pos + size > len) {
  3495. skb_shinfo(skb1)->frags[k] = skb_shinfo(skb)->frags[i];
  3496. if (pos < len) {
  3497. /* Split frag.
  3498. * We have two variants in this case:
  3499. * 1. Move all the frag to the second
  3500. * part, if it is possible. F.e.
  3501. * this approach is mandatory for TUX,
  3502. * where splitting is expensive.
  3503. * 2. Split is accurately. We make this.
  3504. */
  3505. skb_frag_ref(skb, i);
  3506. skb_frag_off_add(&skb_shinfo(skb1)->frags[0], len - pos);
  3507. skb_frag_size_sub(&skb_shinfo(skb1)->frags[0], len - pos);
  3508. skb_frag_size_set(&skb_shinfo(skb)->frags[i], len - pos);
  3509. skb_shinfo(skb)->nr_frags++;
  3510. }
  3511. k++;
  3512. } else
  3513. skb_shinfo(skb)->nr_frags++;
  3514. pos += size;
  3515. }
  3516. skb_shinfo(skb1)->nr_frags = k;
  3517. skb1->unreadable = skb->unreadable;
  3518. }
  3519. /**
  3520. * skb_split - Split fragmented skb to two parts at length len.
  3521. * @skb: the buffer to split
  3522. * @skb1: the buffer to receive the second part
  3523. * @len: new length for skb
  3524. */
  3525. void skb_split(struct sk_buff *skb, struct sk_buff *skb1, const u32 len)
  3526. {
  3527. int pos = skb_headlen(skb);
  3528. const int zc_flags = SKBFL_SHARED_FRAG | SKBFL_PURE_ZEROCOPY;
  3529. skb_zcopy_downgrade_managed(skb);
  3530. skb_shinfo(skb1)->flags |= skb_shinfo(skb)->flags & zc_flags;
  3531. skb_zerocopy_clone(skb1, skb, 0);
  3532. if (len < pos) /* Split line is inside header. */
  3533. skb_split_inside_header(skb, skb1, len, pos);
  3534. else /* Second chunk has no header, nothing to copy. */
  3535. skb_split_no_header(skb, skb1, len, pos);
  3536. }
  3537. EXPORT_SYMBOL(skb_split);
  3538. /* Shifting from/to a cloned skb is a no-go.
  3539. *
  3540. * Caller cannot keep skb_shinfo related pointers past calling here!
  3541. */
  3542. static int skb_prepare_for_shift(struct sk_buff *skb)
  3543. {
  3544. return skb_unclone_keeptruesize(skb, GFP_ATOMIC);
  3545. }
  3546. /**
  3547. * skb_shift - Shifts paged data partially from skb to another
  3548. * @tgt: buffer into which tail data gets added
  3549. * @skb: buffer from which the paged data comes from
  3550. * @shiftlen: shift up to this many bytes
  3551. *
  3552. * Attempts to shift up to shiftlen worth of bytes, which may be less than
  3553. * the length of the skb, from skb to tgt. Returns number bytes shifted.
  3554. * It's up to caller to free skb if everything was shifted.
  3555. *
  3556. * If @tgt runs out of frags, the whole operation is aborted.
  3557. *
  3558. * Skb cannot include anything else but paged data while tgt is allowed
  3559. * to have non-paged data as well.
  3560. *
  3561. * TODO: full sized shift could be optimized but that would need
  3562. * specialized skb free'er to handle frags without up-to-date nr_frags.
  3563. */
  3564. int skb_shift(struct sk_buff *tgt, struct sk_buff *skb, int shiftlen)
  3565. {
  3566. int from, to, merge, todo;
  3567. skb_frag_t *fragfrom, *fragto;
  3568. BUG_ON(shiftlen > skb->len);
  3569. if (skb_headlen(skb))
  3570. return 0;
  3571. if (skb_zcopy(tgt) || skb_zcopy(skb))
  3572. return 0;
  3573. DEBUG_NET_WARN_ON_ONCE(tgt->pp_recycle != skb->pp_recycle);
  3574. DEBUG_NET_WARN_ON_ONCE(skb_cmp_decrypted(tgt, skb));
  3575. todo = shiftlen;
  3576. from = 0;
  3577. to = skb_shinfo(tgt)->nr_frags;
  3578. fragfrom = &skb_shinfo(skb)->frags[from];
  3579. /* Actual merge is delayed until the point when we know we can
  3580. * commit all, so that we don't have to undo partial changes
  3581. */
  3582. if (!skb_can_coalesce(tgt, to, skb_frag_page(fragfrom),
  3583. skb_frag_off(fragfrom))) {
  3584. merge = -1;
  3585. } else {
  3586. merge = to - 1;
  3587. todo -= skb_frag_size(fragfrom);
  3588. if (todo < 0) {
  3589. if (skb_prepare_for_shift(skb) ||
  3590. skb_prepare_for_shift(tgt))
  3591. return 0;
  3592. /* All previous frag pointers might be stale! */
  3593. fragfrom = &skb_shinfo(skb)->frags[from];
  3594. fragto = &skb_shinfo(tgt)->frags[merge];
  3595. skb_frag_size_add(fragto, shiftlen);
  3596. skb_frag_size_sub(fragfrom, shiftlen);
  3597. skb_frag_off_add(fragfrom, shiftlen);
  3598. goto onlymerged;
  3599. }
  3600. from++;
  3601. }
  3602. /* Skip full, not-fitting skb to avoid expensive operations */
  3603. if ((shiftlen == skb->len) &&
  3604. (skb_shinfo(skb)->nr_frags - from) > (MAX_SKB_FRAGS - to))
  3605. return 0;
  3606. if (skb_prepare_for_shift(skb) || skb_prepare_for_shift(tgt))
  3607. return 0;
  3608. while ((todo > 0) && (from < skb_shinfo(skb)->nr_frags)) {
  3609. if (to == MAX_SKB_FRAGS)
  3610. return 0;
  3611. fragfrom = &skb_shinfo(skb)->frags[from];
  3612. fragto = &skb_shinfo(tgt)->frags[to];
  3613. if (todo >= skb_frag_size(fragfrom)) {
  3614. *fragto = *fragfrom;
  3615. todo -= skb_frag_size(fragfrom);
  3616. from++;
  3617. to++;
  3618. } else {
  3619. __skb_frag_ref(fragfrom);
  3620. skb_frag_page_copy(fragto, fragfrom);
  3621. skb_frag_off_copy(fragto, fragfrom);
  3622. skb_frag_size_set(fragto, todo);
  3623. skb_frag_off_add(fragfrom, todo);
  3624. skb_frag_size_sub(fragfrom, todo);
  3625. todo = 0;
  3626. to++;
  3627. break;
  3628. }
  3629. }
  3630. /* Ready to "commit" this state change to tgt */
  3631. skb_shinfo(tgt)->nr_frags = to;
  3632. if (merge >= 0) {
  3633. fragfrom = &skb_shinfo(skb)->frags[0];
  3634. fragto = &skb_shinfo(tgt)->frags[merge];
  3635. skb_frag_size_add(fragto, skb_frag_size(fragfrom));
  3636. __skb_frag_unref(fragfrom, skb->pp_recycle);
  3637. }
  3638. /* Reposition in the original skb */
  3639. to = 0;
  3640. while (from < skb_shinfo(skb)->nr_frags)
  3641. skb_shinfo(skb)->frags[to++] = skb_shinfo(skb)->frags[from++];
  3642. skb_shinfo(skb)->nr_frags = to;
  3643. BUG_ON(todo > 0 && !skb_shinfo(skb)->nr_frags);
  3644. onlymerged:
  3645. /* Most likely the tgt won't ever need its checksum anymore, skb on
  3646. * the other hand might need it if it needs to be resent
  3647. */
  3648. tgt->ip_summed = CHECKSUM_PARTIAL;
  3649. skb->ip_summed = CHECKSUM_PARTIAL;
  3650. skb_len_add(skb, -shiftlen);
  3651. skb_len_add(tgt, shiftlen);
  3652. return shiftlen;
  3653. }
  3654. /**
  3655. * skb_prepare_seq_read - Prepare a sequential read of skb data
  3656. * @skb: the buffer to read
  3657. * @from: lower offset of data to be read
  3658. * @to: upper offset of data to be read
  3659. * @st: state variable
  3660. *
  3661. * Initializes the specified state variable. Must be called before
  3662. * invoking skb_seq_read() for the first time.
  3663. */
  3664. void skb_prepare_seq_read(struct sk_buff *skb, unsigned int from,
  3665. unsigned int to, struct skb_seq_state *st)
  3666. {
  3667. st->lower_offset = from;
  3668. st->upper_offset = to;
  3669. st->root_skb = st->cur_skb = skb;
  3670. st->frag_idx = st->stepped_offset = 0;
  3671. st->frag_data = NULL;
  3672. st->frag_off = 0;
  3673. }
  3674. EXPORT_SYMBOL(skb_prepare_seq_read);
  3675. /**
  3676. * skb_seq_read - Sequentially read skb data
  3677. * @consumed: number of bytes consumed by the caller so far
  3678. * @data: destination pointer for data to be returned
  3679. * @st: state variable
  3680. *
  3681. * Reads a block of skb data at @consumed relative to the
  3682. * lower offset specified to skb_prepare_seq_read(). Assigns
  3683. * the head of the data block to @data and returns the length
  3684. * of the block or 0 if the end of the skb data or the upper
  3685. * offset has been reached.
  3686. *
  3687. * The caller is not required to consume all of the data
  3688. * returned, i.e. @consumed is typically set to the number
  3689. * of bytes already consumed and the next call to
  3690. * skb_seq_read() will return the remaining part of the block.
  3691. *
  3692. * Note 1: The size of each block of data returned can be arbitrary,
  3693. * this limitation is the cost for zerocopy sequential
  3694. * reads of potentially non linear data.
  3695. *
  3696. * Note 2: Fragment lists within fragments are not implemented
  3697. * at the moment, state->root_skb could be replaced with
  3698. * a stack for this purpose.
  3699. */
  3700. unsigned int skb_seq_read(unsigned int consumed, const u8 **data,
  3701. struct skb_seq_state *st)
  3702. {
  3703. unsigned int block_limit, abs_offset = consumed + st->lower_offset;
  3704. skb_frag_t *frag;
  3705. if (unlikely(abs_offset >= st->upper_offset)) {
  3706. if (st->frag_data) {
  3707. kunmap_atomic(st->frag_data);
  3708. st->frag_data = NULL;
  3709. }
  3710. return 0;
  3711. }
  3712. next_skb:
  3713. block_limit = skb_headlen(st->cur_skb) + st->stepped_offset;
  3714. if (abs_offset < block_limit && !st->frag_data) {
  3715. *data = st->cur_skb->data + (abs_offset - st->stepped_offset);
  3716. return block_limit - abs_offset;
  3717. }
  3718. if (!skb_frags_readable(st->cur_skb))
  3719. return 0;
  3720. if (st->frag_idx == 0 && !st->frag_data)
  3721. st->stepped_offset += skb_headlen(st->cur_skb);
  3722. while (st->frag_idx < skb_shinfo(st->cur_skb)->nr_frags) {
  3723. unsigned int pg_idx, pg_off, pg_sz;
  3724. frag = &skb_shinfo(st->cur_skb)->frags[st->frag_idx];
  3725. pg_idx = 0;
  3726. pg_off = skb_frag_off(frag);
  3727. pg_sz = skb_frag_size(frag);
  3728. if (skb_frag_must_loop(skb_frag_page(frag))) {
  3729. pg_idx = (pg_off + st->frag_off) >> PAGE_SHIFT;
  3730. pg_off = offset_in_page(pg_off + st->frag_off);
  3731. pg_sz = min_t(unsigned int, pg_sz - st->frag_off,
  3732. PAGE_SIZE - pg_off);
  3733. }
  3734. block_limit = pg_sz + st->stepped_offset;
  3735. if (abs_offset < block_limit) {
  3736. if (!st->frag_data)
  3737. st->frag_data = kmap_atomic(skb_frag_page(frag) + pg_idx);
  3738. *data = (u8 *)st->frag_data + pg_off +
  3739. (abs_offset - st->stepped_offset);
  3740. return block_limit - abs_offset;
  3741. }
  3742. if (st->frag_data) {
  3743. kunmap_atomic(st->frag_data);
  3744. st->frag_data = NULL;
  3745. }
  3746. st->stepped_offset += pg_sz;
  3747. st->frag_off += pg_sz;
  3748. if (st->frag_off == skb_frag_size(frag)) {
  3749. st->frag_off = 0;
  3750. st->frag_idx++;
  3751. }
  3752. }
  3753. if (st->frag_data) {
  3754. kunmap_atomic(st->frag_data);
  3755. st->frag_data = NULL;
  3756. }
  3757. if (st->root_skb == st->cur_skb && skb_has_frag_list(st->root_skb)) {
  3758. st->cur_skb = skb_shinfo(st->root_skb)->frag_list;
  3759. st->frag_idx = 0;
  3760. goto next_skb;
  3761. } else if (st->cur_skb->next) {
  3762. st->cur_skb = st->cur_skb->next;
  3763. st->frag_idx = 0;
  3764. goto next_skb;
  3765. }
  3766. return 0;
  3767. }
  3768. EXPORT_SYMBOL(skb_seq_read);
  3769. /**
  3770. * skb_abort_seq_read - Abort a sequential read of skb data
  3771. * @st: state variable
  3772. *
  3773. * Must be called if skb_seq_read() was not called until it
  3774. * returned 0.
  3775. */
  3776. void skb_abort_seq_read(struct skb_seq_state *st)
  3777. {
  3778. if (st->frag_data)
  3779. kunmap_atomic(st->frag_data);
  3780. }
  3781. EXPORT_SYMBOL(skb_abort_seq_read);
  3782. /**
  3783. * skb_copy_seq_read() - copy from a skb_seq_state to a buffer
  3784. * @st: source skb_seq_state
  3785. * @offset: offset in source
  3786. * @to: destination buffer
  3787. * @len: number of bytes to copy
  3788. *
  3789. * Copy @len bytes from @offset bytes into the source @st to the destination
  3790. * buffer @to. `offset` should increase (or be unchanged) with each subsequent
  3791. * call to this function. If offset needs to decrease from the previous use `st`
  3792. * should be reset first.
  3793. *
  3794. * Return: 0 on success or -EINVAL if the copy ended early
  3795. */
  3796. int skb_copy_seq_read(struct skb_seq_state *st, int offset, void *to, int len)
  3797. {
  3798. const u8 *data;
  3799. u32 sqlen;
  3800. for (;;) {
  3801. sqlen = skb_seq_read(offset, &data, st);
  3802. if (sqlen == 0)
  3803. return -EINVAL;
  3804. if (sqlen >= len) {
  3805. memcpy(to, data, len);
  3806. return 0;
  3807. }
  3808. memcpy(to, data, sqlen);
  3809. to += sqlen;
  3810. offset += sqlen;
  3811. len -= sqlen;
  3812. }
  3813. }
  3814. EXPORT_SYMBOL(skb_copy_seq_read);
  3815. #define TS_SKB_CB(state) ((struct skb_seq_state *) &((state)->cb))
  3816. static unsigned int skb_ts_get_next_block(unsigned int offset, const u8 **text,
  3817. struct ts_config *conf,
  3818. struct ts_state *state)
  3819. {
  3820. return skb_seq_read(offset, text, TS_SKB_CB(state));
  3821. }
  3822. static void skb_ts_finish(struct ts_config *conf, struct ts_state *state)
  3823. {
  3824. skb_abort_seq_read(TS_SKB_CB(state));
  3825. }
  3826. /**
  3827. * skb_find_text - Find a text pattern in skb data
  3828. * @skb: the buffer to look in
  3829. * @from: search offset
  3830. * @to: search limit
  3831. * @config: textsearch configuration
  3832. *
  3833. * Finds a pattern in the skb data according to the specified
  3834. * textsearch configuration. Use textsearch_next() to retrieve
  3835. * subsequent occurrences of the pattern. Returns the offset
  3836. * to the first occurrence or UINT_MAX if no match was found.
  3837. */
  3838. unsigned int skb_find_text(struct sk_buff *skb, unsigned int from,
  3839. unsigned int to, struct ts_config *config)
  3840. {
  3841. unsigned int patlen = config->ops->get_pattern_len(config);
  3842. struct ts_state state;
  3843. unsigned int ret;
  3844. BUILD_BUG_ON(sizeof(struct skb_seq_state) > sizeof(state.cb));
  3845. config->get_next_block = skb_ts_get_next_block;
  3846. config->finish = skb_ts_finish;
  3847. skb_prepare_seq_read(skb, from, to, TS_SKB_CB(&state));
  3848. ret = textsearch_find(config, &state);
  3849. return (ret + patlen <= to - from ? ret : UINT_MAX);
  3850. }
  3851. EXPORT_SYMBOL(skb_find_text);
  3852. int skb_append_pagefrags(struct sk_buff *skb, struct page *page,
  3853. int offset, size_t size, size_t max_frags)
  3854. {
  3855. int i = skb_shinfo(skb)->nr_frags;
  3856. if (skb_can_coalesce(skb, i, page, offset)) {
  3857. skb_frag_size_add(&skb_shinfo(skb)->frags[i - 1], size);
  3858. } else if (i < max_frags) {
  3859. skb_zcopy_downgrade_managed(skb);
  3860. get_page(page);
  3861. skb_fill_page_desc_noacc(skb, i, page, offset, size);
  3862. } else {
  3863. return -EMSGSIZE;
  3864. }
  3865. return 0;
  3866. }
  3867. EXPORT_SYMBOL_GPL(skb_append_pagefrags);
  3868. /**
  3869. * skb_pull_rcsum - pull skb and update receive checksum
  3870. * @skb: buffer to update
  3871. * @len: length of data pulled
  3872. *
  3873. * This function performs an skb_pull on the packet and updates
  3874. * the CHECKSUM_COMPLETE checksum. It should be used on
  3875. * receive path processing instead of skb_pull unless you know
  3876. * that the checksum difference is zero (e.g., a valid IP header)
  3877. * or you are setting ip_summed to CHECKSUM_NONE.
  3878. */
  3879. void *skb_pull_rcsum(struct sk_buff *skb, unsigned int len)
  3880. {
  3881. unsigned char *data = skb->data;
  3882. BUG_ON(len > skb->len);
  3883. __skb_pull(skb, len);
  3884. skb_postpull_rcsum(skb, data, len);
  3885. return skb->data;
  3886. }
  3887. EXPORT_SYMBOL_GPL(skb_pull_rcsum);
  3888. static inline skb_frag_t skb_head_frag_to_page_desc(struct sk_buff *frag_skb)
  3889. {
  3890. skb_frag_t head_frag;
  3891. struct page *page;
  3892. page = virt_to_head_page(frag_skb->head);
  3893. skb_frag_fill_page_desc(&head_frag, page, frag_skb->data -
  3894. (unsigned char *)page_address(page),
  3895. skb_headlen(frag_skb));
  3896. return head_frag;
  3897. }
  3898. struct sk_buff *skb_segment_list(struct sk_buff *skb,
  3899. netdev_features_t features,
  3900. unsigned int offset)
  3901. {
  3902. struct sk_buff *list_skb = skb_shinfo(skb)->frag_list;
  3903. unsigned int tnl_hlen = skb_tnl_header_len(skb);
  3904. unsigned int delta_truesize = 0;
  3905. unsigned int delta_len = 0;
  3906. struct sk_buff *tail = NULL;
  3907. struct sk_buff *nskb, *tmp;
  3908. int len_diff, err;
  3909. skb_push(skb, -skb_network_offset(skb) + offset);
  3910. /* Ensure the head is writeable before touching the shared info */
  3911. err = skb_unclone(skb, GFP_ATOMIC);
  3912. if (err)
  3913. goto err_linearize;
  3914. skb_shinfo(skb)->frag_list = NULL;
  3915. while (list_skb) {
  3916. nskb = list_skb;
  3917. list_skb = list_skb->next;
  3918. err = 0;
  3919. delta_truesize += nskb->truesize;
  3920. if (skb_shared(nskb)) {
  3921. tmp = skb_clone(nskb, GFP_ATOMIC);
  3922. if (tmp) {
  3923. consume_skb(nskb);
  3924. nskb = tmp;
  3925. err = skb_unclone(nskb, GFP_ATOMIC);
  3926. } else {
  3927. err = -ENOMEM;
  3928. }
  3929. }
  3930. if (!tail)
  3931. skb->next = nskb;
  3932. else
  3933. tail->next = nskb;
  3934. if (unlikely(err)) {
  3935. nskb->next = list_skb;
  3936. goto err_linearize;
  3937. }
  3938. tail = nskb;
  3939. delta_len += nskb->len;
  3940. skb_push(nskb, -skb_network_offset(nskb) + offset);
  3941. skb_release_head_state(nskb);
  3942. len_diff = skb_network_header_len(nskb) - skb_network_header_len(skb);
  3943. __copy_skb_header(nskb, skb);
  3944. skb_headers_offset_update(nskb, skb_headroom(nskb) - skb_headroom(skb));
  3945. nskb->transport_header += len_diff;
  3946. skb_copy_from_linear_data_offset(skb, -tnl_hlen,
  3947. nskb->data - tnl_hlen,
  3948. offset + tnl_hlen);
  3949. if (skb_needs_linearize(nskb, features) &&
  3950. __skb_linearize(nskb))
  3951. goto err_linearize;
  3952. }
  3953. skb->truesize = skb->truesize - delta_truesize;
  3954. skb->data_len = skb->data_len - delta_len;
  3955. skb->len = skb->len - delta_len;
  3956. skb_gso_reset(skb);
  3957. skb->prev = tail;
  3958. if (skb_needs_linearize(skb, features) &&
  3959. __skb_linearize(skb))
  3960. goto err_linearize;
  3961. skb_get(skb);
  3962. return skb;
  3963. err_linearize:
  3964. kfree_skb_list(skb->next);
  3965. skb->next = NULL;
  3966. return ERR_PTR(-ENOMEM);
  3967. }
  3968. EXPORT_SYMBOL_GPL(skb_segment_list);
  3969. /**
  3970. * skb_segment - Perform protocol segmentation on skb.
  3971. * @head_skb: buffer to segment
  3972. * @features: features for the output path (see dev->features)
  3973. *
  3974. * This function performs segmentation on the given skb. It returns
  3975. * a pointer to the first in a list of new skbs for the segments.
  3976. * In case of error it returns ERR_PTR(err).
  3977. */
  3978. struct sk_buff *skb_segment(struct sk_buff *head_skb,
  3979. netdev_features_t features)
  3980. {
  3981. struct sk_buff *segs = NULL;
  3982. struct sk_buff *tail = NULL;
  3983. struct sk_buff *list_skb = skb_shinfo(head_skb)->frag_list;
  3984. unsigned int mss = skb_shinfo(head_skb)->gso_size;
  3985. unsigned int doffset = head_skb->data - skb_mac_header(head_skb);
  3986. unsigned int offset = doffset;
  3987. unsigned int tnl_hlen = skb_tnl_header_len(head_skb);
  3988. unsigned int partial_segs = 0;
  3989. unsigned int headroom;
  3990. unsigned int len = head_skb->len;
  3991. struct sk_buff *frag_skb;
  3992. skb_frag_t *frag;
  3993. __be16 proto;
  3994. bool csum, sg;
  3995. int err = -ENOMEM;
  3996. int i = 0;
  3997. int nfrags, pos;
  3998. if ((skb_shinfo(head_skb)->gso_type & SKB_GSO_DODGY) &&
  3999. mss != GSO_BY_FRAGS && mss != skb_headlen(head_skb)) {
  4000. struct sk_buff *check_skb;
  4001. for (check_skb = list_skb; check_skb; check_skb = check_skb->next) {
  4002. if (skb_headlen(check_skb) && !check_skb->head_frag) {
  4003. /* gso_size is untrusted, and we have a frag_list with
  4004. * a linear non head_frag item.
  4005. *
  4006. * If head_skb's headlen does not fit requested gso_size,
  4007. * it means that the frag_list members do NOT terminate
  4008. * on exact gso_size boundaries. Hence we cannot perform
  4009. * skb_frag_t page sharing. Therefore we must fallback to
  4010. * copying the frag_list skbs; we do so by disabling SG.
  4011. */
  4012. features &= ~NETIF_F_SG;
  4013. break;
  4014. }
  4015. }
  4016. }
  4017. __skb_push(head_skb, doffset);
  4018. proto = skb_network_protocol(head_skb, NULL);
  4019. if (unlikely(!proto))
  4020. return ERR_PTR(-EINVAL);
  4021. sg = !!(features & NETIF_F_SG);
  4022. csum = !!can_checksum_protocol(features, proto);
  4023. if (sg && csum && (mss != GSO_BY_FRAGS)) {
  4024. if (!(features & NETIF_F_GSO_PARTIAL)) {
  4025. struct sk_buff *iter;
  4026. unsigned int frag_len;
  4027. if (!list_skb ||
  4028. !net_gso_ok(features, skb_shinfo(head_skb)->gso_type))
  4029. goto normal;
  4030. /* If we get here then all the required
  4031. * GSO features except frag_list are supported.
  4032. * Try to split the SKB to multiple GSO SKBs
  4033. * with no frag_list.
  4034. * Currently we can do that only when the buffers don't
  4035. * have a linear part and all the buffers except
  4036. * the last are of the same length.
  4037. */
  4038. frag_len = list_skb->len;
  4039. skb_walk_frags(head_skb, iter) {
  4040. if (frag_len != iter->len && iter->next)
  4041. goto normal;
  4042. if (skb_headlen(iter) && !iter->head_frag)
  4043. goto normal;
  4044. len -= iter->len;
  4045. }
  4046. if (len != frag_len)
  4047. goto normal;
  4048. }
  4049. /* GSO partial only requires that we trim off any excess that
  4050. * doesn't fit into an MSS sized block, so take care of that
  4051. * now.
  4052. * Cap len to not accidentally hit GSO_BY_FRAGS.
  4053. */
  4054. partial_segs = min(len, GSO_BY_FRAGS - 1) / mss;
  4055. if (partial_segs > 1)
  4056. mss *= partial_segs;
  4057. else
  4058. partial_segs = 0;
  4059. }
  4060. normal:
  4061. headroom = skb_headroom(head_skb);
  4062. pos = skb_headlen(head_skb);
  4063. if (skb_orphan_frags(head_skb, GFP_ATOMIC))
  4064. return ERR_PTR(-ENOMEM);
  4065. nfrags = skb_shinfo(head_skb)->nr_frags;
  4066. frag = skb_shinfo(head_skb)->frags;
  4067. frag_skb = head_skb;
  4068. do {
  4069. struct sk_buff *nskb;
  4070. skb_frag_t *nskb_frag;
  4071. int hsize;
  4072. int size;
  4073. if (unlikely(mss == GSO_BY_FRAGS)) {
  4074. len = list_skb->len;
  4075. } else {
  4076. len = head_skb->len - offset;
  4077. if (len > mss)
  4078. len = mss;
  4079. }
  4080. hsize = skb_headlen(head_skb) - offset;
  4081. if (hsize <= 0 && i >= nfrags && skb_headlen(list_skb) &&
  4082. (skb_headlen(list_skb) == len || sg)) {
  4083. BUG_ON(skb_headlen(list_skb) > len);
  4084. nskb = skb_clone(list_skb, GFP_ATOMIC);
  4085. if (unlikely(!nskb))
  4086. goto err;
  4087. i = 0;
  4088. nfrags = skb_shinfo(list_skb)->nr_frags;
  4089. frag = skb_shinfo(list_skb)->frags;
  4090. frag_skb = list_skb;
  4091. pos += skb_headlen(list_skb);
  4092. while (pos < offset + len) {
  4093. BUG_ON(i >= nfrags);
  4094. size = skb_frag_size(frag);
  4095. if (pos + size > offset + len)
  4096. break;
  4097. i++;
  4098. pos += size;
  4099. frag++;
  4100. }
  4101. list_skb = list_skb->next;
  4102. if (unlikely(pskb_trim(nskb, len))) {
  4103. kfree_skb(nskb);
  4104. goto err;
  4105. }
  4106. hsize = skb_end_offset(nskb);
  4107. if (skb_cow_head(nskb, doffset + headroom)) {
  4108. kfree_skb(nskb);
  4109. goto err;
  4110. }
  4111. nskb->truesize += skb_end_offset(nskb) - hsize;
  4112. skb_release_head_state(nskb);
  4113. __skb_push(nskb, doffset);
  4114. } else {
  4115. if (hsize < 0)
  4116. hsize = 0;
  4117. if (hsize > len || !sg)
  4118. hsize = len;
  4119. nskb = __alloc_skb(hsize + doffset + headroom,
  4120. GFP_ATOMIC, skb_alloc_rx_flag(head_skb),
  4121. NUMA_NO_NODE);
  4122. if (unlikely(!nskb))
  4123. goto err;
  4124. skb_reserve(nskb, headroom);
  4125. __skb_put(nskb, doffset);
  4126. }
  4127. if (segs)
  4128. tail->next = nskb;
  4129. else
  4130. segs = nskb;
  4131. tail = nskb;
  4132. __copy_skb_header(nskb, head_skb);
  4133. skb_headers_offset_update(nskb, skb_headroom(nskb) - headroom);
  4134. skb_reset_mac_len(nskb);
  4135. skb_copy_from_linear_data_offset(head_skb, -tnl_hlen,
  4136. nskb->data - tnl_hlen,
  4137. doffset + tnl_hlen);
  4138. if (nskb->len == len + doffset)
  4139. goto perform_csum_check;
  4140. if (!sg) {
  4141. if (!csum) {
  4142. if (!nskb->remcsum_offload)
  4143. nskb->ip_summed = CHECKSUM_NONE;
  4144. SKB_GSO_CB(nskb)->csum =
  4145. skb_copy_and_csum_bits(head_skb, offset,
  4146. skb_put(nskb,
  4147. len),
  4148. len);
  4149. SKB_GSO_CB(nskb)->csum_start =
  4150. skb_headroom(nskb) + doffset;
  4151. } else {
  4152. if (skb_copy_bits(head_skb, offset, skb_put(nskb, len), len))
  4153. goto err;
  4154. }
  4155. continue;
  4156. }
  4157. nskb_frag = skb_shinfo(nskb)->frags;
  4158. skb_copy_from_linear_data_offset(head_skb, offset,
  4159. skb_put(nskb, hsize), hsize);
  4160. skb_shinfo(nskb)->flags |= skb_shinfo(head_skb)->flags &
  4161. SKBFL_SHARED_FRAG;
  4162. if (skb_zerocopy_clone(nskb, frag_skb, GFP_ATOMIC))
  4163. goto err;
  4164. while (pos < offset + len) {
  4165. if (i >= nfrags) {
  4166. if (skb_orphan_frags(list_skb, GFP_ATOMIC) ||
  4167. skb_zerocopy_clone(nskb, list_skb,
  4168. GFP_ATOMIC))
  4169. goto err;
  4170. i = 0;
  4171. nfrags = skb_shinfo(list_skb)->nr_frags;
  4172. frag = skb_shinfo(list_skb)->frags;
  4173. frag_skb = list_skb;
  4174. if (!skb_headlen(list_skb)) {
  4175. BUG_ON(!nfrags);
  4176. } else {
  4177. BUG_ON(!list_skb->head_frag);
  4178. /* to make room for head_frag. */
  4179. i--;
  4180. frag--;
  4181. }
  4182. list_skb = list_skb->next;
  4183. }
  4184. if (unlikely(skb_shinfo(nskb)->nr_frags >=
  4185. MAX_SKB_FRAGS)) {
  4186. net_warn_ratelimited(
  4187. "skb_segment: too many frags: %u %u\n",
  4188. pos, mss);
  4189. err = -EINVAL;
  4190. goto err;
  4191. }
  4192. *nskb_frag = (i < 0) ? skb_head_frag_to_page_desc(frag_skb) : *frag;
  4193. __skb_frag_ref(nskb_frag);
  4194. size = skb_frag_size(nskb_frag);
  4195. if (pos < offset) {
  4196. skb_frag_off_add(nskb_frag, offset - pos);
  4197. skb_frag_size_sub(nskb_frag, offset - pos);
  4198. }
  4199. skb_shinfo(nskb)->nr_frags++;
  4200. if (pos + size <= offset + len) {
  4201. i++;
  4202. frag++;
  4203. pos += size;
  4204. } else {
  4205. skb_frag_size_sub(nskb_frag, pos + size - (offset + len));
  4206. goto skip_fraglist;
  4207. }
  4208. nskb_frag++;
  4209. }
  4210. skip_fraglist:
  4211. nskb->data_len = len - hsize;
  4212. nskb->len += nskb->data_len;
  4213. nskb->truesize += nskb->data_len;
  4214. perform_csum_check:
  4215. if (!csum) {
  4216. if (skb_has_shared_frag(nskb) &&
  4217. __skb_linearize(nskb))
  4218. goto err;
  4219. if (!nskb->remcsum_offload)
  4220. nskb->ip_summed = CHECKSUM_NONE;
  4221. SKB_GSO_CB(nskb)->csum =
  4222. skb_checksum(nskb, doffset,
  4223. nskb->len - doffset, 0);
  4224. SKB_GSO_CB(nskb)->csum_start =
  4225. skb_headroom(nskb) + doffset;
  4226. }
  4227. } while ((offset += len) < head_skb->len);
  4228. /* Some callers want to get the end of the list.
  4229. * Put it in segs->prev to avoid walking the list.
  4230. * (see validate_xmit_skb_list() for example)
  4231. */
  4232. segs->prev = tail;
  4233. if (partial_segs) {
  4234. struct sk_buff *iter;
  4235. int type = skb_shinfo(head_skb)->gso_type;
  4236. unsigned short gso_size = skb_shinfo(head_skb)->gso_size;
  4237. /* Update type to add partial and then remove dodgy if set */
  4238. type |= (features & NETIF_F_GSO_PARTIAL) / NETIF_F_GSO_PARTIAL * SKB_GSO_PARTIAL;
  4239. type &= ~SKB_GSO_DODGY;
  4240. /* Update GSO info and prepare to start updating headers on
  4241. * our way back down the stack of protocols.
  4242. */
  4243. for (iter = segs; iter; iter = iter->next) {
  4244. skb_shinfo(iter)->gso_size = gso_size;
  4245. skb_shinfo(iter)->gso_segs = partial_segs;
  4246. skb_shinfo(iter)->gso_type = type;
  4247. SKB_GSO_CB(iter)->data_offset = skb_headroom(iter) + doffset;
  4248. }
  4249. if (tail->len - doffset <= gso_size)
  4250. skb_shinfo(tail)->gso_size = 0;
  4251. else if (tail != segs)
  4252. skb_shinfo(tail)->gso_segs = DIV_ROUND_UP(tail->len - doffset, gso_size);
  4253. }
  4254. /* Following permits correct backpressure, for protocols
  4255. * using skb_set_owner_w().
  4256. * Idea is to tranfert ownership from head_skb to last segment.
  4257. */
  4258. if (head_skb->destructor == sock_wfree) {
  4259. swap(tail->truesize, head_skb->truesize);
  4260. swap(tail->destructor, head_skb->destructor);
  4261. swap(tail->sk, head_skb->sk);
  4262. }
  4263. return segs;
  4264. err:
  4265. kfree_skb_list(segs);
  4266. return ERR_PTR(err);
  4267. }
  4268. EXPORT_SYMBOL_GPL(skb_segment);
  4269. #ifdef CONFIG_SKB_EXTENSIONS
  4270. #define SKB_EXT_ALIGN_VALUE 8
  4271. #define SKB_EXT_CHUNKSIZEOF(x) (ALIGN((sizeof(x)), SKB_EXT_ALIGN_VALUE) / SKB_EXT_ALIGN_VALUE)
  4272. static const u8 skb_ext_type_len[] = {
  4273. #if IS_ENABLED(CONFIG_BRIDGE_NETFILTER)
  4274. [SKB_EXT_BRIDGE_NF] = SKB_EXT_CHUNKSIZEOF(struct nf_bridge_info),
  4275. #endif
  4276. #ifdef CONFIG_XFRM
  4277. [SKB_EXT_SEC_PATH] = SKB_EXT_CHUNKSIZEOF(struct sec_path),
  4278. #endif
  4279. #if IS_ENABLED(CONFIG_NET_TC_SKB_EXT)
  4280. [TC_SKB_EXT] = SKB_EXT_CHUNKSIZEOF(struct tc_skb_ext),
  4281. #endif
  4282. #if IS_ENABLED(CONFIG_MPTCP)
  4283. [SKB_EXT_MPTCP] = SKB_EXT_CHUNKSIZEOF(struct mptcp_ext),
  4284. #endif
  4285. #if IS_ENABLED(CONFIG_MCTP_FLOWS)
  4286. [SKB_EXT_MCTP] = SKB_EXT_CHUNKSIZEOF(struct mctp_flow),
  4287. #endif
  4288. };
  4289. static __always_inline unsigned int skb_ext_total_length(void)
  4290. {
  4291. unsigned int l = SKB_EXT_CHUNKSIZEOF(struct skb_ext);
  4292. int i;
  4293. for (i = 0; i < ARRAY_SIZE(skb_ext_type_len); i++)
  4294. l += skb_ext_type_len[i];
  4295. return l;
  4296. }
  4297. static void skb_extensions_init(void)
  4298. {
  4299. BUILD_BUG_ON(SKB_EXT_NUM >= 8);
  4300. #if !IS_ENABLED(CONFIG_KCOV_INSTRUMENT_ALL)
  4301. BUILD_BUG_ON(skb_ext_total_length() > 255);
  4302. #endif
  4303. skbuff_ext_cache = kmem_cache_create("skbuff_ext_cache",
  4304. SKB_EXT_ALIGN_VALUE * skb_ext_total_length(),
  4305. 0,
  4306. SLAB_HWCACHE_ALIGN|SLAB_PANIC,
  4307. NULL);
  4308. }
  4309. #else
  4310. static void skb_extensions_init(void) {}
  4311. #endif
  4312. /* The SKB kmem_cache slab is critical for network performance. Never
  4313. * merge/alias the slab with similar sized objects. This avoids fragmentation
  4314. * that hurts performance of kmem_cache_{alloc,free}_bulk APIs.
  4315. */
  4316. #ifndef CONFIG_SLUB_TINY
  4317. #define FLAG_SKB_NO_MERGE SLAB_NO_MERGE
  4318. #else /* CONFIG_SLUB_TINY - simple loop in kmem_cache_alloc_bulk */
  4319. #define FLAG_SKB_NO_MERGE 0
  4320. #endif
  4321. void __init skb_init(void)
  4322. {
  4323. net_hotdata.skbuff_cache = kmem_cache_create_usercopy("skbuff_head_cache",
  4324. sizeof(struct sk_buff),
  4325. 0,
  4326. SLAB_HWCACHE_ALIGN|SLAB_PANIC|
  4327. FLAG_SKB_NO_MERGE,
  4328. offsetof(struct sk_buff, cb),
  4329. sizeof_field(struct sk_buff, cb),
  4330. NULL);
  4331. net_hotdata.skbuff_fclone_cache = kmem_cache_create("skbuff_fclone_cache",
  4332. sizeof(struct sk_buff_fclones),
  4333. 0,
  4334. SLAB_HWCACHE_ALIGN|SLAB_PANIC,
  4335. NULL);
  4336. /* usercopy should only access first SKB_SMALL_HEAD_HEADROOM bytes.
  4337. * struct skb_shared_info is located at the end of skb->head,
  4338. * and should not be copied to/from user.
  4339. */
  4340. net_hotdata.skb_small_head_cache = kmem_cache_create_usercopy("skbuff_small_head",
  4341. SKB_SMALL_HEAD_CACHE_SIZE,
  4342. 0,
  4343. SLAB_HWCACHE_ALIGN | SLAB_PANIC,
  4344. 0,
  4345. SKB_SMALL_HEAD_HEADROOM,
  4346. NULL);
  4347. skb_extensions_init();
  4348. }
  4349. static int
  4350. __skb_to_sgvec(struct sk_buff *skb, struct scatterlist *sg, int offset, int len,
  4351. unsigned int recursion_level)
  4352. {
  4353. int start = skb_headlen(skb);
  4354. int i, copy = start - offset;
  4355. struct sk_buff *frag_iter;
  4356. int elt = 0;
  4357. if (unlikely(recursion_level >= 24))
  4358. return -EMSGSIZE;
  4359. if (copy > 0) {
  4360. if (copy > len)
  4361. copy = len;
  4362. sg_set_buf(sg, skb->data + offset, copy);
  4363. elt++;
  4364. if ((len -= copy) == 0)
  4365. return elt;
  4366. offset += copy;
  4367. }
  4368. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++) {
  4369. int end;
  4370. WARN_ON(start > offset + len);
  4371. end = start + skb_frag_size(&skb_shinfo(skb)->frags[i]);
  4372. if ((copy = end - offset) > 0) {
  4373. skb_frag_t *frag = &skb_shinfo(skb)->frags[i];
  4374. if (unlikely(elt && sg_is_last(&sg[elt - 1])))
  4375. return -EMSGSIZE;
  4376. if (copy > len)
  4377. copy = len;
  4378. sg_set_page(&sg[elt], skb_frag_page(frag), copy,
  4379. skb_frag_off(frag) + offset - start);
  4380. elt++;
  4381. if (!(len -= copy))
  4382. return elt;
  4383. offset += copy;
  4384. }
  4385. start = end;
  4386. }
  4387. skb_walk_frags(skb, frag_iter) {
  4388. int end, ret;
  4389. WARN_ON(start > offset + len);
  4390. end = start + frag_iter->len;
  4391. if ((copy = end - offset) > 0) {
  4392. if (unlikely(elt && sg_is_last(&sg[elt - 1])))
  4393. return -EMSGSIZE;
  4394. if (copy > len)
  4395. copy = len;
  4396. ret = __skb_to_sgvec(frag_iter, sg+elt, offset - start,
  4397. copy, recursion_level + 1);
  4398. if (unlikely(ret < 0))
  4399. return ret;
  4400. elt += ret;
  4401. if ((len -= copy) == 0)
  4402. return elt;
  4403. offset += copy;
  4404. }
  4405. start = end;
  4406. }
  4407. BUG_ON(len);
  4408. return elt;
  4409. }
  4410. /**
  4411. * skb_to_sgvec - Fill a scatter-gather list from a socket buffer
  4412. * @skb: Socket buffer containing the buffers to be mapped
  4413. * @sg: The scatter-gather list to map into
  4414. * @offset: The offset into the buffer's contents to start mapping
  4415. * @len: Length of buffer space to be mapped
  4416. *
  4417. * Fill the specified scatter-gather list with mappings/pointers into a
  4418. * region of the buffer space attached to a socket buffer. Returns either
  4419. * the number of scatterlist items used, or -EMSGSIZE if the contents
  4420. * could not fit.
  4421. */
  4422. int skb_to_sgvec(struct sk_buff *skb, struct scatterlist *sg, int offset, int len)
  4423. {
  4424. int nsg = __skb_to_sgvec(skb, sg, offset, len, 0);
  4425. if (nsg <= 0)
  4426. return nsg;
  4427. sg_mark_end(&sg[nsg - 1]);
  4428. return nsg;
  4429. }
  4430. EXPORT_SYMBOL_GPL(skb_to_sgvec);
  4431. /* As compared with skb_to_sgvec, skb_to_sgvec_nomark only map skb to given
  4432. * sglist without mark the sg which contain last skb data as the end.
  4433. * So the caller can mannipulate sg list as will when padding new data after
  4434. * the first call without calling sg_unmark_end to expend sg list.
  4435. *
  4436. * Scenario to use skb_to_sgvec_nomark:
  4437. * 1. sg_init_table
  4438. * 2. skb_to_sgvec_nomark(payload1)
  4439. * 3. skb_to_sgvec_nomark(payload2)
  4440. *
  4441. * This is equivalent to:
  4442. * 1. sg_init_table
  4443. * 2. skb_to_sgvec(payload1)
  4444. * 3. sg_unmark_end
  4445. * 4. skb_to_sgvec(payload2)
  4446. *
  4447. * When mapping multiple payload conditionally, skb_to_sgvec_nomark
  4448. * is more preferable.
  4449. */
  4450. int skb_to_sgvec_nomark(struct sk_buff *skb, struct scatterlist *sg,
  4451. int offset, int len)
  4452. {
  4453. return __skb_to_sgvec(skb, sg, offset, len, 0);
  4454. }
  4455. EXPORT_SYMBOL_GPL(skb_to_sgvec_nomark);
  4456. /**
  4457. * skb_cow_data - Check that a socket buffer's data buffers are writable
  4458. * @skb: The socket buffer to check.
  4459. * @tailbits: Amount of trailing space to be added
  4460. * @trailer: Returned pointer to the skb where the @tailbits space begins
  4461. *
  4462. * Make sure that the data buffers attached to a socket buffer are
  4463. * writable. If they are not, private copies are made of the data buffers
  4464. * and the socket buffer is set to use these instead.
  4465. *
  4466. * If @tailbits is given, make sure that there is space to write @tailbits
  4467. * bytes of data beyond current end of socket buffer. @trailer will be
  4468. * set to point to the skb in which this space begins.
  4469. *
  4470. * The number of scatterlist elements required to completely map the
  4471. * COW'd and extended socket buffer will be returned.
  4472. */
  4473. int skb_cow_data(struct sk_buff *skb, int tailbits, struct sk_buff **trailer)
  4474. {
  4475. int copyflag;
  4476. int elt;
  4477. struct sk_buff *skb1, **skb_p;
  4478. /* If skb is cloned or its head is paged, reallocate
  4479. * head pulling out all the pages (pages are considered not writable
  4480. * at the moment even if they are anonymous).
  4481. */
  4482. if ((skb_cloned(skb) || skb_shinfo(skb)->nr_frags) &&
  4483. !__pskb_pull_tail(skb, __skb_pagelen(skb)))
  4484. return -ENOMEM;
  4485. /* Easy case. Most of packets will go this way. */
  4486. if (!skb_has_frag_list(skb)) {
  4487. /* A little of trouble, not enough of space for trailer.
  4488. * This should not happen, when stack is tuned to generate
  4489. * good frames. OK, on miss we reallocate and reserve even more
  4490. * space, 128 bytes is fair. */
  4491. if (skb_tailroom(skb) < tailbits &&
  4492. pskb_expand_head(skb, 0, tailbits-skb_tailroom(skb)+128, GFP_ATOMIC))
  4493. return -ENOMEM;
  4494. /* Voila! */
  4495. *trailer = skb;
  4496. return 1;
  4497. }
  4498. /* Misery. We are in troubles, going to mincer fragments... */
  4499. elt = 1;
  4500. skb_p = &skb_shinfo(skb)->frag_list;
  4501. copyflag = 0;
  4502. while ((skb1 = *skb_p) != NULL) {
  4503. int ntail = 0;
  4504. /* The fragment is partially pulled by someone,
  4505. * this can happen on input. Copy it and everything
  4506. * after it. */
  4507. if (skb_shared(skb1))
  4508. copyflag = 1;
  4509. /* If the skb is the last, worry about trailer. */
  4510. if (skb1->next == NULL && tailbits) {
  4511. if (skb_shinfo(skb1)->nr_frags ||
  4512. skb_has_frag_list(skb1) ||
  4513. skb_tailroom(skb1) < tailbits)
  4514. ntail = tailbits + 128;
  4515. }
  4516. if (copyflag ||
  4517. skb_cloned(skb1) ||
  4518. ntail ||
  4519. skb_shinfo(skb1)->nr_frags ||
  4520. skb_has_frag_list(skb1)) {
  4521. struct sk_buff *skb2;
  4522. /* Fuck, we are miserable poor guys... */
  4523. if (ntail == 0)
  4524. skb2 = skb_copy(skb1, GFP_ATOMIC);
  4525. else
  4526. skb2 = skb_copy_expand(skb1,
  4527. skb_headroom(skb1),
  4528. ntail,
  4529. GFP_ATOMIC);
  4530. if (unlikely(skb2 == NULL))
  4531. return -ENOMEM;
  4532. if (skb1->sk)
  4533. skb_set_owner_w(skb2, skb1->sk);
  4534. /* Looking around. Are we still alive?
  4535. * OK, link new skb, drop old one */
  4536. skb2->next = skb1->next;
  4537. *skb_p = skb2;
  4538. kfree_skb(skb1);
  4539. skb1 = skb2;
  4540. }
  4541. elt++;
  4542. *trailer = skb1;
  4543. skb_p = &skb1->next;
  4544. }
  4545. return elt;
  4546. }
  4547. EXPORT_SYMBOL_GPL(skb_cow_data);
  4548. static void sock_rmem_free(struct sk_buff *skb)
  4549. {
  4550. struct sock *sk = skb->sk;
  4551. atomic_sub(skb->truesize, &sk->sk_rmem_alloc);
  4552. }
  4553. static void skb_set_err_queue(struct sk_buff *skb)
  4554. {
  4555. /* pkt_type of skbs received on local sockets is never PACKET_OUTGOING.
  4556. * So, it is safe to (mis)use it to mark skbs on the error queue.
  4557. */
  4558. skb->pkt_type = PACKET_OUTGOING;
  4559. BUILD_BUG_ON(PACKET_OUTGOING == 0);
  4560. }
  4561. /*
  4562. * Note: We dont mem charge error packets (no sk_forward_alloc changes)
  4563. */
  4564. int sock_queue_err_skb(struct sock *sk, struct sk_buff *skb)
  4565. {
  4566. if (atomic_read(&sk->sk_rmem_alloc) + skb->truesize >=
  4567. (unsigned int)READ_ONCE(sk->sk_rcvbuf))
  4568. return -ENOMEM;
  4569. skb_orphan(skb);
  4570. skb->sk = sk;
  4571. skb->destructor = sock_rmem_free;
  4572. atomic_add(skb->truesize, &sk->sk_rmem_alloc);
  4573. skb_set_err_queue(skb);
  4574. /* before exiting rcu section, make sure dst is refcounted */
  4575. skb_dst_force(skb);
  4576. skb_queue_tail(&sk->sk_error_queue, skb);
  4577. if (!sock_flag(sk, SOCK_DEAD))
  4578. sk_error_report(sk);
  4579. return 0;
  4580. }
  4581. EXPORT_SYMBOL(sock_queue_err_skb);
  4582. static bool is_icmp_err_skb(const struct sk_buff *skb)
  4583. {
  4584. return skb && (SKB_EXT_ERR(skb)->ee.ee_origin == SO_EE_ORIGIN_ICMP ||
  4585. SKB_EXT_ERR(skb)->ee.ee_origin == SO_EE_ORIGIN_ICMP6);
  4586. }
  4587. struct sk_buff *sock_dequeue_err_skb(struct sock *sk)
  4588. {
  4589. struct sk_buff_head *q = &sk->sk_error_queue;
  4590. struct sk_buff *skb, *skb_next = NULL;
  4591. bool icmp_next = false;
  4592. unsigned long flags;
  4593. if (skb_queue_empty_lockless(q))
  4594. return NULL;
  4595. spin_lock_irqsave(&q->lock, flags);
  4596. skb = __skb_dequeue(q);
  4597. if (skb && (skb_next = skb_peek(q))) {
  4598. icmp_next = is_icmp_err_skb(skb_next);
  4599. if (icmp_next)
  4600. sk->sk_err = SKB_EXT_ERR(skb_next)->ee.ee_errno;
  4601. }
  4602. spin_unlock_irqrestore(&q->lock, flags);
  4603. if (is_icmp_err_skb(skb) && !icmp_next)
  4604. sk->sk_err = 0;
  4605. if (skb_next)
  4606. sk_error_report(sk);
  4607. return skb;
  4608. }
  4609. EXPORT_SYMBOL(sock_dequeue_err_skb);
  4610. /**
  4611. * skb_clone_sk - create clone of skb, and take reference to socket
  4612. * @skb: the skb to clone
  4613. *
  4614. * This function creates a clone of a buffer that holds a reference on
  4615. * sk_refcnt. Buffers created via this function are meant to be
  4616. * returned using sock_queue_err_skb, or free via kfree_skb.
  4617. *
  4618. * When passing buffers allocated with this function to sock_queue_err_skb
  4619. * it is necessary to wrap the call with sock_hold/sock_put in order to
  4620. * prevent the socket from being released prior to being enqueued on
  4621. * the sk_error_queue.
  4622. */
  4623. struct sk_buff *skb_clone_sk(struct sk_buff *skb)
  4624. {
  4625. struct sock *sk = skb->sk;
  4626. struct sk_buff *clone;
  4627. if (!sk || !refcount_inc_not_zero(&sk->sk_refcnt))
  4628. return NULL;
  4629. clone = skb_clone(skb, GFP_ATOMIC);
  4630. if (!clone) {
  4631. sock_put(sk);
  4632. return NULL;
  4633. }
  4634. clone->sk = sk;
  4635. clone->destructor = sock_efree;
  4636. return clone;
  4637. }
  4638. EXPORT_SYMBOL(skb_clone_sk);
  4639. static void __skb_complete_tx_timestamp(struct sk_buff *skb,
  4640. struct sock *sk,
  4641. int tstype,
  4642. bool opt_stats)
  4643. {
  4644. struct sock_exterr_skb *serr;
  4645. int err;
  4646. BUILD_BUG_ON(sizeof(struct sock_exterr_skb) > sizeof(skb->cb));
  4647. serr = SKB_EXT_ERR(skb);
  4648. memset(serr, 0, sizeof(*serr));
  4649. serr->ee.ee_errno = ENOMSG;
  4650. serr->ee.ee_origin = SO_EE_ORIGIN_TIMESTAMPING;
  4651. serr->ee.ee_info = tstype;
  4652. serr->opt_stats = opt_stats;
  4653. serr->header.h4.iif = skb->dev ? skb->dev->ifindex : 0;
  4654. if (READ_ONCE(sk->sk_tsflags) & SOF_TIMESTAMPING_OPT_ID) {
  4655. serr->ee.ee_data = skb_shinfo(skb)->tskey;
  4656. if (sk_is_tcp(sk))
  4657. serr->ee.ee_data -= atomic_read(&sk->sk_tskey);
  4658. }
  4659. err = sock_queue_err_skb(sk, skb);
  4660. if (err)
  4661. kfree_skb(skb);
  4662. }
  4663. static bool skb_may_tx_timestamp(struct sock *sk, bool tsonly)
  4664. {
  4665. bool ret;
  4666. if (likely(READ_ONCE(sysctl_tstamp_allow_data) || tsonly))
  4667. return true;
  4668. read_lock_bh(&sk->sk_callback_lock);
  4669. ret = sk->sk_socket && sk->sk_socket->file &&
  4670. file_ns_capable(sk->sk_socket->file, &init_user_ns, CAP_NET_RAW);
  4671. read_unlock_bh(&sk->sk_callback_lock);
  4672. return ret;
  4673. }
  4674. void skb_complete_tx_timestamp(struct sk_buff *skb,
  4675. struct skb_shared_hwtstamps *hwtstamps)
  4676. {
  4677. struct sock *sk = skb->sk;
  4678. if (!skb_may_tx_timestamp(sk, false))
  4679. goto err;
  4680. /* Take a reference to prevent skb_orphan() from freeing the socket,
  4681. * but only if the socket refcount is not zero.
  4682. */
  4683. if (likely(refcount_inc_not_zero(&sk->sk_refcnt))) {
  4684. *skb_hwtstamps(skb) = *hwtstamps;
  4685. __skb_complete_tx_timestamp(skb, sk, SCM_TSTAMP_SND, false);
  4686. sock_put(sk);
  4687. return;
  4688. }
  4689. err:
  4690. kfree_skb(skb);
  4691. }
  4692. EXPORT_SYMBOL_GPL(skb_complete_tx_timestamp);
  4693. void __skb_tstamp_tx(struct sk_buff *orig_skb,
  4694. const struct sk_buff *ack_skb,
  4695. struct skb_shared_hwtstamps *hwtstamps,
  4696. struct sock *sk, int tstype)
  4697. {
  4698. struct sk_buff *skb;
  4699. bool tsonly, opt_stats = false;
  4700. u32 tsflags;
  4701. if (!sk)
  4702. return;
  4703. tsflags = READ_ONCE(sk->sk_tsflags);
  4704. if (!hwtstamps && !(tsflags & SOF_TIMESTAMPING_OPT_TX_SWHW) &&
  4705. skb_shinfo(orig_skb)->tx_flags & SKBTX_IN_PROGRESS)
  4706. return;
  4707. tsonly = tsflags & SOF_TIMESTAMPING_OPT_TSONLY;
  4708. if (!skb_may_tx_timestamp(sk, tsonly))
  4709. return;
  4710. if (tsonly) {
  4711. #ifdef CONFIG_INET
  4712. if ((tsflags & SOF_TIMESTAMPING_OPT_STATS) &&
  4713. sk_is_tcp(sk)) {
  4714. skb = tcp_get_timestamping_opt_stats(sk, orig_skb,
  4715. ack_skb);
  4716. opt_stats = true;
  4717. } else
  4718. #endif
  4719. skb = alloc_skb(0, GFP_ATOMIC);
  4720. } else {
  4721. skb = skb_clone(orig_skb, GFP_ATOMIC);
  4722. if (skb_orphan_frags_rx(skb, GFP_ATOMIC)) {
  4723. kfree_skb(skb);
  4724. return;
  4725. }
  4726. }
  4727. if (!skb)
  4728. return;
  4729. if (tsonly) {
  4730. skb_shinfo(skb)->tx_flags |= skb_shinfo(orig_skb)->tx_flags &
  4731. SKBTX_ANY_TSTAMP;
  4732. skb_shinfo(skb)->tskey = skb_shinfo(orig_skb)->tskey;
  4733. }
  4734. if (hwtstamps)
  4735. *skb_hwtstamps(skb) = *hwtstamps;
  4736. else
  4737. __net_timestamp(skb);
  4738. __skb_complete_tx_timestamp(skb, sk, tstype, opt_stats);
  4739. }
  4740. EXPORT_SYMBOL_GPL(__skb_tstamp_tx);
  4741. void skb_tstamp_tx(struct sk_buff *orig_skb,
  4742. struct skb_shared_hwtstamps *hwtstamps)
  4743. {
  4744. return __skb_tstamp_tx(orig_skb, NULL, hwtstamps, orig_skb->sk,
  4745. SCM_TSTAMP_SND);
  4746. }
  4747. EXPORT_SYMBOL_GPL(skb_tstamp_tx);
  4748. #ifdef CONFIG_WIRELESS
  4749. void skb_complete_wifi_ack(struct sk_buff *skb, bool acked)
  4750. {
  4751. struct sock *sk = skb->sk;
  4752. struct sock_exterr_skb *serr;
  4753. int err = 1;
  4754. skb->wifi_acked_valid = 1;
  4755. skb->wifi_acked = acked;
  4756. serr = SKB_EXT_ERR(skb);
  4757. memset(serr, 0, sizeof(*serr));
  4758. serr->ee.ee_errno = ENOMSG;
  4759. serr->ee.ee_origin = SO_EE_ORIGIN_TXSTATUS;
  4760. /* Take a reference to prevent skb_orphan() from freeing the socket,
  4761. * but only if the socket refcount is not zero.
  4762. */
  4763. if (likely(refcount_inc_not_zero(&sk->sk_refcnt))) {
  4764. err = sock_queue_err_skb(sk, skb);
  4765. sock_put(sk);
  4766. }
  4767. if (err)
  4768. kfree_skb(skb);
  4769. }
  4770. EXPORT_SYMBOL_GPL(skb_complete_wifi_ack);
  4771. #endif /* CONFIG_WIRELESS */
  4772. /**
  4773. * skb_partial_csum_set - set up and verify partial csum values for packet
  4774. * @skb: the skb to set
  4775. * @start: the number of bytes after skb->data to start checksumming.
  4776. * @off: the offset from start to place the checksum.
  4777. *
  4778. * For untrusted partially-checksummed packets, we need to make sure the values
  4779. * for skb->csum_start and skb->csum_offset are valid so we don't oops.
  4780. *
  4781. * This function checks and sets those values and skb->ip_summed: if this
  4782. * returns false you should drop the packet.
  4783. */
  4784. bool skb_partial_csum_set(struct sk_buff *skb, u16 start, u16 off)
  4785. {
  4786. u32 csum_end = (u32)start + (u32)off + sizeof(__sum16);
  4787. u32 csum_start = skb_headroom(skb) + (u32)start;
  4788. if (unlikely(csum_start >= U16_MAX || csum_end > skb_headlen(skb))) {
  4789. net_warn_ratelimited("bad partial csum: csum=%u/%u headroom=%u headlen=%u\n",
  4790. start, off, skb_headroom(skb), skb_headlen(skb));
  4791. return false;
  4792. }
  4793. skb->ip_summed = CHECKSUM_PARTIAL;
  4794. skb->csum_start = csum_start;
  4795. skb->csum_offset = off;
  4796. skb->transport_header = csum_start;
  4797. return true;
  4798. }
  4799. EXPORT_SYMBOL_GPL(skb_partial_csum_set);
  4800. static int skb_maybe_pull_tail(struct sk_buff *skb, unsigned int len,
  4801. unsigned int max)
  4802. {
  4803. if (skb_headlen(skb) >= len)
  4804. return 0;
  4805. /* If we need to pullup then pullup to the max, so we
  4806. * won't need to do it again.
  4807. */
  4808. if (max > skb->len)
  4809. max = skb->len;
  4810. if (__pskb_pull_tail(skb, max - skb_headlen(skb)) == NULL)
  4811. return -ENOMEM;
  4812. if (skb_headlen(skb) < len)
  4813. return -EPROTO;
  4814. return 0;
  4815. }
  4816. #define MAX_TCP_HDR_LEN (15 * 4)
  4817. static __sum16 *skb_checksum_setup_ip(struct sk_buff *skb,
  4818. typeof(IPPROTO_IP) proto,
  4819. unsigned int off)
  4820. {
  4821. int err;
  4822. switch (proto) {
  4823. case IPPROTO_TCP:
  4824. err = skb_maybe_pull_tail(skb, off + sizeof(struct tcphdr),
  4825. off + MAX_TCP_HDR_LEN);
  4826. if (!err && !skb_partial_csum_set(skb, off,
  4827. offsetof(struct tcphdr,
  4828. check)))
  4829. err = -EPROTO;
  4830. return err ? ERR_PTR(err) : &tcp_hdr(skb)->check;
  4831. case IPPROTO_UDP:
  4832. err = skb_maybe_pull_tail(skb, off + sizeof(struct udphdr),
  4833. off + sizeof(struct udphdr));
  4834. if (!err && !skb_partial_csum_set(skb, off,
  4835. offsetof(struct udphdr,
  4836. check)))
  4837. err = -EPROTO;
  4838. return err ? ERR_PTR(err) : &udp_hdr(skb)->check;
  4839. }
  4840. return ERR_PTR(-EPROTO);
  4841. }
  4842. /* This value should be large enough to cover a tagged ethernet header plus
  4843. * maximally sized IP and TCP or UDP headers.
  4844. */
  4845. #define MAX_IP_HDR_LEN 128
  4846. static int skb_checksum_setup_ipv4(struct sk_buff *skb, bool recalculate)
  4847. {
  4848. unsigned int off;
  4849. bool fragment;
  4850. __sum16 *csum;
  4851. int err;
  4852. fragment = false;
  4853. err = skb_maybe_pull_tail(skb,
  4854. sizeof(struct iphdr),
  4855. MAX_IP_HDR_LEN);
  4856. if (err < 0)
  4857. goto out;
  4858. if (ip_is_fragment(ip_hdr(skb)))
  4859. fragment = true;
  4860. off = ip_hdrlen(skb);
  4861. err = -EPROTO;
  4862. if (fragment)
  4863. goto out;
  4864. csum = skb_checksum_setup_ip(skb, ip_hdr(skb)->protocol, off);
  4865. if (IS_ERR(csum))
  4866. return PTR_ERR(csum);
  4867. if (recalculate)
  4868. *csum = ~csum_tcpudp_magic(ip_hdr(skb)->saddr,
  4869. ip_hdr(skb)->daddr,
  4870. skb->len - off,
  4871. ip_hdr(skb)->protocol, 0);
  4872. err = 0;
  4873. out:
  4874. return err;
  4875. }
  4876. /* This value should be large enough to cover a tagged ethernet header plus
  4877. * an IPv6 header, all options, and a maximal TCP or UDP header.
  4878. */
  4879. #define MAX_IPV6_HDR_LEN 256
  4880. #define OPT_HDR(type, skb, off) \
  4881. (type *)(skb_network_header(skb) + (off))
  4882. static int skb_checksum_setup_ipv6(struct sk_buff *skb, bool recalculate)
  4883. {
  4884. int err;
  4885. u8 nexthdr;
  4886. unsigned int off;
  4887. unsigned int len;
  4888. bool fragment;
  4889. bool done;
  4890. __sum16 *csum;
  4891. fragment = false;
  4892. done = false;
  4893. off = sizeof(struct ipv6hdr);
  4894. err = skb_maybe_pull_tail(skb, off, MAX_IPV6_HDR_LEN);
  4895. if (err < 0)
  4896. goto out;
  4897. nexthdr = ipv6_hdr(skb)->nexthdr;
  4898. len = sizeof(struct ipv6hdr) + ntohs(ipv6_hdr(skb)->payload_len);
  4899. while (off <= len && !done) {
  4900. switch (nexthdr) {
  4901. case IPPROTO_DSTOPTS:
  4902. case IPPROTO_HOPOPTS:
  4903. case IPPROTO_ROUTING: {
  4904. struct ipv6_opt_hdr *hp;
  4905. err = skb_maybe_pull_tail(skb,
  4906. off +
  4907. sizeof(struct ipv6_opt_hdr),
  4908. MAX_IPV6_HDR_LEN);
  4909. if (err < 0)
  4910. goto out;
  4911. hp = OPT_HDR(struct ipv6_opt_hdr, skb, off);
  4912. nexthdr = hp->nexthdr;
  4913. off += ipv6_optlen(hp);
  4914. break;
  4915. }
  4916. case IPPROTO_AH: {
  4917. struct ip_auth_hdr *hp;
  4918. err = skb_maybe_pull_tail(skb,
  4919. off +
  4920. sizeof(struct ip_auth_hdr),
  4921. MAX_IPV6_HDR_LEN);
  4922. if (err < 0)
  4923. goto out;
  4924. hp = OPT_HDR(struct ip_auth_hdr, skb, off);
  4925. nexthdr = hp->nexthdr;
  4926. off += ipv6_authlen(hp);
  4927. break;
  4928. }
  4929. case IPPROTO_FRAGMENT: {
  4930. struct frag_hdr *hp;
  4931. err = skb_maybe_pull_tail(skb,
  4932. off +
  4933. sizeof(struct frag_hdr),
  4934. MAX_IPV6_HDR_LEN);
  4935. if (err < 0)
  4936. goto out;
  4937. hp = OPT_HDR(struct frag_hdr, skb, off);
  4938. if (hp->frag_off & htons(IP6_OFFSET | IP6_MF))
  4939. fragment = true;
  4940. nexthdr = hp->nexthdr;
  4941. off += sizeof(struct frag_hdr);
  4942. break;
  4943. }
  4944. default:
  4945. done = true;
  4946. break;
  4947. }
  4948. }
  4949. err = -EPROTO;
  4950. if (!done || fragment)
  4951. goto out;
  4952. csum = skb_checksum_setup_ip(skb, nexthdr, off);
  4953. if (IS_ERR(csum))
  4954. return PTR_ERR(csum);
  4955. if (recalculate)
  4956. *csum = ~csum_ipv6_magic(&ipv6_hdr(skb)->saddr,
  4957. &ipv6_hdr(skb)->daddr,
  4958. skb->len - off, nexthdr, 0);
  4959. err = 0;
  4960. out:
  4961. return err;
  4962. }
  4963. /**
  4964. * skb_checksum_setup - set up partial checksum offset
  4965. * @skb: the skb to set up
  4966. * @recalculate: if true the pseudo-header checksum will be recalculated
  4967. */
  4968. int skb_checksum_setup(struct sk_buff *skb, bool recalculate)
  4969. {
  4970. int err;
  4971. switch (skb->protocol) {
  4972. case htons(ETH_P_IP):
  4973. err = skb_checksum_setup_ipv4(skb, recalculate);
  4974. break;
  4975. case htons(ETH_P_IPV6):
  4976. err = skb_checksum_setup_ipv6(skb, recalculate);
  4977. break;
  4978. default:
  4979. err = -EPROTO;
  4980. break;
  4981. }
  4982. return err;
  4983. }
  4984. EXPORT_SYMBOL(skb_checksum_setup);
  4985. /**
  4986. * skb_checksum_maybe_trim - maybe trims the given skb
  4987. * @skb: the skb to check
  4988. * @transport_len: the data length beyond the network header
  4989. *
  4990. * Checks whether the given skb has data beyond the given transport length.
  4991. * If so, returns a cloned skb trimmed to this transport length.
  4992. * Otherwise returns the provided skb. Returns NULL in error cases
  4993. * (e.g. transport_len exceeds skb length or out-of-memory).
  4994. *
  4995. * Caller needs to set the skb transport header and free any returned skb if it
  4996. * differs from the provided skb.
  4997. */
  4998. static struct sk_buff *skb_checksum_maybe_trim(struct sk_buff *skb,
  4999. unsigned int transport_len)
  5000. {
  5001. struct sk_buff *skb_chk;
  5002. unsigned int len = skb_transport_offset(skb) + transport_len;
  5003. int ret;
  5004. if (skb->len < len)
  5005. return NULL;
  5006. else if (skb->len == len)
  5007. return skb;
  5008. skb_chk = skb_clone(skb, GFP_ATOMIC);
  5009. if (!skb_chk)
  5010. return NULL;
  5011. ret = pskb_trim_rcsum(skb_chk, len);
  5012. if (ret) {
  5013. kfree_skb(skb_chk);
  5014. return NULL;
  5015. }
  5016. return skb_chk;
  5017. }
  5018. /**
  5019. * skb_checksum_trimmed - validate checksum of an skb
  5020. * @skb: the skb to check
  5021. * @transport_len: the data length beyond the network header
  5022. * @skb_chkf: checksum function to use
  5023. *
  5024. * Applies the given checksum function skb_chkf to the provided skb.
  5025. * Returns a checked and maybe trimmed skb. Returns NULL on error.
  5026. *
  5027. * If the skb has data beyond the given transport length, then a
  5028. * trimmed & cloned skb is checked and returned.
  5029. *
  5030. * Caller needs to set the skb transport header and free any returned skb if it
  5031. * differs from the provided skb.
  5032. */
  5033. struct sk_buff *skb_checksum_trimmed(struct sk_buff *skb,
  5034. unsigned int transport_len,
  5035. __sum16(*skb_chkf)(struct sk_buff *skb))
  5036. {
  5037. struct sk_buff *skb_chk;
  5038. unsigned int offset = skb_transport_offset(skb);
  5039. __sum16 ret;
  5040. skb_chk = skb_checksum_maybe_trim(skb, transport_len);
  5041. if (!skb_chk)
  5042. goto err;
  5043. if (!pskb_may_pull(skb_chk, offset))
  5044. goto err;
  5045. skb_pull_rcsum(skb_chk, offset);
  5046. ret = skb_chkf(skb_chk);
  5047. skb_push_rcsum(skb_chk, offset);
  5048. if (ret)
  5049. goto err;
  5050. return skb_chk;
  5051. err:
  5052. if (skb_chk && skb_chk != skb)
  5053. kfree_skb(skb_chk);
  5054. return NULL;
  5055. }
  5056. EXPORT_SYMBOL(skb_checksum_trimmed);
  5057. void __skb_warn_lro_forwarding(const struct sk_buff *skb)
  5058. {
  5059. net_warn_ratelimited("%s: received packets cannot be forwarded while LRO is enabled\n",
  5060. skb->dev->name);
  5061. }
  5062. EXPORT_SYMBOL(__skb_warn_lro_forwarding);
  5063. void kfree_skb_partial(struct sk_buff *skb, bool head_stolen)
  5064. {
  5065. if (head_stolen) {
  5066. skb_release_head_state(skb);
  5067. kmem_cache_free(net_hotdata.skbuff_cache, skb);
  5068. } else {
  5069. __kfree_skb(skb);
  5070. }
  5071. }
  5072. EXPORT_SYMBOL(kfree_skb_partial);
  5073. /**
  5074. * skb_try_coalesce - try to merge skb to prior one
  5075. * @to: prior buffer
  5076. * @from: buffer to add
  5077. * @fragstolen: pointer to boolean
  5078. * @delta_truesize: how much more was allocated than was requested
  5079. */
  5080. bool skb_try_coalesce(struct sk_buff *to, struct sk_buff *from,
  5081. bool *fragstolen, int *delta_truesize)
  5082. {
  5083. struct skb_shared_info *to_shinfo, *from_shinfo;
  5084. int i, delta, len = from->len;
  5085. *fragstolen = false;
  5086. if (skb_cloned(to))
  5087. return false;
  5088. /* In general, avoid mixing page_pool and non-page_pool allocated
  5089. * pages within the same SKB. In theory we could take full
  5090. * references if @from is cloned and !@to->pp_recycle but its
  5091. * tricky (due to potential race with the clone disappearing) and
  5092. * rare, so not worth dealing with.
  5093. */
  5094. if (to->pp_recycle != from->pp_recycle)
  5095. return false;
  5096. if (skb_frags_readable(from) != skb_frags_readable(to))
  5097. return false;
  5098. if (len <= skb_tailroom(to) && skb_frags_readable(from)) {
  5099. if (len)
  5100. BUG_ON(skb_copy_bits(from, 0, skb_put(to, len), len));
  5101. *delta_truesize = 0;
  5102. return true;
  5103. }
  5104. to_shinfo = skb_shinfo(to);
  5105. from_shinfo = skb_shinfo(from);
  5106. if (to_shinfo->frag_list || from_shinfo->frag_list)
  5107. return false;
  5108. if (skb_zcopy(to) || skb_zcopy(from))
  5109. return false;
  5110. if (skb_headlen(from) != 0) {
  5111. struct page *page;
  5112. unsigned int offset;
  5113. if (to_shinfo->nr_frags +
  5114. from_shinfo->nr_frags >= MAX_SKB_FRAGS)
  5115. return false;
  5116. if (skb_head_is_locked(from))
  5117. return false;
  5118. delta = from->truesize - SKB_DATA_ALIGN(sizeof(struct sk_buff));
  5119. page = virt_to_head_page(from->head);
  5120. offset = from->data - (unsigned char *)page_address(page);
  5121. skb_fill_page_desc(to, to_shinfo->nr_frags,
  5122. page, offset, skb_headlen(from));
  5123. *fragstolen = true;
  5124. } else {
  5125. if (to_shinfo->nr_frags +
  5126. from_shinfo->nr_frags > MAX_SKB_FRAGS)
  5127. return false;
  5128. delta = from->truesize - SKB_TRUESIZE(skb_end_offset(from));
  5129. }
  5130. WARN_ON_ONCE(delta < len);
  5131. memcpy(to_shinfo->frags + to_shinfo->nr_frags,
  5132. from_shinfo->frags,
  5133. from_shinfo->nr_frags * sizeof(skb_frag_t));
  5134. to_shinfo->nr_frags += from_shinfo->nr_frags;
  5135. if (!skb_cloned(from))
  5136. from_shinfo->nr_frags = 0;
  5137. /* if the skb is not cloned this does nothing
  5138. * since we set nr_frags to 0.
  5139. */
  5140. if (skb_pp_frag_ref(from)) {
  5141. for (i = 0; i < from_shinfo->nr_frags; i++)
  5142. __skb_frag_ref(&from_shinfo->frags[i]);
  5143. }
  5144. to->truesize += delta;
  5145. to->len += len;
  5146. to->data_len += len;
  5147. *delta_truesize = delta;
  5148. return true;
  5149. }
  5150. EXPORT_SYMBOL(skb_try_coalesce);
  5151. /**
  5152. * skb_scrub_packet - scrub an skb
  5153. *
  5154. * @skb: buffer to clean
  5155. * @xnet: packet is crossing netns
  5156. *
  5157. * skb_scrub_packet can be used after encapsulating or decapsulating a packet
  5158. * into/from a tunnel. Some information have to be cleared during these
  5159. * operations.
  5160. * skb_scrub_packet can also be used to clean a skb before injecting it in
  5161. * another namespace (@xnet == true). We have to clear all information in the
  5162. * skb that could impact namespace isolation.
  5163. */
  5164. void skb_scrub_packet(struct sk_buff *skb, bool xnet)
  5165. {
  5166. skb->pkt_type = PACKET_HOST;
  5167. skb->skb_iif = 0;
  5168. skb->ignore_df = 0;
  5169. skb_dst_drop(skb);
  5170. skb_ext_reset(skb);
  5171. nf_reset_ct(skb);
  5172. nf_reset_trace(skb);
  5173. #ifdef CONFIG_NET_SWITCHDEV
  5174. skb->offload_fwd_mark = 0;
  5175. skb->offload_l3_fwd_mark = 0;
  5176. #endif
  5177. if (!xnet)
  5178. return;
  5179. ipvs_reset(skb);
  5180. skb->mark = 0;
  5181. skb_clear_tstamp(skb);
  5182. }
  5183. EXPORT_SYMBOL_GPL(skb_scrub_packet);
  5184. static struct sk_buff *skb_reorder_vlan_header(struct sk_buff *skb)
  5185. {
  5186. int mac_len, meta_len;
  5187. void *meta;
  5188. if (skb_cow(skb, skb_headroom(skb)) < 0) {
  5189. kfree_skb(skb);
  5190. return NULL;
  5191. }
  5192. mac_len = skb->data - skb_mac_header(skb);
  5193. if (likely(mac_len > VLAN_HLEN + ETH_TLEN)) {
  5194. memmove(skb_mac_header(skb) + VLAN_HLEN, skb_mac_header(skb),
  5195. mac_len - VLAN_HLEN - ETH_TLEN);
  5196. }
  5197. meta_len = skb_metadata_len(skb);
  5198. if (meta_len) {
  5199. meta = skb_metadata_end(skb) - meta_len;
  5200. memmove(meta + VLAN_HLEN, meta, meta_len);
  5201. }
  5202. skb->mac_header += VLAN_HLEN;
  5203. return skb;
  5204. }
  5205. struct sk_buff *skb_vlan_untag(struct sk_buff *skb)
  5206. {
  5207. struct vlan_hdr *vhdr;
  5208. u16 vlan_tci;
  5209. if (unlikely(skb_vlan_tag_present(skb))) {
  5210. /* vlan_tci is already set-up so leave this for another time */
  5211. return skb;
  5212. }
  5213. skb = skb_share_check(skb, GFP_ATOMIC);
  5214. if (unlikely(!skb))
  5215. goto err_free;
  5216. /* We may access the two bytes after vlan_hdr in vlan_set_encap_proto(). */
  5217. if (unlikely(!pskb_may_pull(skb, VLAN_HLEN + sizeof(unsigned short))))
  5218. goto err_free;
  5219. vhdr = (struct vlan_hdr *)skb->data;
  5220. vlan_tci = ntohs(vhdr->h_vlan_TCI);
  5221. __vlan_hwaccel_put_tag(skb, skb->protocol, vlan_tci);
  5222. skb_pull_rcsum(skb, VLAN_HLEN);
  5223. vlan_set_encap_proto(skb, vhdr);
  5224. skb = skb_reorder_vlan_header(skb);
  5225. if (unlikely(!skb))
  5226. goto err_free;
  5227. skb_reset_network_header(skb);
  5228. if (!skb_transport_header_was_set(skb))
  5229. skb_reset_transport_header(skb);
  5230. skb_reset_mac_len(skb);
  5231. return skb;
  5232. err_free:
  5233. kfree_skb(skb);
  5234. return NULL;
  5235. }
  5236. EXPORT_SYMBOL(skb_vlan_untag);
  5237. int skb_ensure_writable(struct sk_buff *skb, unsigned int write_len)
  5238. {
  5239. if (!pskb_may_pull(skb, write_len))
  5240. return -ENOMEM;
  5241. if (!skb_frags_readable(skb))
  5242. return -EFAULT;
  5243. if (!skb_cloned(skb) || skb_clone_writable(skb, write_len))
  5244. return 0;
  5245. return pskb_expand_head(skb, 0, 0, GFP_ATOMIC);
  5246. }
  5247. EXPORT_SYMBOL(skb_ensure_writable);
  5248. int skb_ensure_writable_head_tail(struct sk_buff *skb, struct net_device *dev)
  5249. {
  5250. int needed_headroom = dev->needed_headroom;
  5251. int needed_tailroom = dev->needed_tailroom;
  5252. /* For tail taggers, we need to pad short frames ourselves, to ensure
  5253. * that the tail tag does not fail at its role of being at the end of
  5254. * the packet, once the conduit interface pads the frame. Account for
  5255. * that pad length here, and pad later.
  5256. */
  5257. if (unlikely(needed_tailroom && skb->len < ETH_ZLEN))
  5258. needed_tailroom += ETH_ZLEN - skb->len;
  5259. /* skb_headroom() returns unsigned int... */
  5260. needed_headroom = max_t(int, needed_headroom - skb_headroom(skb), 0);
  5261. needed_tailroom = max_t(int, needed_tailroom - skb_tailroom(skb), 0);
  5262. if (likely(!needed_headroom && !needed_tailroom && !skb_cloned(skb)))
  5263. /* No reallocation needed, yay! */
  5264. return 0;
  5265. return pskb_expand_head(skb, needed_headroom, needed_tailroom,
  5266. GFP_ATOMIC);
  5267. }
  5268. EXPORT_SYMBOL(skb_ensure_writable_head_tail);
  5269. /* remove VLAN header from packet and update csum accordingly.
  5270. * expects a non skb_vlan_tag_present skb with a vlan tag payload
  5271. */
  5272. int __skb_vlan_pop(struct sk_buff *skb, u16 *vlan_tci)
  5273. {
  5274. int offset = skb->data - skb_mac_header(skb);
  5275. int err;
  5276. if (WARN_ONCE(offset,
  5277. "__skb_vlan_pop got skb with skb->data not at mac header (offset %d)\n",
  5278. offset)) {
  5279. return -EINVAL;
  5280. }
  5281. err = skb_ensure_writable(skb, VLAN_ETH_HLEN);
  5282. if (unlikely(err))
  5283. return err;
  5284. skb_postpull_rcsum(skb, skb->data + (2 * ETH_ALEN), VLAN_HLEN);
  5285. vlan_remove_tag(skb, vlan_tci);
  5286. skb->mac_header += VLAN_HLEN;
  5287. if (skb_network_offset(skb) < ETH_HLEN)
  5288. skb_set_network_header(skb, ETH_HLEN);
  5289. skb_reset_mac_len(skb);
  5290. return err;
  5291. }
  5292. EXPORT_SYMBOL(__skb_vlan_pop);
  5293. /* Pop a vlan tag either from hwaccel or from payload.
  5294. * Expects skb->data at mac header.
  5295. */
  5296. int skb_vlan_pop(struct sk_buff *skb)
  5297. {
  5298. u16 vlan_tci;
  5299. __be16 vlan_proto;
  5300. int err;
  5301. if (likely(skb_vlan_tag_present(skb))) {
  5302. __vlan_hwaccel_clear_tag(skb);
  5303. } else {
  5304. if (unlikely(!eth_type_vlan(skb->protocol)))
  5305. return 0;
  5306. err = __skb_vlan_pop(skb, &vlan_tci);
  5307. if (err)
  5308. return err;
  5309. }
  5310. /* move next vlan tag to hw accel tag */
  5311. if (likely(!eth_type_vlan(skb->protocol)))
  5312. return 0;
  5313. vlan_proto = skb->protocol;
  5314. err = __skb_vlan_pop(skb, &vlan_tci);
  5315. if (unlikely(err))
  5316. return err;
  5317. __vlan_hwaccel_put_tag(skb, vlan_proto, vlan_tci);
  5318. return 0;
  5319. }
  5320. EXPORT_SYMBOL(skb_vlan_pop);
  5321. /* Push a vlan tag either into hwaccel or into payload (if hwaccel tag present).
  5322. * Expects skb->data at mac header.
  5323. */
  5324. int skb_vlan_push(struct sk_buff *skb, __be16 vlan_proto, u16 vlan_tci)
  5325. {
  5326. if (skb_vlan_tag_present(skb)) {
  5327. int offset = skb->data - skb_mac_header(skb);
  5328. int err;
  5329. if (WARN_ONCE(offset,
  5330. "skb_vlan_push got skb with skb->data not at mac header (offset %d)\n",
  5331. offset)) {
  5332. return -EINVAL;
  5333. }
  5334. err = __vlan_insert_tag(skb, skb->vlan_proto,
  5335. skb_vlan_tag_get(skb));
  5336. if (err)
  5337. return err;
  5338. skb->protocol = skb->vlan_proto;
  5339. skb->network_header -= VLAN_HLEN;
  5340. skb_postpush_rcsum(skb, skb->data + (2 * ETH_ALEN), VLAN_HLEN);
  5341. }
  5342. __vlan_hwaccel_put_tag(skb, vlan_proto, vlan_tci);
  5343. return 0;
  5344. }
  5345. EXPORT_SYMBOL(skb_vlan_push);
  5346. /**
  5347. * skb_eth_pop() - Drop the Ethernet header at the head of a packet
  5348. *
  5349. * @skb: Socket buffer to modify
  5350. *
  5351. * Drop the Ethernet header of @skb.
  5352. *
  5353. * Expects that skb->data points to the mac header and that no VLAN tags are
  5354. * present.
  5355. *
  5356. * Returns 0 on success, -errno otherwise.
  5357. */
  5358. int skb_eth_pop(struct sk_buff *skb)
  5359. {
  5360. if (!pskb_may_pull(skb, ETH_HLEN) || skb_vlan_tagged(skb) ||
  5361. skb_network_offset(skb) < ETH_HLEN)
  5362. return -EPROTO;
  5363. skb_pull_rcsum(skb, ETH_HLEN);
  5364. skb_reset_mac_header(skb);
  5365. skb_reset_mac_len(skb);
  5366. return 0;
  5367. }
  5368. EXPORT_SYMBOL(skb_eth_pop);
  5369. /**
  5370. * skb_eth_push() - Add a new Ethernet header at the head of a packet
  5371. *
  5372. * @skb: Socket buffer to modify
  5373. * @dst: Destination MAC address of the new header
  5374. * @src: Source MAC address of the new header
  5375. *
  5376. * Prepend @skb with a new Ethernet header.
  5377. *
  5378. * Expects that skb->data points to the mac header, which must be empty.
  5379. *
  5380. * Returns 0 on success, -errno otherwise.
  5381. */
  5382. int skb_eth_push(struct sk_buff *skb, const unsigned char *dst,
  5383. const unsigned char *src)
  5384. {
  5385. struct ethhdr *eth;
  5386. int err;
  5387. if (skb_network_offset(skb) || skb_vlan_tag_present(skb))
  5388. return -EPROTO;
  5389. err = skb_cow_head(skb, sizeof(*eth));
  5390. if (err < 0)
  5391. return err;
  5392. skb_push(skb, sizeof(*eth));
  5393. skb_reset_mac_header(skb);
  5394. skb_reset_mac_len(skb);
  5395. eth = eth_hdr(skb);
  5396. ether_addr_copy(eth->h_dest, dst);
  5397. ether_addr_copy(eth->h_source, src);
  5398. eth->h_proto = skb->protocol;
  5399. skb_postpush_rcsum(skb, eth, sizeof(*eth));
  5400. return 0;
  5401. }
  5402. EXPORT_SYMBOL(skb_eth_push);
  5403. /* Update the ethertype of hdr and the skb csum value if required. */
  5404. static void skb_mod_eth_type(struct sk_buff *skb, struct ethhdr *hdr,
  5405. __be16 ethertype)
  5406. {
  5407. if (skb->ip_summed == CHECKSUM_COMPLETE) {
  5408. __be16 diff[] = { ~hdr->h_proto, ethertype };
  5409. skb->csum = csum_partial((char *)diff, sizeof(diff), skb->csum);
  5410. }
  5411. hdr->h_proto = ethertype;
  5412. }
  5413. /**
  5414. * skb_mpls_push() - push a new MPLS header after mac_len bytes from start of
  5415. * the packet
  5416. *
  5417. * @skb: buffer
  5418. * @mpls_lse: MPLS label stack entry to push
  5419. * @mpls_proto: ethertype of the new MPLS header (expects 0x8847 or 0x8848)
  5420. * @mac_len: length of the MAC header
  5421. * @ethernet: flag to indicate if the resulting packet after skb_mpls_push is
  5422. * ethernet
  5423. *
  5424. * Expects skb->data at mac header.
  5425. *
  5426. * Returns 0 on success, -errno otherwise.
  5427. */
  5428. int skb_mpls_push(struct sk_buff *skb, __be32 mpls_lse, __be16 mpls_proto,
  5429. int mac_len, bool ethernet)
  5430. {
  5431. struct mpls_shim_hdr *lse;
  5432. int err;
  5433. if (unlikely(!eth_p_mpls(mpls_proto)))
  5434. return -EINVAL;
  5435. /* Networking stack does not allow simultaneous Tunnel and MPLS GSO. */
  5436. if (skb->encapsulation)
  5437. return -EINVAL;
  5438. err = skb_cow_head(skb, MPLS_HLEN);
  5439. if (unlikely(err))
  5440. return err;
  5441. if (!skb->inner_protocol) {
  5442. skb_set_inner_network_header(skb, skb_network_offset(skb));
  5443. skb_set_inner_protocol(skb, skb->protocol);
  5444. }
  5445. skb_push(skb, MPLS_HLEN);
  5446. memmove(skb_mac_header(skb) - MPLS_HLEN, skb_mac_header(skb),
  5447. mac_len);
  5448. skb_reset_mac_header(skb);
  5449. skb_set_network_header(skb, mac_len);
  5450. skb_reset_mac_len(skb);
  5451. lse = mpls_hdr(skb);
  5452. lse->label_stack_entry = mpls_lse;
  5453. skb_postpush_rcsum(skb, lse, MPLS_HLEN);
  5454. if (ethernet && mac_len >= ETH_HLEN)
  5455. skb_mod_eth_type(skb, eth_hdr(skb), mpls_proto);
  5456. skb->protocol = mpls_proto;
  5457. return 0;
  5458. }
  5459. EXPORT_SYMBOL_GPL(skb_mpls_push);
  5460. /**
  5461. * skb_mpls_pop() - pop the outermost MPLS header
  5462. *
  5463. * @skb: buffer
  5464. * @next_proto: ethertype of header after popped MPLS header
  5465. * @mac_len: length of the MAC header
  5466. * @ethernet: flag to indicate if the packet is ethernet
  5467. *
  5468. * Expects skb->data at mac header.
  5469. *
  5470. * Returns 0 on success, -errno otherwise.
  5471. */
  5472. int skb_mpls_pop(struct sk_buff *skb, __be16 next_proto, int mac_len,
  5473. bool ethernet)
  5474. {
  5475. int err;
  5476. if (unlikely(!eth_p_mpls(skb->protocol)))
  5477. return 0;
  5478. err = skb_ensure_writable(skb, mac_len + MPLS_HLEN);
  5479. if (unlikely(err))
  5480. return err;
  5481. skb_postpull_rcsum(skb, mpls_hdr(skb), MPLS_HLEN);
  5482. memmove(skb_mac_header(skb) + MPLS_HLEN, skb_mac_header(skb),
  5483. mac_len);
  5484. __skb_pull(skb, MPLS_HLEN);
  5485. skb_reset_mac_header(skb);
  5486. skb_set_network_header(skb, mac_len);
  5487. if (ethernet && mac_len >= ETH_HLEN) {
  5488. struct ethhdr *hdr;
  5489. /* use mpls_hdr() to get ethertype to account for VLANs. */
  5490. hdr = (struct ethhdr *)((void *)mpls_hdr(skb) - ETH_HLEN);
  5491. skb_mod_eth_type(skb, hdr, next_proto);
  5492. }
  5493. skb->protocol = next_proto;
  5494. return 0;
  5495. }
  5496. EXPORT_SYMBOL_GPL(skb_mpls_pop);
  5497. /**
  5498. * skb_mpls_update_lse() - modify outermost MPLS header and update csum
  5499. *
  5500. * @skb: buffer
  5501. * @mpls_lse: new MPLS label stack entry to update to
  5502. *
  5503. * Expects skb->data at mac header.
  5504. *
  5505. * Returns 0 on success, -errno otherwise.
  5506. */
  5507. int skb_mpls_update_lse(struct sk_buff *skb, __be32 mpls_lse)
  5508. {
  5509. int err;
  5510. if (unlikely(!eth_p_mpls(skb->protocol)))
  5511. return -EINVAL;
  5512. err = skb_ensure_writable(skb, skb->mac_len + MPLS_HLEN);
  5513. if (unlikely(err))
  5514. return err;
  5515. if (skb->ip_summed == CHECKSUM_COMPLETE) {
  5516. __be32 diff[] = { ~mpls_hdr(skb)->label_stack_entry, mpls_lse };
  5517. skb->csum = csum_partial((char *)diff, sizeof(diff), skb->csum);
  5518. }
  5519. mpls_hdr(skb)->label_stack_entry = mpls_lse;
  5520. return 0;
  5521. }
  5522. EXPORT_SYMBOL_GPL(skb_mpls_update_lse);
  5523. /**
  5524. * skb_mpls_dec_ttl() - decrement the TTL of the outermost MPLS header
  5525. *
  5526. * @skb: buffer
  5527. *
  5528. * Expects skb->data at mac header.
  5529. *
  5530. * Returns 0 on success, -errno otherwise.
  5531. */
  5532. int skb_mpls_dec_ttl(struct sk_buff *skb)
  5533. {
  5534. u32 lse;
  5535. u8 ttl;
  5536. if (unlikely(!eth_p_mpls(skb->protocol)))
  5537. return -EINVAL;
  5538. if (!pskb_may_pull(skb, skb_network_offset(skb) + MPLS_HLEN))
  5539. return -ENOMEM;
  5540. lse = be32_to_cpu(mpls_hdr(skb)->label_stack_entry);
  5541. ttl = (lse & MPLS_LS_TTL_MASK) >> MPLS_LS_TTL_SHIFT;
  5542. if (!--ttl)
  5543. return -EINVAL;
  5544. lse &= ~MPLS_LS_TTL_MASK;
  5545. lse |= ttl << MPLS_LS_TTL_SHIFT;
  5546. return skb_mpls_update_lse(skb, cpu_to_be32(lse));
  5547. }
  5548. EXPORT_SYMBOL_GPL(skb_mpls_dec_ttl);
  5549. /**
  5550. * alloc_skb_with_frags - allocate skb with page frags
  5551. *
  5552. * @header_len: size of linear part
  5553. * @data_len: needed length in frags
  5554. * @order: max page order desired.
  5555. * @errcode: pointer to error code if any
  5556. * @gfp_mask: allocation mask
  5557. *
  5558. * This can be used to allocate a paged skb, given a maximal order for frags.
  5559. */
  5560. struct sk_buff *alloc_skb_with_frags(unsigned long header_len,
  5561. unsigned long data_len,
  5562. int order,
  5563. int *errcode,
  5564. gfp_t gfp_mask)
  5565. {
  5566. unsigned long chunk;
  5567. struct sk_buff *skb;
  5568. struct page *page;
  5569. int nr_frags = 0;
  5570. *errcode = -EMSGSIZE;
  5571. if (unlikely(data_len > MAX_SKB_FRAGS * (PAGE_SIZE << order)))
  5572. return NULL;
  5573. *errcode = -ENOBUFS;
  5574. skb = alloc_skb(header_len, gfp_mask);
  5575. if (!skb)
  5576. return NULL;
  5577. while (data_len) {
  5578. if (nr_frags == MAX_SKB_FRAGS - 1)
  5579. goto failure;
  5580. while (order && PAGE_ALIGN(data_len) < (PAGE_SIZE << order))
  5581. order--;
  5582. if (order) {
  5583. page = alloc_pages((gfp_mask & ~__GFP_DIRECT_RECLAIM) |
  5584. __GFP_COMP |
  5585. __GFP_NOWARN,
  5586. order);
  5587. if (!page) {
  5588. order--;
  5589. continue;
  5590. }
  5591. } else {
  5592. page = alloc_page(gfp_mask);
  5593. if (!page)
  5594. goto failure;
  5595. }
  5596. chunk = min_t(unsigned long, data_len,
  5597. PAGE_SIZE << order);
  5598. skb_fill_page_desc(skb, nr_frags, page, 0, chunk);
  5599. nr_frags++;
  5600. skb->truesize += (PAGE_SIZE << order);
  5601. data_len -= chunk;
  5602. }
  5603. return skb;
  5604. failure:
  5605. kfree_skb(skb);
  5606. return NULL;
  5607. }
  5608. EXPORT_SYMBOL(alloc_skb_with_frags);
  5609. /* carve out the first off bytes from skb when off < headlen */
  5610. static int pskb_carve_inside_header(struct sk_buff *skb, const u32 off,
  5611. const int headlen, gfp_t gfp_mask)
  5612. {
  5613. int i;
  5614. unsigned int size = skb_end_offset(skb);
  5615. int new_hlen = headlen - off;
  5616. u8 *data;
  5617. if (skb_pfmemalloc(skb))
  5618. gfp_mask |= __GFP_MEMALLOC;
  5619. data = kmalloc_reserve(&size, gfp_mask, NUMA_NO_NODE, NULL);
  5620. if (!data)
  5621. return -ENOMEM;
  5622. size = SKB_WITH_OVERHEAD(size);
  5623. /* Copy real data, and all frags */
  5624. skb_copy_from_linear_data_offset(skb, off, data, new_hlen);
  5625. skb->len -= off;
  5626. memcpy((struct skb_shared_info *)(data + size),
  5627. skb_shinfo(skb),
  5628. offsetof(struct skb_shared_info,
  5629. frags[skb_shinfo(skb)->nr_frags]));
  5630. if (skb_cloned(skb)) {
  5631. /* drop the old head gracefully */
  5632. if (skb_orphan_frags(skb, gfp_mask)) {
  5633. skb_kfree_head(data, size);
  5634. return -ENOMEM;
  5635. }
  5636. for (i = 0; i < skb_shinfo(skb)->nr_frags; i++)
  5637. skb_frag_ref(skb, i);
  5638. if (skb_has_frag_list(skb))
  5639. skb_clone_fraglist(skb);
  5640. skb_release_data(skb, SKB_CONSUMED);
  5641. } else {
  5642. /* we can reuse existing recount- all we did was
  5643. * relocate values
  5644. */
  5645. skb_free_head(skb);
  5646. }
  5647. skb->head = data;
  5648. skb->data = data;
  5649. skb->head_frag = 0;
  5650. skb_set_end_offset(skb, size);
  5651. skb_set_tail_pointer(skb, skb_headlen(skb));
  5652. skb_headers_offset_update(skb, 0);
  5653. skb->cloned = 0;
  5654. skb->hdr_len = 0;
  5655. skb->nohdr = 0;
  5656. atomic_set(&skb_shinfo(skb)->dataref, 1);
  5657. return 0;
  5658. }
  5659. static int pskb_carve(struct sk_buff *skb, const u32 off, gfp_t gfp);
  5660. /* carve out the first eat bytes from skb's frag_list. May recurse into
  5661. * pskb_carve()
  5662. */
  5663. static int pskb_carve_frag_list(struct sk_buff *skb,
  5664. struct skb_shared_info *shinfo, int eat,
  5665. gfp_t gfp_mask)
  5666. {
  5667. struct sk_buff *list = shinfo->frag_list;
  5668. struct sk_buff *clone = NULL;
  5669. struct sk_buff *insp = NULL;
  5670. do {
  5671. if (!list) {
  5672. pr_err("Not enough bytes to eat. Want %d\n", eat);
  5673. return -EFAULT;
  5674. }
  5675. if (list->len <= eat) {
  5676. /* Eaten as whole. */
  5677. eat -= list->len;
  5678. list = list->next;
  5679. insp = list;
  5680. } else {
  5681. /* Eaten partially. */
  5682. if (skb_shared(list)) {
  5683. clone = skb_clone(list, gfp_mask);
  5684. if (!clone)
  5685. return -ENOMEM;
  5686. insp = list->next;
  5687. list = clone;
  5688. } else {
  5689. /* This may be pulled without problems. */
  5690. insp = list;
  5691. }
  5692. if (pskb_carve(list, eat, gfp_mask) < 0) {
  5693. kfree_skb(clone);
  5694. return -ENOMEM;
  5695. }
  5696. break;
  5697. }
  5698. } while (eat);
  5699. /* Free pulled out fragments. */
  5700. while ((list = shinfo->frag_list) != insp) {
  5701. shinfo->frag_list = list->next;
  5702. consume_skb(list);
  5703. }
  5704. /* And insert new clone at head. */
  5705. if (clone) {
  5706. clone->next = list;
  5707. shinfo->frag_list = clone;
  5708. }
  5709. return 0;
  5710. }
  5711. /* carve off first len bytes from skb. Split line (off) is in the
  5712. * non-linear part of skb
  5713. */
  5714. static int pskb_carve_inside_nonlinear(struct sk_buff *skb, const u32 off,
  5715. int pos, gfp_t gfp_mask)
  5716. {
  5717. int i, k = 0;
  5718. unsigned int size = skb_end_offset(skb);
  5719. u8 *data;
  5720. const int nfrags = skb_shinfo(skb)->nr_frags;
  5721. struct skb_shared_info *shinfo;
  5722. if (skb_pfmemalloc(skb))
  5723. gfp_mask |= __GFP_MEMALLOC;
  5724. data = kmalloc_reserve(&size, gfp_mask, NUMA_NO_NODE, NULL);
  5725. if (!data)
  5726. return -ENOMEM;
  5727. size = SKB_WITH_OVERHEAD(size);
  5728. memcpy((struct skb_shared_info *)(data + size),
  5729. skb_shinfo(skb), offsetof(struct skb_shared_info, frags[0]));
  5730. if (skb_orphan_frags(skb, gfp_mask)) {
  5731. skb_kfree_head(data, size);
  5732. return -ENOMEM;
  5733. }
  5734. shinfo = (struct skb_shared_info *)(data + size);
  5735. for (i = 0; i < nfrags; i++) {
  5736. int fsize = skb_frag_size(&skb_shinfo(skb)->frags[i]);
  5737. if (pos + fsize > off) {
  5738. shinfo->frags[k] = skb_shinfo(skb)->frags[i];
  5739. if (pos < off) {
  5740. /* Split frag.
  5741. * We have two variants in this case:
  5742. * 1. Move all the frag to the second
  5743. * part, if it is possible. F.e.
  5744. * this approach is mandatory for TUX,
  5745. * where splitting is expensive.
  5746. * 2. Split is accurately. We make this.
  5747. */
  5748. skb_frag_off_add(&shinfo->frags[0], off - pos);
  5749. skb_frag_size_sub(&shinfo->frags[0], off - pos);
  5750. }
  5751. skb_frag_ref(skb, i);
  5752. k++;
  5753. }
  5754. pos += fsize;
  5755. }
  5756. shinfo->nr_frags = k;
  5757. if (skb_has_frag_list(skb))
  5758. skb_clone_fraglist(skb);
  5759. /* split line is in frag list */
  5760. if (k == 0 && pskb_carve_frag_list(skb, shinfo, off - pos, gfp_mask)) {
  5761. /* skb_frag_unref() is not needed here as shinfo->nr_frags = 0. */
  5762. if (skb_has_frag_list(skb))
  5763. kfree_skb_list(skb_shinfo(skb)->frag_list);
  5764. skb_kfree_head(data, size);
  5765. return -ENOMEM;
  5766. }
  5767. skb_release_data(skb, SKB_CONSUMED);
  5768. skb->head = data;
  5769. skb->head_frag = 0;
  5770. skb->data = data;
  5771. skb_set_end_offset(skb, size);
  5772. skb_reset_tail_pointer(skb);
  5773. skb_headers_offset_update(skb, 0);
  5774. skb->cloned = 0;
  5775. skb->hdr_len = 0;
  5776. skb->nohdr = 0;
  5777. skb->len -= off;
  5778. skb->data_len = skb->len;
  5779. atomic_set(&skb_shinfo(skb)->dataref, 1);
  5780. return 0;
  5781. }
  5782. /* remove len bytes from the beginning of the skb */
  5783. static int pskb_carve(struct sk_buff *skb, const u32 len, gfp_t gfp)
  5784. {
  5785. int headlen = skb_headlen(skb);
  5786. if (len < headlen)
  5787. return pskb_carve_inside_header(skb, len, headlen, gfp);
  5788. else
  5789. return pskb_carve_inside_nonlinear(skb, len, headlen, gfp);
  5790. }
  5791. /* Extract to_copy bytes starting at off from skb, and return this in
  5792. * a new skb
  5793. */
  5794. struct sk_buff *pskb_extract(struct sk_buff *skb, int off,
  5795. int to_copy, gfp_t gfp)
  5796. {
  5797. struct sk_buff *clone = skb_clone(skb, gfp);
  5798. if (!clone)
  5799. return NULL;
  5800. if (pskb_carve(clone, off, gfp) < 0 ||
  5801. pskb_trim(clone, to_copy)) {
  5802. kfree_skb(clone);
  5803. return NULL;
  5804. }
  5805. return clone;
  5806. }
  5807. EXPORT_SYMBOL(pskb_extract);
  5808. /**
  5809. * skb_condense - try to get rid of fragments/frag_list if possible
  5810. * @skb: buffer
  5811. *
  5812. * Can be used to save memory before skb is added to a busy queue.
  5813. * If packet has bytes in frags and enough tail room in skb->head,
  5814. * pull all of them, so that we can free the frags right now and adjust
  5815. * truesize.
  5816. * Notes:
  5817. * We do not reallocate skb->head thus can not fail.
  5818. * Caller must re-evaluate skb->truesize if needed.
  5819. */
  5820. void skb_condense(struct sk_buff *skb)
  5821. {
  5822. if (skb->data_len) {
  5823. if (skb->data_len > skb->end - skb->tail ||
  5824. skb_cloned(skb) || !skb_frags_readable(skb))
  5825. return;
  5826. /* Nice, we can free page frag(s) right now */
  5827. __pskb_pull_tail(skb, skb->data_len);
  5828. }
  5829. /* At this point, skb->truesize might be over estimated,
  5830. * because skb had a fragment, and fragments do not tell
  5831. * their truesize.
  5832. * When we pulled its content into skb->head, fragment
  5833. * was freed, but __pskb_pull_tail() could not possibly
  5834. * adjust skb->truesize, not knowing the frag truesize.
  5835. */
  5836. skb->truesize = SKB_TRUESIZE(skb_end_offset(skb));
  5837. }
  5838. EXPORT_SYMBOL(skb_condense);
  5839. #ifdef CONFIG_SKB_EXTENSIONS
  5840. static void *skb_ext_get_ptr(struct skb_ext *ext, enum skb_ext_id id)
  5841. {
  5842. return (void *)ext + (ext->offset[id] * SKB_EXT_ALIGN_VALUE);
  5843. }
  5844. /**
  5845. * __skb_ext_alloc - allocate a new skb extensions storage
  5846. *
  5847. * @flags: See kmalloc().
  5848. *
  5849. * Returns the newly allocated pointer. The pointer can later attached to a
  5850. * skb via __skb_ext_set().
  5851. * Note: caller must handle the skb_ext as an opaque data.
  5852. */
  5853. struct skb_ext *__skb_ext_alloc(gfp_t flags)
  5854. {
  5855. struct skb_ext *new = kmem_cache_alloc(skbuff_ext_cache, flags);
  5856. if (new) {
  5857. memset(new->offset, 0, sizeof(new->offset));
  5858. refcount_set(&new->refcnt, 1);
  5859. }
  5860. return new;
  5861. }
  5862. static struct skb_ext *skb_ext_maybe_cow(struct skb_ext *old,
  5863. unsigned int old_active)
  5864. {
  5865. struct skb_ext *new;
  5866. if (refcount_read(&old->refcnt) == 1)
  5867. return old;
  5868. new = kmem_cache_alloc(skbuff_ext_cache, GFP_ATOMIC);
  5869. if (!new)
  5870. return NULL;
  5871. memcpy(new, old, old->chunks * SKB_EXT_ALIGN_VALUE);
  5872. refcount_set(&new->refcnt, 1);
  5873. #ifdef CONFIG_XFRM
  5874. if (old_active & (1 << SKB_EXT_SEC_PATH)) {
  5875. struct sec_path *sp = skb_ext_get_ptr(old, SKB_EXT_SEC_PATH);
  5876. unsigned int i;
  5877. for (i = 0; i < sp->len; i++)
  5878. xfrm_state_hold(sp->xvec[i]);
  5879. }
  5880. #endif
  5881. #ifdef CONFIG_MCTP_FLOWS
  5882. if (old_active & (1 << SKB_EXT_MCTP)) {
  5883. struct mctp_flow *flow = skb_ext_get_ptr(old, SKB_EXT_MCTP);
  5884. if (flow->key)
  5885. refcount_inc(&flow->key->refs);
  5886. }
  5887. #endif
  5888. __skb_ext_put(old);
  5889. return new;
  5890. }
  5891. /**
  5892. * __skb_ext_set - attach the specified extension storage to this skb
  5893. * @skb: buffer
  5894. * @id: extension id
  5895. * @ext: extension storage previously allocated via __skb_ext_alloc()
  5896. *
  5897. * Existing extensions, if any, are cleared.
  5898. *
  5899. * Returns the pointer to the extension.
  5900. */
  5901. void *__skb_ext_set(struct sk_buff *skb, enum skb_ext_id id,
  5902. struct skb_ext *ext)
  5903. {
  5904. unsigned int newlen, newoff = SKB_EXT_CHUNKSIZEOF(*ext);
  5905. skb_ext_put(skb);
  5906. newlen = newoff + skb_ext_type_len[id];
  5907. ext->chunks = newlen;
  5908. ext->offset[id] = newoff;
  5909. skb->extensions = ext;
  5910. skb->active_extensions = 1 << id;
  5911. return skb_ext_get_ptr(ext, id);
  5912. }
  5913. /**
  5914. * skb_ext_add - allocate space for given extension, COW if needed
  5915. * @skb: buffer
  5916. * @id: extension to allocate space for
  5917. *
  5918. * Allocates enough space for the given extension.
  5919. * If the extension is already present, a pointer to that extension
  5920. * is returned.
  5921. *
  5922. * If the skb was cloned, COW applies and the returned memory can be
  5923. * modified without changing the extension space of clones buffers.
  5924. *
  5925. * Returns pointer to the extension or NULL on allocation failure.
  5926. */
  5927. void *skb_ext_add(struct sk_buff *skb, enum skb_ext_id id)
  5928. {
  5929. struct skb_ext *new, *old = NULL;
  5930. unsigned int newlen, newoff;
  5931. if (skb->active_extensions) {
  5932. old = skb->extensions;
  5933. new = skb_ext_maybe_cow(old, skb->active_extensions);
  5934. if (!new)
  5935. return NULL;
  5936. if (__skb_ext_exist(new, id))
  5937. goto set_active;
  5938. newoff = new->chunks;
  5939. } else {
  5940. newoff = SKB_EXT_CHUNKSIZEOF(*new);
  5941. new = __skb_ext_alloc(GFP_ATOMIC);
  5942. if (!new)
  5943. return NULL;
  5944. }
  5945. newlen = newoff + skb_ext_type_len[id];
  5946. new->chunks = newlen;
  5947. new->offset[id] = newoff;
  5948. set_active:
  5949. skb->slow_gro = 1;
  5950. skb->extensions = new;
  5951. skb->active_extensions |= 1 << id;
  5952. return skb_ext_get_ptr(new, id);
  5953. }
  5954. EXPORT_SYMBOL(skb_ext_add);
  5955. #ifdef CONFIG_XFRM
  5956. static void skb_ext_put_sp(struct sec_path *sp)
  5957. {
  5958. unsigned int i;
  5959. for (i = 0; i < sp->len; i++)
  5960. xfrm_state_put(sp->xvec[i]);
  5961. }
  5962. #endif
  5963. #ifdef CONFIG_MCTP_FLOWS
  5964. static void skb_ext_put_mctp(struct mctp_flow *flow)
  5965. {
  5966. if (flow->key)
  5967. mctp_key_unref(flow->key);
  5968. }
  5969. #endif
  5970. void __skb_ext_del(struct sk_buff *skb, enum skb_ext_id id)
  5971. {
  5972. struct skb_ext *ext = skb->extensions;
  5973. skb->active_extensions &= ~(1 << id);
  5974. if (skb->active_extensions == 0) {
  5975. skb->extensions = NULL;
  5976. __skb_ext_put(ext);
  5977. #ifdef CONFIG_XFRM
  5978. } else if (id == SKB_EXT_SEC_PATH &&
  5979. refcount_read(&ext->refcnt) == 1) {
  5980. struct sec_path *sp = skb_ext_get_ptr(ext, SKB_EXT_SEC_PATH);
  5981. skb_ext_put_sp(sp);
  5982. sp->len = 0;
  5983. #endif
  5984. }
  5985. }
  5986. EXPORT_SYMBOL(__skb_ext_del);
  5987. void __skb_ext_put(struct skb_ext *ext)
  5988. {
  5989. /* If this is last clone, nothing can increment
  5990. * it after check passes. Avoids one atomic op.
  5991. */
  5992. if (refcount_read(&ext->refcnt) == 1)
  5993. goto free_now;
  5994. if (!refcount_dec_and_test(&ext->refcnt))
  5995. return;
  5996. free_now:
  5997. #ifdef CONFIG_XFRM
  5998. if (__skb_ext_exist(ext, SKB_EXT_SEC_PATH))
  5999. skb_ext_put_sp(skb_ext_get_ptr(ext, SKB_EXT_SEC_PATH));
  6000. #endif
  6001. #ifdef CONFIG_MCTP_FLOWS
  6002. if (__skb_ext_exist(ext, SKB_EXT_MCTP))
  6003. skb_ext_put_mctp(skb_ext_get_ptr(ext, SKB_EXT_MCTP));
  6004. #endif
  6005. kmem_cache_free(skbuff_ext_cache, ext);
  6006. }
  6007. EXPORT_SYMBOL(__skb_ext_put);
  6008. #endif /* CONFIG_SKB_EXTENSIONS */
  6009. static void kfree_skb_napi_cache(struct sk_buff *skb)
  6010. {
  6011. /* if SKB is a clone, don't handle this case */
  6012. if (skb->fclone != SKB_FCLONE_UNAVAILABLE) {
  6013. __kfree_skb(skb);
  6014. return;
  6015. }
  6016. local_bh_disable();
  6017. __napi_kfree_skb(skb, SKB_CONSUMED);
  6018. local_bh_enable();
  6019. }
  6020. /**
  6021. * skb_attempt_defer_free - queue skb for remote freeing
  6022. * @skb: buffer
  6023. *
  6024. * Put @skb in a per-cpu list, using the cpu which
  6025. * allocated the skb/pages to reduce false sharing
  6026. * and memory zone spinlock contention.
  6027. */
  6028. void skb_attempt_defer_free(struct sk_buff *skb)
  6029. {
  6030. int cpu = skb->alloc_cpu;
  6031. struct softnet_data *sd;
  6032. unsigned int defer_max;
  6033. bool kick;
  6034. if (cpu == raw_smp_processor_id() ||
  6035. WARN_ON_ONCE(cpu >= nr_cpu_ids) ||
  6036. !cpu_online(cpu)) {
  6037. nodefer: kfree_skb_napi_cache(skb);
  6038. return;
  6039. }
  6040. DEBUG_NET_WARN_ON_ONCE(skb_dst(skb));
  6041. DEBUG_NET_WARN_ON_ONCE(skb->destructor);
  6042. sd = &per_cpu(softnet_data, cpu);
  6043. defer_max = READ_ONCE(net_hotdata.sysctl_skb_defer_max);
  6044. if (READ_ONCE(sd->defer_count) >= defer_max)
  6045. goto nodefer;
  6046. spin_lock_bh(&sd->defer_lock);
  6047. /* Send an IPI every time queue reaches half capacity. */
  6048. kick = sd->defer_count == (defer_max >> 1);
  6049. /* Paired with the READ_ONCE() few lines above */
  6050. WRITE_ONCE(sd->defer_count, sd->defer_count + 1);
  6051. skb->next = sd->defer_list;
  6052. /* Paired with READ_ONCE() in skb_defer_free_flush() */
  6053. WRITE_ONCE(sd->defer_list, skb);
  6054. spin_unlock_bh(&sd->defer_lock);
  6055. /* Make sure to trigger NET_RX_SOFTIRQ on the remote CPU
  6056. * if we are unlucky enough (this seems very unlikely).
  6057. */
  6058. if (unlikely(kick))
  6059. kick_defer_list_purge(sd, cpu);
  6060. }
  6061. static void skb_splice_csum_page(struct sk_buff *skb, struct page *page,
  6062. size_t offset, size_t len)
  6063. {
  6064. const char *kaddr;
  6065. __wsum csum;
  6066. kaddr = kmap_local_page(page);
  6067. csum = csum_partial(kaddr + offset, len, 0);
  6068. kunmap_local(kaddr);
  6069. skb->csum = csum_block_add(skb->csum, csum, skb->len);
  6070. }
  6071. /**
  6072. * skb_splice_from_iter - Splice (or copy) pages to skbuff
  6073. * @skb: The buffer to add pages to
  6074. * @iter: Iterator representing the pages to be added
  6075. * @maxsize: Maximum amount of pages to be added
  6076. * @gfp: Allocation flags
  6077. *
  6078. * This is a common helper function for supporting MSG_SPLICE_PAGES. It
  6079. * extracts pages from an iterator and adds them to the socket buffer if
  6080. * possible, copying them to fragments if not possible (such as if they're slab
  6081. * pages).
  6082. *
  6083. * Returns the amount of data spliced/copied or -EMSGSIZE if there's
  6084. * insufficient space in the buffer to transfer anything.
  6085. */
  6086. ssize_t skb_splice_from_iter(struct sk_buff *skb, struct iov_iter *iter,
  6087. ssize_t maxsize, gfp_t gfp)
  6088. {
  6089. size_t frag_limit = READ_ONCE(net_hotdata.sysctl_max_skb_frags);
  6090. struct page *pages[8], **ppages = pages;
  6091. ssize_t spliced = 0, ret = 0;
  6092. unsigned int i;
  6093. while (iter->count > 0) {
  6094. ssize_t space, nr, len;
  6095. size_t off;
  6096. ret = -EMSGSIZE;
  6097. space = frag_limit - skb_shinfo(skb)->nr_frags;
  6098. if (space < 0)
  6099. break;
  6100. /* We might be able to coalesce without increasing nr_frags */
  6101. nr = clamp_t(size_t, space, 1, ARRAY_SIZE(pages));
  6102. len = iov_iter_extract_pages(iter, &ppages, maxsize, nr, 0, &off);
  6103. if (len <= 0) {
  6104. ret = len ?: -EIO;
  6105. break;
  6106. }
  6107. i = 0;
  6108. do {
  6109. struct page *page = pages[i++];
  6110. size_t part = min_t(size_t, PAGE_SIZE - off, len);
  6111. ret = -EIO;
  6112. if (WARN_ON_ONCE(!sendpage_ok(page)))
  6113. goto out;
  6114. ret = skb_append_pagefrags(skb, page, off, part,
  6115. frag_limit);
  6116. if (ret < 0) {
  6117. iov_iter_revert(iter, len);
  6118. goto out;
  6119. }
  6120. if (skb->ip_summed == CHECKSUM_NONE)
  6121. skb_splice_csum_page(skb, page, off, part);
  6122. off = 0;
  6123. spliced += part;
  6124. maxsize -= part;
  6125. len -= part;
  6126. } while (len > 0);
  6127. if (maxsize <= 0)
  6128. break;
  6129. }
  6130. out:
  6131. skb_len_add(skb, spliced);
  6132. return spliced ?: ret;
  6133. }
  6134. EXPORT_SYMBOL(skb_splice_from_iter);
  6135. static __always_inline
  6136. size_t memcpy_from_iter_csum(void *iter_from, size_t progress,
  6137. size_t len, void *to, void *priv2)
  6138. {
  6139. __wsum *csum = priv2;
  6140. __wsum next = csum_partial_copy_nocheck(iter_from, to + progress, len);
  6141. *csum = csum_block_add(*csum, next, progress);
  6142. return 0;
  6143. }
  6144. static __always_inline
  6145. size_t copy_from_user_iter_csum(void __user *iter_from, size_t progress,
  6146. size_t len, void *to, void *priv2)
  6147. {
  6148. __wsum next, *csum = priv2;
  6149. next = csum_and_copy_from_user(iter_from, to + progress, len);
  6150. *csum = csum_block_add(*csum, next, progress);
  6151. return next ? 0 : len;
  6152. }
  6153. bool csum_and_copy_from_iter_full(void *addr, size_t bytes,
  6154. __wsum *csum, struct iov_iter *i)
  6155. {
  6156. size_t copied;
  6157. if (WARN_ON_ONCE(!i->data_source))
  6158. return false;
  6159. copied = iterate_and_advance2(i, bytes, addr, csum,
  6160. copy_from_user_iter_csum,
  6161. memcpy_from_iter_csum);
  6162. if (likely(copied == bytes))
  6163. return true;
  6164. iov_iter_revert(i, copied);
  6165. return false;
  6166. }
  6167. EXPORT_SYMBOL(csum_and_copy_from_iter_full);