ext.c 209 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020102110221023102410251026102710281029103010311032103310341035103610371038103910401041104210431044104510461047104810491050105110521053105410551056105710581059106010611062106310641065106610671068106910701071107210731074107510761077107810791080108110821083108410851086108710881089109010911092109310941095109610971098109911001101110211031104110511061107110811091110111111121113111411151116111711181119112011211122112311241125112611271128112911301131113211331134113511361137113811391140114111421143114411451146114711481149115011511152115311541155115611571158115911601161116211631164116511661167116811691170117111721173117411751176117711781179118011811182118311841185118611871188118911901191119211931194119511961197119811991200120112021203120412051206120712081209121012111212121312141215121612171218121912201221122212231224122512261227122812291230123112321233123412351236123712381239124012411242124312441245124612471248124912501251125212531254125512561257125812591260126112621263126412651266126712681269127012711272127312741275127612771278127912801281128212831284128512861287128812891290129112921293129412951296129712981299130013011302130313041305130613071308130913101311131213131314131513161317131813191320132113221323132413251326132713281329133013311332133313341335133613371338133913401341134213431344134513461347134813491350135113521353135413551356135713581359136013611362136313641365136613671368136913701371137213731374137513761377137813791380138113821383138413851386138713881389139013911392139313941395139613971398139914001401140214031404140514061407140814091410141114121413141414151416141714181419142014211422142314241425142614271428142914301431143214331434143514361437143814391440144114421443144414451446144714481449145014511452145314541455145614571458145914601461146214631464146514661467146814691470147114721473147414751476147714781479148014811482148314841485148614871488148914901491149214931494149514961497149814991500150115021503150415051506150715081509151015111512151315141515151615171518151915201521152215231524152515261527152815291530153115321533153415351536153715381539154015411542154315441545154615471548154915501551155215531554155515561557155815591560156115621563156415651566156715681569157015711572157315741575157615771578157915801581158215831584158515861587158815891590159115921593159415951596159715981599160016011602160316041605160616071608160916101611161216131614161516161617161816191620162116221623162416251626162716281629163016311632163316341635163616371638163916401641164216431644164516461647164816491650165116521653165416551656165716581659166016611662166316641665166616671668166916701671167216731674167516761677167816791680168116821683168416851686168716881689169016911692169316941695169616971698169917001701170217031704170517061707170817091710171117121713171417151716171717181719172017211722172317241725172617271728172917301731173217331734173517361737173817391740174117421743174417451746174717481749175017511752175317541755175617571758175917601761176217631764176517661767176817691770177117721773177417751776177717781779178017811782178317841785178617871788178917901791179217931794179517961797179817991800180118021803180418051806180718081809181018111812181318141815181618171818181918201821182218231824182518261827182818291830183118321833183418351836183718381839184018411842184318441845184618471848184918501851185218531854185518561857185818591860186118621863186418651866186718681869187018711872187318741875187618771878187918801881188218831884188518861887188818891890189118921893189418951896189718981899190019011902190319041905190619071908190919101911191219131914191519161917191819191920192119221923192419251926192719281929193019311932193319341935193619371938193919401941194219431944194519461947194819491950195119521953195419551956195719581959196019611962196319641965196619671968196919701971197219731974197519761977197819791980198119821983198419851986198719881989199019911992199319941995199619971998199920002001200220032004200520062007200820092010201120122013201420152016201720182019202020212022202320242025202620272028202920302031203220332034203520362037203820392040204120422043204420452046204720482049205020512052205320542055205620572058205920602061206220632064206520662067206820692070207120722073207420752076207720782079208020812082208320842085208620872088208920902091209220932094209520962097209820992100210121022103210421052106210721082109211021112112211321142115211621172118211921202121212221232124212521262127212821292130213121322133213421352136213721382139214021412142214321442145214621472148214921502151215221532154215521562157215821592160216121622163216421652166216721682169217021712172217321742175217621772178217921802181218221832184218521862187218821892190219121922193219421952196219721982199220022012202220322042205220622072208220922102211221222132214221522162217221822192220222122222223222422252226222722282229223022312232223322342235223622372238223922402241224222432244224522462247224822492250225122522253225422552256225722582259226022612262226322642265226622672268226922702271227222732274227522762277227822792280228122822283228422852286228722882289229022912292229322942295229622972298229923002301230223032304230523062307230823092310231123122313231423152316231723182319232023212322232323242325232623272328232923302331233223332334233523362337233823392340234123422343234423452346234723482349235023512352235323542355235623572358235923602361236223632364236523662367236823692370237123722373237423752376237723782379238023812382238323842385238623872388238923902391239223932394239523962397239823992400240124022403240424052406240724082409241024112412241324142415241624172418241924202421242224232424242524262427242824292430243124322433243424352436243724382439244024412442244324442445244624472448244924502451245224532454245524562457245824592460246124622463246424652466246724682469247024712472247324742475247624772478247924802481248224832484248524862487248824892490249124922493249424952496249724982499250025012502250325042505250625072508250925102511251225132514251525162517251825192520252125222523252425252526252725282529253025312532253325342535253625372538253925402541254225432544254525462547254825492550255125522553255425552556255725582559256025612562256325642565256625672568256925702571257225732574257525762577257825792580258125822583258425852586258725882589259025912592259325942595259625972598259926002601260226032604260526062607260826092610261126122613261426152616261726182619262026212622262326242625262626272628262926302631263226332634263526362637263826392640264126422643264426452646264726482649265026512652265326542655265626572658265926602661266226632664266526662667266826692670267126722673267426752676267726782679268026812682268326842685268626872688268926902691269226932694269526962697269826992700270127022703270427052706270727082709271027112712271327142715271627172718271927202721272227232724272527262727272827292730273127322733273427352736273727382739274027412742274327442745274627472748274927502751275227532754275527562757275827592760276127622763276427652766276727682769277027712772277327742775277627772778277927802781278227832784278527862787278827892790279127922793279427952796279727982799280028012802280328042805280628072808280928102811281228132814281528162817281828192820282128222823282428252826282728282829283028312832283328342835283628372838283928402841284228432844284528462847284828492850285128522853285428552856285728582859286028612862286328642865286628672868286928702871287228732874287528762877287828792880288128822883288428852886288728882889289028912892289328942895289628972898289929002901290229032904290529062907290829092910291129122913291429152916291729182919292029212922292329242925292629272928292929302931293229332934293529362937293829392940294129422943294429452946294729482949295029512952295329542955295629572958295929602961296229632964296529662967296829692970297129722973297429752976297729782979298029812982298329842985298629872988298929902991299229932994299529962997299829993000300130023003300430053006300730083009301030113012301330143015301630173018301930203021302230233024302530263027302830293030303130323033303430353036303730383039304030413042304330443045304630473048304930503051305230533054305530563057305830593060306130623063306430653066306730683069307030713072307330743075307630773078307930803081308230833084308530863087308830893090309130923093309430953096309730983099310031013102310331043105310631073108310931103111311231133114311531163117311831193120312131223123312431253126312731283129313031313132313331343135313631373138313931403141314231433144314531463147314831493150315131523153315431553156315731583159316031613162316331643165316631673168316931703171317231733174317531763177317831793180318131823183318431853186318731883189319031913192319331943195319631973198319932003201320232033204320532063207320832093210321132123213321432153216321732183219322032213222322332243225322632273228322932303231323232333234323532363237323832393240324132423243324432453246324732483249325032513252325332543255325632573258325932603261326232633264326532663267326832693270327132723273327432753276327732783279328032813282328332843285328632873288328932903291329232933294329532963297329832993300330133023303330433053306330733083309331033113312331333143315331633173318331933203321332233233324332533263327332833293330333133323333333433353336333733383339334033413342334333443345334633473348334933503351335233533354335533563357335833593360336133623363336433653366336733683369337033713372337333743375337633773378337933803381338233833384338533863387338833893390339133923393339433953396339733983399340034013402340334043405340634073408340934103411341234133414341534163417341834193420342134223423342434253426342734283429343034313432343334343435343634373438343934403441344234433444344534463447344834493450345134523453345434553456345734583459346034613462346334643465346634673468346934703471347234733474347534763477347834793480348134823483348434853486348734883489349034913492349334943495349634973498349935003501350235033504350535063507350835093510351135123513351435153516351735183519352035213522352335243525352635273528352935303531353235333534353535363537353835393540354135423543354435453546354735483549355035513552355335543555355635573558355935603561356235633564356535663567356835693570357135723573357435753576357735783579358035813582358335843585358635873588358935903591359235933594359535963597359835993600360136023603360436053606360736083609361036113612361336143615361636173618361936203621362236233624362536263627362836293630363136323633363436353636363736383639364036413642364336443645364636473648364936503651365236533654365536563657365836593660366136623663366436653666366736683669367036713672367336743675367636773678367936803681368236833684368536863687368836893690369136923693369436953696369736983699370037013702370337043705370637073708370937103711371237133714371537163717371837193720372137223723372437253726372737283729373037313732373337343735373637373738373937403741374237433744374537463747374837493750375137523753375437553756375737583759376037613762376337643765376637673768376937703771377237733774377537763777377837793780378137823783378437853786378737883789379037913792379337943795379637973798379938003801380238033804380538063807380838093810381138123813381438153816381738183819382038213822382338243825382638273828382938303831383238333834383538363837383838393840384138423843384438453846384738483849385038513852385338543855385638573858385938603861386238633864386538663867386838693870387138723873387438753876387738783879388038813882388338843885388638873888388938903891389238933894389538963897389838993900390139023903390439053906390739083909391039113912391339143915391639173918391939203921392239233924392539263927392839293930393139323933393439353936393739383939394039413942394339443945394639473948394939503951395239533954395539563957395839593960396139623963396439653966396739683969397039713972397339743975397639773978397939803981398239833984398539863987398839893990399139923993399439953996399739983999400040014002400340044005400640074008400940104011401240134014401540164017401840194020402140224023402440254026402740284029403040314032403340344035403640374038403940404041404240434044404540464047404840494050405140524053405440554056405740584059406040614062406340644065406640674068406940704071407240734074407540764077407840794080408140824083408440854086408740884089409040914092409340944095409640974098409941004101410241034104410541064107410841094110411141124113411441154116411741184119412041214122412341244125412641274128412941304131413241334134413541364137413841394140414141424143414441454146414741484149415041514152415341544155415641574158415941604161416241634164416541664167416841694170417141724173417441754176417741784179418041814182418341844185418641874188418941904191419241934194419541964197419841994200420142024203420442054206420742084209421042114212421342144215421642174218421942204221422242234224422542264227422842294230423142324233423442354236423742384239424042414242424342444245424642474248424942504251425242534254425542564257425842594260426142624263426442654266426742684269427042714272427342744275427642774278427942804281428242834284428542864287428842894290429142924293429442954296429742984299430043014302430343044305430643074308430943104311431243134314431543164317431843194320432143224323432443254326432743284329433043314332433343344335433643374338433943404341434243434344434543464347434843494350435143524353435443554356435743584359436043614362436343644365436643674368436943704371437243734374437543764377437843794380438143824383438443854386438743884389439043914392439343944395439643974398439944004401440244034404440544064407440844094410441144124413441444154416441744184419442044214422442344244425442644274428442944304431443244334434443544364437443844394440444144424443444444454446444744484449445044514452445344544455445644574458445944604461446244634464446544664467446844694470447144724473447444754476447744784479448044814482448344844485448644874488448944904491449244934494449544964497449844994500450145024503450445054506450745084509451045114512451345144515451645174518451945204521452245234524452545264527452845294530453145324533453445354536453745384539454045414542454345444545454645474548454945504551455245534554455545564557455845594560456145624563456445654566456745684569457045714572457345744575457645774578457945804581458245834584458545864587458845894590459145924593459445954596459745984599460046014602460346044605460646074608460946104611461246134614461546164617461846194620462146224623462446254626462746284629463046314632463346344635463646374638463946404641464246434644464546464647464846494650465146524653465446554656465746584659466046614662466346644665466646674668466946704671467246734674467546764677467846794680468146824683468446854686468746884689469046914692469346944695469646974698469947004701470247034704470547064707470847094710471147124713471447154716471747184719472047214722472347244725472647274728472947304731473247334734473547364737473847394740474147424743474447454746474747484749475047514752475347544755475647574758475947604761476247634764476547664767476847694770477147724773477447754776477747784779478047814782478347844785478647874788478947904791479247934794479547964797479847994800480148024803480448054806480748084809481048114812481348144815481648174818481948204821482248234824482548264827482848294830483148324833483448354836483748384839484048414842484348444845484648474848484948504851485248534854485548564857485848594860486148624863486448654866486748684869487048714872487348744875487648774878487948804881488248834884488548864887488848894890489148924893489448954896489748984899490049014902490349044905490649074908490949104911491249134914491549164917491849194920492149224923492449254926492749284929493049314932493349344935493649374938493949404941494249434944494549464947494849494950495149524953495449554956495749584959496049614962496349644965496649674968496949704971497249734974497549764977497849794980498149824983498449854986498749884989499049914992499349944995499649974998499950005001500250035004500550065007500850095010501150125013501450155016501750185019502050215022502350245025502650275028502950305031503250335034503550365037503850395040504150425043504450455046504750485049505050515052505350545055505650575058505950605061506250635064506550665067506850695070507150725073507450755076507750785079508050815082508350845085508650875088508950905091509250935094509550965097509850995100510151025103510451055106510751085109511051115112511351145115511651175118511951205121512251235124512551265127512851295130513151325133513451355136513751385139514051415142514351445145514651475148514951505151515251535154515551565157515851595160516151625163516451655166516751685169517051715172517351745175517651775178517951805181518251835184518551865187518851895190519151925193519451955196519751985199520052015202520352045205520652075208520952105211521252135214521552165217521852195220522152225223522452255226522752285229523052315232523352345235523652375238523952405241524252435244524552465247524852495250525152525253525452555256525752585259526052615262526352645265526652675268526952705271527252735274527552765277527852795280528152825283528452855286528752885289529052915292529352945295529652975298529953005301530253035304530553065307530853095310531153125313531453155316531753185319532053215322532353245325532653275328532953305331533253335334533553365337533853395340534153425343534453455346534753485349535053515352535353545355535653575358535953605361536253635364536553665367536853695370537153725373537453755376537753785379538053815382538353845385538653875388538953905391539253935394539553965397539853995400540154025403540454055406540754085409541054115412541354145415541654175418541954205421542254235424542554265427542854295430543154325433543454355436543754385439544054415442544354445445544654475448544954505451545254535454545554565457545854595460546154625463546454655466546754685469547054715472547354745475547654775478547954805481548254835484548554865487548854895490549154925493549454955496549754985499550055015502550355045505550655075508550955105511551255135514551555165517551855195520552155225523552455255526552755285529553055315532553355345535553655375538553955405541554255435544554555465547554855495550555155525553555455555556555755585559556055615562556355645565556655675568556955705571557255735574557555765577557855795580558155825583558455855586558755885589559055915592559355945595559655975598559956005601560256035604560556065607560856095610561156125613561456155616561756185619562056215622562356245625562656275628562956305631563256335634563556365637563856395640564156425643564456455646564756485649565056515652565356545655565656575658565956605661566256635664566556665667566856695670567156725673567456755676567756785679568056815682568356845685568656875688568956905691569256935694569556965697569856995700570157025703570457055706570757085709571057115712571357145715571657175718571957205721572257235724572557265727572857295730573157325733573457355736573757385739574057415742574357445745574657475748574957505751575257535754575557565757575857595760576157625763576457655766576757685769577057715772577357745775577657775778577957805781578257835784578557865787578857895790579157925793579457955796579757985799580058015802580358045805580658075808580958105811581258135814581558165817581858195820582158225823582458255826582758285829583058315832583358345835583658375838583958405841584258435844584558465847584858495850585158525853585458555856585758585859586058615862586358645865586658675868586958705871587258735874587558765877587858795880588158825883588458855886588758885889589058915892589358945895589658975898589959005901590259035904590559065907590859095910591159125913591459155916591759185919592059215922592359245925592659275928592959305931593259335934593559365937593859395940594159425943594459455946594759485949595059515952595359545955595659575958595959605961596259635964596559665967596859695970597159725973597459755976597759785979598059815982598359845985598659875988598959905991599259935994599559965997599859996000600160026003600460056006600760086009601060116012601360146015601660176018601960206021602260236024602560266027602860296030603160326033603460356036603760386039604060416042604360446045604660476048604960506051605260536054605560566057605860596060606160626063606460656066606760686069607060716072607360746075607660776078607960806081608260836084608560866087608860896090609160926093609460956096609760986099610061016102610361046105610661076108610961106111611261136114611561166117611861196120612161226123612461256126612761286129613061316132613361346135613661376138613961406141614261436144614561466147614861496150615161526153615461556156615761586159616061616162616361646165616661676168616961706171617261736174617561766177617861796180618161826183618461856186618761886189619061916192619361946195619661976198619962006201620262036204620562066207620862096210621162126213621462156216621762186219622062216222622362246225622662276228622962306231623262336234623562366237623862396240624162426243624462456246624762486249625062516252625362546255625662576258625962606261626262636264626562666267626862696270627162726273627462756276627762786279628062816282628362846285628662876288628962906291629262936294629562966297629862996300630163026303630463056306630763086309631063116312631363146315631663176318631963206321632263236324632563266327632863296330633163326333633463356336633763386339634063416342634363446345634663476348634963506351635263536354635563566357635863596360636163626363636463656366636763686369637063716372637363746375637663776378637963806381638263836384638563866387638863896390639163926393639463956396639763986399640064016402640364046405640664076408640964106411641264136414641564166417641864196420642164226423642464256426642764286429643064316432643364346435643664376438643964406441644264436444644564466447644864496450645164526453645464556456645764586459646064616462646364646465646664676468646964706471647264736474647564766477647864796480648164826483648464856486648764886489649064916492649364946495649664976498649965006501650265036504650565066507650865096510651165126513651465156516651765186519652065216522652365246525652665276528652965306531653265336534653565366537653865396540654165426543654465456546654765486549655065516552655365546555655665576558655965606561656265636564656565666567656865696570657165726573657465756576657765786579658065816582658365846585658665876588658965906591659265936594659565966597659865996600660166026603660466056606660766086609661066116612661366146615661666176618661966206621662266236624662566266627662866296630663166326633663466356636663766386639664066416642664366446645664666476648664966506651665266536654665566566657665866596660666166626663666466656666666766686669667066716672667366746675667666776678667966806681668266836684668566866687668866896690669166926693669466956696669766986699670067016702670367046705670667076708670967106711671267136714671567166717671867196720672167226723672467256726672767286729673067316732673367346735673667376738673967406741674267436744674567466747674867496750675167526753675467556756675767586759676067616762676367646765676667676768676967706771677267736774677567766777677867796780678167826783678467856786678767886789679067916792679367946795679667976798679968006801680268036804680568066807680868096810681168126813681468156816681768186819682068216822682368246825682668276828682968306831683268336834683568366837683868396840684168426843684468456846684768486849685068516852685368546855685668576858685968606861686268636864686568666867686868696870687168726873687468756876687768786879688068816882688368846885688668876888688968906891689268936894689568966897689868996900690169026903690469056906690769086909691069116912691369146915691669176918691969206921692269236924692569266927692869296930693169326933693469356936693769386939694069416942694369446945694669476948694969506951695269536954695569566957695869596960696169626963696469656966696769686969697069716972697369746975697669776978697969806981698269836984698569866987698869896990699169926993699469956996699769986999700070017002700370047005700670077008700970107011701270137014701570167017701870197020702170227023702470257026702770287029703070317032703370347035703670377038703970407041704270437044704570467047704870497050705170527053705470557056705770587059706070617062706370647065706670677068706970707071707270737074707570767077707870797080708170827083708470857086708770887089709070917092709370947095709670977098709971007101710271037104710571067107710871097110711171127113711471157116711771187119712071217122712371247125712671277128712971307131713271337134713571367137713871397140714171427143714471457146714771487149715071517152715371547155715671577158715971607161716271637164716571667167716871697170717171727173717471757176717771787179718071817182718371847185718671877188718971907191719271937194719571967197719871997200720172027203720472057206720772087209721072117212721372147215721672177218721972207221722272237224722572267227722872297230723172327233723472357236723772387239724072417242724372447245724672477248724972507251725272537254725572567257725872597260726172627263726472657266726772687269727072717272727372747275727672777278727972807281728272837284728572867287728872897290729172927293729472957296729772987299730073017302730373047305730673077308730973107311731273137314731573167317731873197320732173227323732473257326732773287329733073317332733373347335733673377338733973407341734273437344734573467347734873497350735173527353735473557356735773587359736073617362736373647365736673677368736973707371737273737374737573767377737873797380738173827383738473857386738773887389739073917392739373947395
  1. /* SPDX-License-Identifier: GPL-2.0 */
  2. /*
  3. * BPF extensible scheduler class: Documentation/scheduler/sched-ext.rst
  4. *
  5. * Copyright (c) 2022 Meta Platforms, Inc. and affiliates.
  6. * Copyright (c) 2022 Tejun Heo <tj@kernel.org>
  7. * Copyright (c) 2022 David Vernet <dvernet@meta.com>
  8. */
  9. #define SCX_OP_IDX(op) (offsetof(struct sched_ext_ops, op) / sizeof(void (*)(void)))
  10. enum scx_consts {
  11. SCX_DSP_DFL_MAX_BATCH = 32,
  12. SCX_DSP_MAX_LOOPS = 32,
  13. SCX_WATCHDOG_MAX_TIMEOUT = 30 * HZ,
  14. SCX_EXIT_BT_LEN = 64,
  15. SCX_EXIT_MSG_LEN = 1024,
  16. SCX_EXIT_DUMP_DFL_LEN = 32768,
  17. SCX_CPUPERF_ONE = SCHED_CAPACITY_SCALE,
  18. /*
  19. * Iterating all tasks may take a while. Periodically drop
  20. * scx_tasks_lock to avoid causing e.g. CSD and RCU stalls.
  21. */
  22. SCX_OPS_TASK_ITER_BATCH = 32,
  23. };
  24. enum scx_exit_kind {
  25. SCX_EXIT_NONE,
  26. SCX_EXIT_DONE,
  27. SCX_EXIT_UNREG = 64, /* user-space initiated unregistration */
  28. SCX_EXIT_UNREG_BPF, /* BPF-initiated unregistration */
  29. SCX_EXIT_UNREG_KERN, /* kernel-initiated unregistration */
  30. SCX_EXIT_SYSRQ, /* requested by 'S' sysrq */
  31. SCX_EXIT_ERROR = 1024, /* runtime error, error msg contains details */
  32. SCX_EXIT_ERROR_BPF, /* ERROR but triggered through scx_bpf_error() */
  33. SCX_EXIT_ERROR_STALL, /* watchdog detected stalled runnable tasks */
  34. };
  35. /*
  36. * An exit code can be specified when exiting with scx_bpf_exit() or
  37. * scx_ops_exit(), corresponding to exit_kind UNREG_BPF and UNREG_KERN
  38. * respectively. The codes are 64bit of the format:
  39. *
  40. * Bits: [63 .. 48 47 .. 32 31 .. 0]
  41. * [ SYS ACT ] [ SYS RSN ] [ USR ]
  42. *
  43. * SYS ACT: System-defined exit actions
  44. * SYS RSN: System-defined exit reasons
  45. * USR : User-defined exit codes and reasons
  46. *
  47. * Using the above, users may communicate intention and context by ORing system
  48. * actions and/or system reasons with a user-defined exit code.
  49. */
  50. enum scx_exit_code {
  51. /* Reasons */
  52. SCX_ECODE_RSN_HOTPLUG = 1LLU << 32,
  53. /* Actions */
  54. SCX_ECODE_ACT_RESTART = 1LLU << 48,
  55. };
  56. /*
  57. * scx_exit_info is passed to ops.exit() to describe why the BPF scheduler is
  58. * being disabled.
  59. */
  60. struct scx_exit_info {
  61. /* %SCX_EXIT_* - broad category of the exit reason */
  62. enum scx_exit_kind kind;
  63. /* exit code if gracefully exiting */
  64. s64 exit_code;
  65. /* textual representation of the above */
  66. const char *reason;
  67. /* backtrace if exiting due to an error */
  68. unsigned long *bt;
  69. u32 bt_len;
  70. /* informational message */
  71. char *msg;
  72. /* debug dump */
  73. char *dump;
  74. };
  75. /* sched_ext_ops.flags */
  76. enum scx_ops_flags {
  77. /*
  78. * Keep built-in idle tracking even if ops.update_idle() is implemented.
  79. */
  80. SCX_OPS_KEEP_BUILTIN_IDLE = 1LLU << 0,
  81. /*
  82. * By default, if there are no other task to run on the CPU, ext core
  83. * keeps running the current task even after its slice expires. If this
  84. * flag is specified, such tasks are passed to ops.enqueue() with
  85. * %SCX_ENQ_LAST. See the comment above %SCX_ENQ_LAST for more info.
  86. */
  87. SCX_OPS_ENQ_LAST = 1LLU << 1,
  88. /*
  89. * An exiting task may schedule after PF_EXITING is set. In such cases,
  90. * bpf_task_from_pid() may not be able to find the task and if the BPF
  91. * scheduler depends on pid lookup for dispatching, the task will be
  92. * lost leading to various issues including RCU grace period stalls.
  93. *
  94. * To mask this problem, by default, unhashed tasks are automatically
  95. * dispatched to the local DSQ on enqueue. If the BPF scheduler doesn't
  96. * depend on pid lookups and wants to handle these tasks directly, the
  97. * following flag can be used.
  98. */
  99. SCX_OPS_ENQ_EXITING = 1LLU << 2,
  100. /*
  101. * If set, only tasks with policy set to SCHED_EXT are attached to
  102. * sched_ext. If clear, SCHED_NORMAL tasks are also included.
  103. */
  104. SCX_OPS_SWITCH_PARTIAL = 1LLU << 3,
  105. /*
  106. * CPU cgroup support flags
  107. */
  108. SCX_OPS_HAS_CGROUP_WEIGHT = 1LLU << 16, /* cpu.weight */
  109. SCX_OPS_ALL_FLAGS = SCX_OPS_KEEP_BUILTIN_IDLE |
  110. SCX_OPS_ENQ_LAST |
  111. SCX_OPS_ENQ_EXITING |
  112. SCX_OPS_SWITCH_PARTIAL |
  113. SCX_OPS_HAS_CGROUP_WEIGHT,
  114. };
  115. /* argument container for ops.init_task() */
  116. struct scx_init_task_args {
  117. /*
  118. * Set if ops.init_task() is being invoked on the fork path, as opposed
  119. * to the scheduler transition path.
  120. */
  121. bool fork;
  122. #ifdef CONFIG_EXT_GROUP_SCHED
  123. /* the cgroup the task is joining */
  124. struct cgroup *cgroup;
  125. #endif
  126. };
  127. /* argument container for ops.exit_task() */
  128. struct scx_exit_task_args {
  129. /* Whether the task exited before running on sched_ext. */
  130. bool cancelled;
  131. };
  132. /* argument container for ops->cgroup_init() */
  133. struct scx_cgroup_init_args {
  134. /* the weight of the cgroup [1..10000] */
  135. u32 weight;
  136. };
  137. enum scx_cpu_preempt_reason {
  138. /* next task is being scheduled by &sched_class_rt */
  139. SCX_CPU_PREEMPT_RT,
  140. /* next task is being scheduled by &sched_class_dl */
  141. SCX_CPU_PREEMPT_DL,
  142. /* next task is being scheduled by &sched_class_stop */
  143. SCX_CPU_PREEMPT_STOP,
  144. /* unknown reason for SCX being preempted */
  145. SCX_CPU_PREEMPT_UNKNOWN,
  146. };
  147. /*
  148. * Argument container for ops->cpu_acquire(). Currently empty, but may be
  149. * expanded in the future.
  150. */
  151. struct scx_cpu_acquire_args {};
  152. /* argument container for ops->cpu_release() */
  153. struct scx_cpu_release_args {
  154. /* the reason the CPU was preempted */
  155. enum scx_cpu_preempt_reason reason;
  156. /* the task that's going to be scheduled on the CPU */
  157. struct task_struct *task;
  158. };
  159. /*
  160. * Informational context provided to dump operations.
  161. */
  162. struct scx_dump_ctx {
  163. enum scx_exit_kind kind;
  164. s64 exit_code;
  165. const char *reason;
  166. u64 at_ns;
  167. u64 at_jiffies;
  168. };
  169. /**
  170. * struct sched_ext_ops - Operation table for BPF scheduler implementation
  171. *
  172. * Userland can implement an arbitrary scheduling policy by implementing and
  173. * loading operations in this table.
  174. */
  175. struct sched_ext_ops {
  176. /**
  177. * select_cpu - Pick the target CPU for a task which is being woken up
  178. * @p: task being woken up
  179. * @prev_cpu: the cpu @p was on before sleeping
  180. * @wake_flags: SCX_WAKE_*
  181. *
  182. * Decision made here isn't final. @p may be moved to any CPU while it
  183. * is getting dispatched for execution later. However, as @p is not on
  184. * the rq at this point, getting the eventual execution CPU right here
  185. * saves a small bit of overhead down the line.
  186. *
  187. * If an idle CPU is returned, the CPU is kicked and will try to
  188. * dispatch. While an explicit custom mechanism can be added,
  189. * select_cpu() serves as the default way to wake up idle CPUs.
  190. *
  191. * @p may be dispatched directly by calling scx_bpf_dispatch(). If @p
  192. * is dispatched, the ops.enqueue() callback will be skipped. Finally,
  193. * if @p is dispatched to SCX_DSQ_LOCAL, it will be dispatched to the
  194. * local DSQ of whatever CPU is returned by this callback.
  195. */
  196. s32 (*select_cpu)(struct task_struct *p, s32 prev_cpu, u64 wake_flags);
  197. /**
  198. * enqueue - Enqueue a task on the BPF scheduler
  199. * @p: task being enqueued
  200. * @enq_flags: %SCX_ENQ_*
  201. *
  202. * @p is ready to run. Dispatch directly by calling scx_bpf_dispatch()
  203. * or enqueue on the BPF scheduler. If not directly dispatched, the bpf
  204. * scheduler owns @p and if it fails to dispatch @p, the task will
  205. * stall.
  206. *
  207. * If @p was dispatched from ops.select_cpu(), this callback is
  208. * skipped.
  209. */
  210. void (*enqueue)(struct task_struct *p, u64 enq_flags);
  211. /**
  212. * dequeue - Remove a task from the BPF scheduler
  213. * @p: task being dequeued
  214. * @deq_flags: %SCX_DEQ_*
  215. *
  216. * Remove @p from the BPF scheduler. This is usually called to isolate
  217. * the task while updating its scheduling properties (e.g. priority).
  218. *
  219. * The ext core keeps track of whether the BPF side owns a given task or
  220. * not and can gracefully ignore spurious dispatches from BPF side,
  221. * which makes it safe to not implement this method. However, depending
  222. * on the scheduling logic, this can lead to confusing behaviors - e.g.
  223. * scheduling position not being updated across a priority change.
  224. */
  225. void (*dequeue)(struct task_struct *p, u64 deq_flags);
  226. /**
  227. * dispatch - Dispatch tasks from the BPF scheduler and/or consume DSQs
  228. * @cpu: CPU to dispatch tasks for
  229. * @prev: previous task being switched out
  230. *
  231. * Called when a CPU's local dsq is empty. The operation should dispatch
  232. * one or more tasks from the BPF scheduler into the DSQs using
  233. * scx_bpf_dispatch() and/or consume user DSQs into the local DSQ using
  234. * scx_bpf_consume().
  235. *
  236. * The maximum number of times scx_bpf_dispatch() can be called without
  237. * an intervening scx_bpf_consume() is specified by
  238. * ops.dispatch_max_batch. See the comments on top of the two functions
  239. * for more details.
  240. *
  241. * When not %NULL, @prev is an SCX task with its slice depleted. If
  242. * @prev is still runnable as indicated by set %SCX_TASK_QUEUED in
  243. * @prev->scx.flags, it is not enqueued yet and will be enqueued after
  244. * ops.dispatch() returns. To keep executing @prev, return without
  245. * dispatching or consuming any tasks. Also see %SCX_OPS_ENQ_LAST.
  246. */
  247. void (*dispatch)(s32 cpu, struct task_struct *prev);
  248. /**
  249. * tick - Periodic tick
  250. * @p: task running currently
  251. *
  252. * This operation is called every 1/HZ seconds on CPUs which are
  253. * executing an SCX task. Setting @p->scx.slice to 0 will trigger an
  254. * immediate dispatch cycle on the CPU.
  255. */
  256. void (*tick)(struct task_struct *p);
  257. /**
  258. * runnable - A task is becoming runnable on its associated CPU
  259. * @p: task becoming runnable
  260. * @enq_flags: %SCX_ENQ_*
  261. *
  262. * This and the following three functions can be used to track a task's
  263. * execution state transitions. A task becomes ->runnable() on a CPU,
  264. * and then goes through one or more ->running() and ->stopping() pairs
  265. * as it runs on the CPU, and eventually becomes ->quiescent() when it's
  266. * done running on the CPU.
  267. *
  268. * @p is becoming runnable on the CPU because it's
  269. *
  270. * - waking up (%SCX_ENQ_WAKEUP)
  271. * - being moved from another CPU
  272. * - being restored after temporarily taken off the queue for an
  273. * attribute change.
  274. *
  275. * This and ->enqueue() are related but not coupled. This operation
  276. * notifies @p's state transition and may not be followed by ->enqueue()
  277. * e.g. when @p is being dispatched to a remote CPU, or when @p is
  278. * being enqueued on a CPU experiencing a hotplug event. Likewise, a
  279. * task may be ->enqueue()'d without being preceded by this operation
  280. * e.g. after exhausting its slice.
  281. */
  282. void (*runnable)(struct task_struct *p, u64 enq_flags);
  283. /**
  284. * running - A task is starting to run on its associated CPU
  285. * @p: task starting to run
  286. *
  287. * See ->runnable() for explanation on the task state notifiers.
  288. */
  289. void (*running)(struct task_struct *p);
  290. /**
  291. * stopping - A task is stopping execution
  292. * @p: task stopping to run
  293. * @runnable: is task @p still runnable?
  294. *
  295. * See ->runnable() for explanation on the task state notifiers. If
  296. * !@runnable, ->quiescent() will be invoked after this operation
  297. * returns.
  298. */
  299. void (*stopping)(struct task_struct *p, bool runnable);
  300. /**
  301. * quiescent - A task is becoming not runnable on its associated CPU
  302. * @p: task becoming not runnable
  303. * @deq_flags: %SCX_DEQ_*
  304. *
  305. * See ->runnable() for explanation on the task state notifiers.
  306. *
  307. * @p is becoming quiescent on the CPU because it's
  308. *
  309. * - sleeping (%SCX_DEQ_SLEEP)
  310. * - being moved to another CPU
  311. * - being temporarily taken off the queue for an attribute change
  312. * (%SCX_DEQ_SAVE)
  313. *
  314. * This and ->dequeue() are related but not coupled. This operation
  315. * notifies @p's state transition and may not be preceded by ->dequeue()
  316. * e.g. when @p is being dispatched to a remote CPU.
  317. */
  318. void (*quiescent)(struct task_struct *p, u64 deq_flags);
  319. /**
  320. * yield - Yield CPU
  321. * @from: yielding task
  322. * @to: optional yield target task
  323. *
  324. * If @to is NULL, @from is yielding the CPU to other runnable tasks.
  325. * The BPF scheduler should ensure that other available tasks are
  326. * dispatched before the yielding task. Return value is ignored in this
  327. * case.
  328. *
  329. * If @to is not-NULL, @from wants to yield the CPU to @to. If the bpf
  330. * scheduler can implement the request, return %true; otherwise, %false.
  331. */
  332. bool (*yield)(struct task_struct *from, struct task_struct *to);
  333. /**
  334. * core_sched_before - Task ordering for core-sched
  335. * @a: task A
  336. * @b: task B
  337. *
  338. * Used by core-sched to determine the ordering between two tasks. See
  339. * Documentation/admin-guide/hw-vuln/core-scheduling.rst for details on
  340. * core-sched.
  341. *
  342. * Both @a and @b are runnable and may or may not currently be queued on
  343. * the BPF scheduler. Should return %true if @a should run before @b.
  344. * %false if there's no required ordering or @b should run before @a.
  345. *
  346. * If not specified, the default is ordering them according to when they
  347. * became runnable.
  348. */
  349. bool (*core_sched_before)(struct task_struct *a, struct task_struct *b);
  350. /**
  351. * set_weight - Set task weight
  352. * @p: task to set weight for
  353. * @weight: new weight [1..10000]
  354. *
  355. * Update @p's weight to @weight.
  356. */
  357. void (*set_weight)(struct task_struct *p, u32 weight);
  358. /**
  359. * set_cpumask - Set CPU affinity
  360. * @p: task to set CPU affinity for
  361. * @cpumask: cpumask of cpus that @p can run on
  362. *
  363. * Update @p's CPU affinity to @cpumask.
  364. */
  365. void (*set_cpumask)(struct task_struct *p,
  366. const struct cpumask *cpumask);
  367. /**
  368. * update_idle - Update the idle state of a CPU
  369. * @cpu: CPU to udpate the idle state for
  370. * @idle: whether entering or exiting the idle state
  371. *
  372. * This operation is called when @rq's CPU goes or leaves the idle
  373. * state. By default, implementing this operation disables the built-in
  374. * idle CPU tracking and the following helpers become unavailable:
  375. *
  376. * - scx_bpf_select_cpu_dfl()
  377. * - scx_bpf_test_and_clear_cpu_idle()
  378. * - scx_bpf_pick_idle_cpu()
  379. *
  380. * The user also must implement ops.select_cpu() as the default
  381. * implementation relies on scx_bpf_select_cpu_dfl().
  382. *
  383. * Specify the %SCX_OPS_KEEP_BUILTIN_IDLE flag to keep the built-in idle
  384. * tracking.
  385. */
  386. void (*update_idle)(s32 cpu, bool idle);
  387. /**
  388. * cpu_acquire - A CPU is becoming available to the BPF scheduler
  389. * @cpu: The CPU being acquired by the BPF scheduler.
  390. * @args: Acquire arguments, see the struct definition.
  391. *
  392. * A CPU that was previously released from the BPF scheduler is now once
  393. * again under its control.
  394. */
  395. void (*cpu_acquire)(s32 cpu, struct scx_cpu_acquire_args *args);
  396. /**
  397. * cpu_release - A CPU is taken away from the BPF scheduler
  398. * @cpu: The CPU being released by the BPF scheduler.
  399. * @args: Release arguments, see the struct definition.
  400. *
  401. * The specified CPU is no longer under the control of the BPF
  402. * scheduler. This could be because it was preempted by a higher
  403. * priority sched_class, though there may be other reasons as well. The
  404. * caller should consult @args->reason to determine the cause.
  405. */
  406. void (*cpu_release)(s32 cpu, struct scx_cpu_release_args *args);
  407. /**
  408. * init_task - Initialize a task to run in a BPF scheduler
  409. * @p: task to initialize for BPF scheduling
  410. * @args: init arguments, see the struct definition
  411. *
  412. * Either we're loading a BPF scheduler or a new task is being forked.
  413. * Initialize @p for BPF scheduling. This operation may block and can
  414. * be used for allocations, and is called exactly once for a task.
  415. *
  416. * Return 0 for success, -errno for failure. An error return while
  417. * loading will abort loading of the BPF scheduler. During a fork, it
  418. * will abort that specific fork.
  419. */
  420. s32 (*init_task)(struct task_struct *p, struct scx_init_task_args *args);
  421. /**
  422. * exit_task - Exit a previously-running task from the system
  423. * @p: task to exit
  424. *
  425. * @p is exiting or the BPF scheduler is being unloaded. Perform any
  426. * necessary cleanup for @p.
  427. */
  428. void (*exit_task)(struct task_struct *p, struct scx_exit_task_args *args);
  429. /**
  430. * enable - Enable BPF scheduling for a task
  431. * @p: task to enable BPF scheduling for
  432. *
  433. * Enable @p for BPF scheduling. enable() is called on @p any time it
  434. * enters SCX, and is always paired with a matching disable().
  435. */
  436. void (*enable)(struct task_struct *p);
  437. /**
  438. * disable - Disable BPF scheduling for a task
  439. * @p: task to disable BPF scheduling for
  440. *
  441. * @p is exiting, leaving SCX or the BPF scheduler is being unloaded.
  442. * Disable BPF scheduling for @p. A disable() call is always matched
  443. * with a prior enable() call.
  444. */
  445. void (*disable)(struct task_struct *p);
  446. /**
  447. * dump - Dump BPF scheduler state on error
  448. * @ctx: debug dump context
  449. *
  450. * Use scx_bpf_dump() to generate BPF scheduler specific debug dump.
  451. */
  452. void (*dump)(struct scx_dump_ctx *ctx);
  453. /**
  454. * dump_cpu - Dump BPF scheduler state for a CPU on error
  455. * @ctx: debug dump context
  456. * @cpu: CPU to generate debug dump for
  457. * @idle: @cpu is currently idle without any runnable tasks
  458. *
  459. * Use scx_bpf_dump() to generate BPF scheduler specific debug dump for
  460. * @cpu. If @idle is %true and this operation doesn't produce any
  461. * output, @cpu is skipped for dump.
  462. */
  463. void (*dump_cpu)(struct scx_dump_ctx *ctx, s32 cpu, bool idle);
  464. /**
  465. * dump_task - Dump BPF scheduler state for a runnable task on error
  466. * @ctx: debug dump context
  467. * @p: runnable task to generate debug dump for
  468. *
  469. * Use scx_bpf_dump() to generate BPF scheduler specific debug dump for
  470. * @p.
  471. */
  472. void (*dump_task)(struct scx_dump_ctx *ctx, struct task_struct *p);
  473. #ifdef CONFIG_EXT_GROUP_SCHED
  474. /**
  475. * cgroup_init - Initialize a cgroup
  476. * @cgrp: cgroup being initialized
  477. * @args: init arguments, see the struct definition
  478. *
  479. * Either the BPF scheduler is being loaded or @cgrp created, initialize
  480. * @cgrp for sched_ext. This operation may block.
  481. *
  482. * Return 0 for success, -errno for failure. An error return while
  483. * loading will abort loading of the BPF scheduler. During cgroup
  484. * creation, it will abort the specific cgroup creation.
  485. */
  486. s32 (*cgroup_init)(struct cgroup *cgrp,
  487. struct scx_cgroup_init_args *args);
  488. /**
  489. * cgroup_exit - Exit a cgroup
  490. * @cgrp: cgroup being exited
  491. *
  492. * Either the BPF scheduler is being unloaded or @cgrp destroyed, exit
  493. * @cgrp for sched_ext. This operation my block.
  494. */
  495. void (*cgroup_exit)(struct cgroup *cgrp);
  496. /**
  497. * cgroup_prep_move - Prepare a task to be moved to a different cgroup
  498. * @p: task being moved
  499. * @from: cgroup @p is being moved from
  500. * @to: cgroup @p is being moved to
  501. *
  502. * Prepare @p for move from cgroup @from to @to. This operation may
  503. * block and can be used for allocations.
  504. *
  505. * Return 0 for success, -errno for failure. An error return aborts the
  506. * migration.
  507. */
  508. s32 (*cgroup_prep_move)(struct task_struct *p,
  509. struct cgroup *from, struct cgroup *to);
  510. /**
  511. * cgroup_move - Commit cgroup move
  512. * @p: task being moved
  513. * @from: cgroup @p is being moved from
  514. * @to: cgroup @p is being moved to
  515. *
  516. * Commit the move. @p is dequeued during this operation.
  517. */
  518. void (*cgroup_move)(struct task_struct *p,
  519. struct cgroup *from, struct cgroup *to);
  520. /**
  521. * cgroup_cancel_move - Cancel cgroup move
  522. * @p: task whose cgroup move is being canceled
  523. * @from: cgroup @p was being moved from
  524. * @to: cgroup @p was being moved to
  525. *
  526. * @p was cgroup_prep_move()'d but failed before reaching cgroup_move().
  527. * Undo the preparation.
  528. */
  529. void (*cgroup_cancel_move)(struct task_struct *p,
  530. struct cgroup *from, struct cgroup *to);
  531. /**
  532. * cgroup_set_weight - A cgroup's weight is being changed
  533. * @cgrp: cgroup whose weight is being updated
  534. * @weight: new weight [1..10000]
  535. *
  536. * Update @tg's weight to @weight.
  537. */
  538. void (*cgroup_set_weight)(struct cgroup *cgrp, u32 weight);
  539. #endif /* CONFIG_CGROUPS */
  540. /*
  541. * All online ops must come before ops.cpu_online().
  542. */
  543. /**
  544. * cpu_online - A CPU became online
  545. * @cpu: CPU which just came up
  546. *
  547. * @cpu just came online. @cpu will not call ops.enqueue() or
  548. * ops.dispatch(), nor run tasks associated with other CPUs beforehand.
  549. */
  550. void (*cpu_online)(s32 cpu);
  551. /**
  552. * cpu_offline - A CPU is going offline
  553. * @cpu: CPU which is going offline
  554. *
  555. * @cpu is going offline. @cpu will not call ops.enqueue() or
  556. * ops.dispatch(), nor run tasks associated with other CPUs afterwards.
  557. */
  558. void (*cpu_offline)(s32 cpu);
  559. /*
  560. * All CPU hotplug ops must come before ops.init().
  561. */
  562. /**
  563. * init - Initialize the BPF scheduler
  564. */
  565. s32 (*init)(void);
  566. /**
  567. * exit - Clean up after the BPF scheduler
  568. * @info: Exit info
  569. *
  570. * ops.exit() is also called on ops.init() failure, which is a bit
  571. * unusual. This is to allow rich reporting through @info on how
  572. * ops.init() failed.
  573. */
  574. void (*exit)(struct scx_exit_info *info);
  575. /**
  576. * dispatch_max_batch - Max nr of tasks that dispatch() can dispatch
  577. */
  578. u32 dispatch_max_batch;
  579. /**
  580. * flags - %SCX_OPS_* flags
  581. */
  582. u64 flags;
  583. /**
  584. * timeout_ms - The maximum amount of time, in milliseconds, that a
  585. * runnable task should be able to wait before being scheduled. The
  586. * maximum timeout may not exceed the default timeout of 30 seconds.
  587. *
  588. * Defaults to the maximum allowed timeout value of 30 seconds.
  589. */
  590. u32 timeout_ms;
  591. /**
  592. * exit_dump_len - scx_exit_info.dump buffer length. If 0, the default
  593. * value of 32768 is used.
  594. */
  595. u32 exit_dump_len;
  596. /**
  597. * hotplug_seq - A sequence number that may be set by the scheduler to
  598. * detect when a hotplug event has occurred during the loading process.
  599. * If 0, no detection occurs. Otherwise, the scheduler will fail to
  600. * load if the sequence number does not match @scx_hotplug_seq on the
  601. * enable path.
  602. */
  603. u64 hotplug_seq;
  604. /**
  605. * name - BPF scheduler's name
  606. *
  607. * Must be a non-zero valid BPF object name including only isalnum(),
  608. * '_' and '.' chars. Shows up in kernel.sched_ext_ops sysctl while the
  609. * BPF scheduler is enabled.
  610. */
  611. char name[SCX_OPS_NAME_LEN];
  612. };
  613. enum scx_opi {
  614. SCX_OPI_BEGIN = 0,
  615. SCX_OPI_NORMAL_BEGIN = 0,
  616. SCX_OPI_NORMAL_END = SCX_OP_IDX(cpu_online),
  617. SCX_OPI_CPU_HOTPLUG_BEGIN = SCX_OP_IDX(cpu_online),
  618. SCX_OPI_CPU_HOTPLUG_END = SCX_OP_IDX(init),
  619. SCX_OPI_END = SCX_OP_IDX(init),
  620. };
  621. enum scx_wake_flags {
  622. /* expose select WF_* flags as enums */
  623. SCX_WAKE_FORK = WF_FORK,
  624. SCX_WAKE_TTWU = WF_TTWU,
  625. SCX_WAKE_SYNC = WF_SYNC,
  626. };
  627. enum scx_enq_flags {
  628. /* expose select ENQUEUE_* flags as enums */
  629. SCX_ENQ_WAKEUP = ENQUEUE_WAKEUP,
  630. SCX_ENQ_HEAD = ENQUEUE_HEAD,
  631. SCX_ENQ_CPU_SELECTED = ENQUEUE_RQ_SELECTED,
  632. /* high 32bits are SCX specific */
  633. /*
  634. * Set the following to trigger preemption when calling
  635. * scx_bpf_dispatch() with a local dsq as the target. The slice of the
  636. * current task is cleared to zero and the CPU is kicked into the
  637. * scheduling path. Implies %SCX_ENQ_HEAD.
  638. */
  639. SCX_ENQ_PREEMPT = 1LLU << 32,
  640. /*
  641. * The task being enqueued was previously enqueued on the current CPU's
  642. * %SCX_DSQ_LOCAL, but was removed from it in a call to the
  643. * bpf_scx_reenqueue_local() kfunc. If bpf_scx_reenqueue_local() was
  644. * invoked in a ->cpu_release() callback, and the task is again
  645. * dispatched back to %SCX_LOCAL_DSQ by this current ->enqueue(), the
  646. * task will not be scheduled on the CPU until at least the next invocation
  647. * of the ->cpu_acquire() callback.
  648. */
  649. SCX_ENQ_REENQ = 1LLU << 40,
  650. /*
  651. * The task being enqueued is the only task available for the cpu. By
  652. * default, ext core keeps executing such tasks but when
  653. * %SCX_OPS_ENQ_LAST is specified, they're ops.enqueue()'d with the
  654. * %SCX_ENQ_LAST flag set.
  655. *
  656. * The BPF scheduler is responsible for triggering a follow-up
  657. * scheduling event. Otherwise, Execution may stall.
  658. */
  659. SCX_ENQ_LAST = 1LLU << 41,
  660. /* high 8 bits are internal */
  661. __SCX_ENQ_INTERNAL_MASK = 0xffLLU << 56,
  662. SCX_ENQ_CLEAR_OPSS = 1LLU << 56,
  663. SCX_ENQ_DSQ_PRIQ = 1LLU << 57,
  664. };
  665. enum scx_deq_flags {
  666. /* expose select DEQUEUE_* flags as enums */
  667. SCX_DEQ_SLEEP = DEQUEUE_SLEEP,
  668. /* high 32bits are SCX specific */
  669. /*
  670. * The generic core-sched layer decided to execute the task even though
  671. * it hasn't been dispatched yet. Dequeue from the BPF side.
  672. */
  673. SCX_DEQ_CORE_SCHED_EXEC = 1LLU << 32,
  674. };
  675. enum scx_pick_idle_cpu_flags {
  676. SCX_PICK_IDLE_CORE = 1LLU << 0, /* pick a CPU whose SMT siblings are also idle */
  677. };
  678. enum scx_kick_flags {
  679. /*
  680. * Kick the target CPU if idle. Guarantees that the target CPU goes
  681. * through at least one full scheduling cycle before going idle. If the
  682. * target CPU can be determined to be currently not idle and going to go
  683. * through a scheduling cycle before going idle, noop.
  684. */
  685. SCX_KICK_IDLE = 1LLU << 0,
  686. /*
  687. * Preempt the current task and execute the dispatch path. If the
  688. * current task of the target CPU is an SCX task, its ->scx.slice is
  689. * cleared to zero before the scheduling path is invoked so that the
  690. * task expires and the dispatch path is invoked.
  691. */
  692. SCX_KICK_PREEMPT = 1LLU << 1,
  693. /*
  694. * Wait for the CPU to be rescheduled. The scx_bpf_kick_cpu() call will
  695. * return after the target CPU finishes picking the next task.
  696. */
  697. SCX_KICK_WAIT = 1LLU << 2,
  698. };
  699. enum scx_tg_flags {
  700. SCX_TG_ONLINE = 1U << 0,
  701. SCX_TG_INITED = 1U << 1,
  702. };
  703. enum scx_ops_enable_state {
  704. SCX_OPS_ENABLING,
  705. SCX_OPS_ENABLED,
  706. SCX_OPS_DISABLING,
  707. SCX_OPS_DISABLED,
  708. };
  709. static const char *scx_ops_enable_state_str[] = {
  710. [SCX_OPS_ENABLING] = "enabling",
  711. [SCX_OPS_ENABLED] = "enabled",
  712. [SCX_OPS_DISABLING] = "disabling",
  713. [SCX_OPS_DISABLED] = "disabled",
  714. };
  715. /*
  716. * sched_ext_entity->ops_state
  717. *
  718. * Used to track the task ownership between the SCX core and the BPF scheduler.
  719. * State transitions look as follows:
  720. *
  721. * NONE -> QUEUEING -> QUEUED -> DISPATCHING
  722. * ^ | |
  723. * | v v
  724. * \-------------------------------/
  725. *
  726. * QUEUEING and DISPATCHING states can be waited upon. See wait_ops_state() call
  727. * sites for explanations on the conditions being waited upon and why they are
  728. * safe. Transitions out of them into NONE or QUEUED must store_release and the
  729. * waiters should load_acquire.
  730. *
  731. * Tracking scx_ops_state enables sched_ext core to reliably determine whether
  732. * any given task can be dispatched by the BPF scheduler at all times and thus
  733. * relaxes the requirements on the BPF scheduler. This allows the BPF scheduler
  734. * to try to dispatch any task anytime regardless of its state as the SCX core
  735. * can safely reject invalid dispatches.
  736. */
  737. enum scx_ops_state {
  738. SCX_OPSS_NONE, /* owned by the SCX core */
  739. SCX_OPSS_QUEUEING, /* in transit to the BPF scheduler */
  740. SCX_OPSS_QUEUED, /* owned by the BPF scheduler */
  741. SCX_OPSS_DISPATCHING, /* in transit back to the SCX core */
  742. /*
  743. * QSEQ brands each QUEUED instance so that, when dispatch races
  744. * dequeue/requeue, the dispatcher can tell whether it still has a claim
  745. * on the task being dispatched.
  746. *
  747. * As some 32bit archs can't do 64bit store_release/load_acquire,
  748. * p->scx.ops_state is atomic_long_t which leaves 30 bits for QSEQ on
  749. * 32bit machines. The dispatch race window QSEQ protects is very narrow
  750. * and runs with IRQ disabled. 30 bits should be sufficient.
  751. */
  752. SCX_OPSS_QSEQ_SHIFT = 2,
  753. };
  754. /* Use macros to ensure that the type is unsigned long for the masks */
  755. #define SCX_OPSS_STATE_MASK ((1LU << SCX_OPSS_QSEQ_SHIFT) - 1)
  756. #define SCX_OPSS_QSEQ_MASK (~SCX_OPSS_STATE_MASK)
  757. /*
  758. * During exit, a task may schedule after losing its PIDs. When disabling the
  759. * BPF scheduler, we need to be able to iterate tasks in every state to
  760. * guarantee system safety. Maintain a dedicated task list which contains every
  761. * task between its fork and eventual free.
  762. */
  763. static DEFINE_SPINLOCK(scx_tasks_lock);
  764. static LIST_HEAD(scx_tasks);
  765. /* ops enable/disable */
  766. static struct kthread_worker *scx_ops_helper;
  767. static DEFINE_MUTEX(scx_ops_enable_mutex);
  768. DEFINE_STATIC_KEY_FALSE(__scx_ops_enabled);
  769. DEFINE_STATIC_PERCPU_RWSEM(scx_fork_rwsem);
  770. static atomic_t scx_ops_enable_state_var = ATOMIC_INIT(SCX_OPS_DISABLED);
  771. static int scx_ops_bypass_depth;
  772. static DEFINE_RAW_SPINLOCK(__scx_ops_bypass_lock);
  773. static bool scx_ops_init_task_enabled;
  774. static bool scx_switching_all;
  775. DEFINE_STATIC_KEY_FALSE(__scx_switched_all);
  776. static struct sched_ext_ops scx_ops;
  777. static bool scx_warned_zero_slice;
  778. static DEFINE_STATIC_KEY_FALSE(scx_ops_enq_last);
  779. static DEFINE_STATIC_KEY_FALSE(scx_ops_enq_exiting);
  780. static DEFINE_STATIC_KEY_FALSE(scx_ops_cpu_preempt);
  781. static DEFINE_STATIC_KEY_FALSE(scx_builtin_idle_enabled);
  782. static struct static_key_false scx_has_op[SCX_OPI_END] =
  783. { [0 ... SCX_OPI_END-1] = STATIC_KEY_FALSE_INIT };
  784. static atomic_t scx_exit_kind = ATOMIC_INIT(SCX_EXIT_DONE);
  785. static struct scx_exit_info *scx_exit_info;
  786. static atomic_long_t scx_nr_rejected = ATOMIC_LONG_INIT(0);
  787. static atomic_long_t scx_hotplug_seq = ATOMIC_LONG_INIT(0);
  788. /*
  789. * A monotically increasing sequence number that is incremented every time a
  790. * scheduler is enabled. This can be used by to check if any custom sched_ext
  791. * scheduler has ever been used in the system.
  792. */
  793. static atomic_long_t scx_enable_seq = ATOMIC_LONG_INIT(0);
  794. /*
  795. * The maximum amount of time in jiffies that a task may be runnable without
  796. * being scheduled on a CPU. If this timeout is exceeded, it will trigger
  797. * scx_ops_error().
  798. */
  799. static unsigned long scx_watchdog_timeout;
  800. /*
  801. * The last time the delayed work was run. This delayed work relies on
  802. * ksoftirqd being able to run to service timer interrupts, so it's possible
  803. * that this work itself could get wedged. To account for this, we check that
  804. * it's not stalled in the timer tick, and trigger an error if it is.
  805. */
  806. static unsigned long scx_watchdog_timestamp = INITIAL_JIFFIES;
  807. static struct delayed_work scx_watchdog_work;
  808. /* idle tracking */
  809. #ifdef CONFIG_SMP
  810. #ifdef CONFIG_CPUMASK_OFFSTACK
  811. #define CL_ALIGNED_IF_ONSTACK
  812. #else
  813. #define CL_ALIGNED_IF_ONSTACK __cacheline_aligned_in_smp
  814. #endif
  815. static struct {
  816. cpumask_var_t cpu;
  817. cpumask_var_t smt;
  818. } idle_masks CL_ALIGNED_IF_ONSTACK;
  819. #endif /* CONFIG_SMP */
  820. /* for %SCX_KICK_WAIT */
  821. static unsigned long __percpu *scx_kick_cpus_pnt_seqs;
  822. /*
  823. * Direct dispatch marker.
  824. *
  825. * Non-NULL values are used for direct dispatch from enqueue path. A valid
  826. * pointer points to the task currently being enqueued. An ERR_PTR value is used
  827. * to indicate that direct dispatch has already happened.
  828. */
  829. static DEFINE_PER_CPU(struct task_struct *, direct_dispatch_task);
  830. /*
  831. * Dispatch queues.
  832. *
  833. * The global DSQ (%SCX_DSQ_GLOBAL) is split per-node for scalability. This is
  834. * to avoid live-locking in bypass mode where all tasks are dispatched to
  835. * %SCX_DSQ_GLOBAL and all CPUs consume from it. If per-node split isn't
  836. * sufficient, it can be further split.
  837. */
  838. static struct scx_dispatch_q **global_dsqs;
  839. static const struct rhashtable_params dsq_hash_params = {
  840. .key_len = 8,
  841. .key_offset = offsetof(struct scx_dispatch_q, id),
  842. .head_offset = offsetof(struct scx_dispatch_q, hash_node),
  843. };
  844. static struct rhashtable dsq_hash;
  845. static LLIST_HEAD(dsqs_to_free);
  846. /* dispatch buf */
  847. struct scx_dsp_buf_ent {
  848. struct task_struct *task;
  849. unsigned long qseq;
  850. u64 dsq_id;
  851. u64 enq_flags;
  852. };
  853. static u32 scx_dsp_max_batch;
  854. struct scx_dsp_ctx {
  855. struct rq *rq;
  856. u32 cursor;
  857. u32 nr_tasks;
  858. struct scx_dsp_buf_ent buf[];
  859. };
  860. static struct scx_dsp_ctx __percpu *scx_dsp_ctx;
  861. /* string formatting from BPF */
  862. struct scx_bstr_buf {
  863. u64 data[MAX_BPRINTF_VARARGS];
  864. char line[SCX_EXIT_MSG_LEN];
  865. };
  866. static DEFINE_RAW_SPINLOCK(scx_exit_bstr_buf_lock);
  867. static struct scx_bstr_buf scx_exit_bstr_buf;
  868. /* ops debug dump */
  869. struct scx_dump_data {
  870. s32 cpu;
  871. bool first;
  872. s32 cursor;
  873. struct seq_buf *s;
  874. const char *prefix;
  875. struct scx_bstr_buf buf;
  876. };
  877. static struct scx_dump_data scx_dump_data = {
  878. .cpu = -1,
  879. };
  880. /* /sys/kernel/sched_ext interface */
  881. static struct kset *scx_kset;
  882. static struct kobject *scx_root_kobj;
  883. #define CREATE_TRACE_POINTS
  884. #include <trace/events/sched_ext.h>
  885. static void process_ddsp_deferred_locals(struct rq *rq);
  886. static void scx_bpf_kick_cpu(s32 cpu, u64 flags);
  887. static __printf(3, 4) void scx_ops_exit_kind(enum scx_exit_kind kind,
  888. s64 exit_code,
  889. const char *fmt, ...);
  890. #define scx_ops_error_kind(err, fmt, args...) \
  891. scx_ops_exit_kind((err), 0, fmt, ##args)
  892. #define scx_ops_exit(code, fmt, args...) \
  893. scx_ops_exit_kind(SCX_EXIT_UNREG_KERN, (code), fmt, ##args)
  894. #define scx_ops_error(fmt, args...) \
  895. scx_ops_error_kind(SCX_EXIT_ERROR, fmt, ##args)
  896. #define SCX_HAS_OP(op) static_branch_likely(&scx_has_op[SCX_OP_IDX(op)])
  897. static long jiffies_delta_msecs(unsigned long at, unsigned long now)
  898. {
  899. if (time_after(at, now))
  900. return jiffies_to_msecs(at - now);
  901. else
  902. return -(long)jiffies_to_msecs(now - at);
  903. }
  904. /* if the highest set bit is N, return a mask with bits [N+1, 31] set */
  905. static u32 higher_bits(u32 flags)
  906. {
  907. return ~((1 << fls(flags)) - 1);
  908. }
  909. /* return the mask with only the highest bit set */
  910. static u32 highest_bit(u32 flags)
  911. {
  912. int bit = fls(flags);
  913. return ((u64)1 << bit) >> 1;
  914. }
  915. static bool u32_before(u32 a, u32 b)
  916. {
  917. return (s32)(a - b) < 0;
  918. }
  919. static struct scx_dispatch_q *find_global_dsq(struct task_struct *p)
  920. {
  921. return global_dsqs[cpu_to_node(task_cpu(p))];
  922. }
  923. static struct scx_dispatch_q *find_user_dsq(u64 dsq_id)
  924. {
  925. return rhashtable_lookup_fast(&dsq_hash, &dsq_id, dsq_hash_params);
  926. }
  927. /*
  928. * scx_kf_mask enforcement. Some kfuncs can only be called from specific SCX
  929. * ops. When invoking SCX ops, SCX_CALL_OP[_RET]() should be used to indicate
  930. * the allowed kfuncs and those kfuncs should use scx_kf_allowed() to check
  931. * whether it's running from an allowed context.
  932. *
  933. * @mask is constant, always inline to cull the mask calculations.
  934. */
  935. static __always_inline void scx_kf_allow(u32 mask)
  936. {
  937. /* nesting is allowed only in increasing scx_kf_mask order */
  938. WARN_ONCE((mask | higher_bits(mask)) & current->scx.kf_mask,
  939. "invalid nesting current->scx.kf_mask=0x%x mask=0x%x\n",
  940. current->scx.kf_mask, mask);
  941. current->scx.kf_mask |= mask;
  942. barrier();
  943. }
  944. static void scx_kf_disallow(u32 mask)
  945. {
  946. barrier();
  947. current->scx.kf_mask &= ~mask;
  948. }
  949. #define SCX_CALL_OP(mask, op, args...) \
  950. do { \
  951. if (mask) { \
  952. scx_kf_allow(mask); \
  953. scx_ops.op(args); \
  954. scx_kf_disallow(mask); \
  955. } else { \
  956. scx_ops.op(args); \
  957. } \
  958. } while (0)
  959. #define SCX_CALL_OP_RET(mask, op, args...) \
  960. ({ \
  961. __typeof__(scx_ops.op(args)) __ret; \
  962. if (mask) { \
  963. scx_kf_allow(mask); \
  964. __ret = scx_ops.op(args); \
  965. scx_kf_disallow(mask); \
  966. } else { \
  967. __ret = scx_ops.op(args); \
  968. } \
  969. __ret; \
  970. })
  971. /*
  972. * Some kfuncs are allowed only on the tasks that are subjects of the
  973. * in-progress scx_ops operation for, e.g., locking guarantees. To enforce such
  974. * restrictions, the following SCX_CALL_OP_*() variants should be used when
  975. * invoking scx_ops operations that take task arguments. These can only be used
  976. * for non-nesting operations due to the way the tasks are tracked.
  977. *
  978. * kfuncs which can only operate on such tasks can in turn use
  979. * scx_kf_allowed_on_arg_tasks() to test whether the invocation is allowed on
  980. * the specific task.
  981. */
  982. #define SCX_CALL_OP_TASK(mask, op, task, args...) \
  983. do { \
  984. BUILD_BUG_ON((mask) & ~__SCX_KF_TERMINAL); \
  985. current->scx.kf_tasks[0] = task; \
  986. SCX_CALL_OP(mask, op, task, ##args); \
  987. current->scx.kf_tasks[0] = NULL; \
  988. } while (0)
  989. #define SCX_CALL_OP_TASK_RET(mask, op, task, args...) \
  990. ({ \
  991. __typeof__(scx_ops.op(task, ##args)) __ret; \
  992. BUILD_BUG_ON((mask) & ~__SCX_KF_TERMINAL); \
  993. current->scx.kf_tasks[0] = task; \
  994. __ret = SCX_CALL_OP_RET(mask, op, task, ##args); \
  995. current->scx.kf_tasks[0] = NULL; \
  996. __ret; \
  997. })
  998. #define SCX_CALL_OP_2TASKS_RET(mask, op, task0, task1, args...) \
  999. ({ \
  1000. __typeof__(scx_ops.op(task0, task1, ##args)) __ret; \
  1001. BUILD_BUG_ON((mask) & ~__SCX_KF_TERMINAL); \
  1002. current->scx.kf_tasks[0] = task0; \
  1003. current->scx.kf_tasks[1] = task1; \
  1004. __ret = SCX_CALL_OP_RET(mask, op, task0, task1, ##args); \
  1005. current->scx.kf_tasks[0] = NULL; \
  1006. current->scx.kf_tasks[1] = NULL; \
  1007. __ret; \
  1008. })
  1009. /* @mask is constant, always inline to cull unnecessary branches */
  1010. static __always_inline bool scx_kf_allowed(u32 mask)
  1011. {
  1012. if (unlikely(!(current->scx.kf_mask & mask))) {
  1013. scx_ops_error("kfunc with mask 0x%x called from an operation only allowing 0x%x",
  1014. mask, current->scx.kf_mask);
  1015. return false;
  1016. }
  1017. /*
  1018. * Enforce nesting boundaries. e.g. A kfunc which can be called from
  1019. * DISPATCH must not be called if we're running DEQUEUE which is nested
  1020. * inside ops.dispatch(). We don't need to check boundaries for any
  1021. * blocking kfuncs as the verifier ensures they're only called from
  1022. * sleepable progs.
  1023. */
  1024. if (unlikely(highest_bit(mask) == SCX_KF_CPU_RELEASE &&
  1025. (current->scx.kf_mask & higher_bits(SCX_KF_CPU_RELEASE)))) {
  1026. scx_ops_error("cpu_release kfunc called from a nested operation");
  1027. return false;
  1028. }
  1029. if (unlikely(highest_bit(mask) == SCX_KF_DISPATCH &&
  1030. (current->scx.kf_mask & higher_bits(SCX_KF_DISPATCH)))) {
  1031. scx_ops_error("dispatch kfunc called from a nested operation");
  1032. return false;
  1033. }
  1034. return true;
  1035. }
  1036. /* see SCX_CALL_OP_TASK() */
  1037. static __always_inline bool scx_kf_allowed_on_arg_tasks(u32 mask,
  1038. struct task_struct *p)
  1039. {
  1040. if (!scx_kf_allowed(mask))
  1041. return false;
  1042. if (unlikely((p != current->scx.kf_tasks[0] &&
  1043. p != current->scx.kf_tasks[1]))) {
  1044. scx_ops_error("called on a task not being operated on");
  1045. return false;
  1046. }
  1047. return true;
  1048. }
  1049. static bool scx_kf_allowed_if_unlocked(void)
  1050. {
  1051. return !current->scx.kf_mask;
  1052. }
  1053. /**
  1054. * nldsq_next_task - Iterate to the next task in a non-local DSQ
  1055. * @dsq: user dsq being interated
  1056. * @cur: current position, %NULL to start iteration
  1057. * @rev: walk backwards
  1058. *
  1059. * Returns %NULL when iteration is finished.
  1060. */
  1061. static struct task_struct *nldsq_next_task(struct scx_dispatch_q *dsq,
  1062. struct task_struct *cur, bool rev)
  1063. {
  1064. struct list_head *list_node;
  1065. struct scx_dsq_list_node *dsq_lnode;
  1066. lockdep_assert_held(&dsq->lock);
  1067. if (cur)
  1068. list_node = &cur->scx.dsq_list.node;
  1069. else
  1070. list_node = &dsq->list;
  1071. /* find the next task, need to skip BPF iteration cursors */
  1072. do {
  1073. if (rev)
  1074. list_node = list_node->prev;
  1075. else
  1076. list_node = list_node->next;
  1077. if (list_node == &dsq->list)
  1078. return NULL;
  1079. dsq_lnode = container_of(list_node, struct scx_dsq_list_node,
  1080. node);
  1081. } while (dsq_lnode->flags & SCX_DSQ_LNODE_ITER_CURSOR);
  1082. return container_of(dsq_lnode, struct task_struct, scx.dsq_list);
  1083. }
  1084. #define nldsq_for_each_task(p, dsq) \
  1085. for ((p) = nldsq_next_task((dsq), NULL, false); (p); \
  1086. (p) = nldsq_next_task((dsq), (p), false))
  1087. /*
  1088. * BPF DSQ iterator. Tasks in a non-local DSQ can be iterated in [reverse]
  1089. * dispatch order. BPF-visible iterator is opaque and larger to allow future
  1090. * changes without breaking backward compatibility. Can be used with
  1091. * bpf_for_each(). See bpf_iter_scx_dsq_*().
  1092. */
  1093. enum scx_dsq_iter_flags {
  1094. /* iterate in the reverse dispatch order */
  1095. SCX_DSQ_ITER_REV = 1U << 16,
  1096. __SCX_DSQ_ITER_HAS_SLICE = 1U << 30,
  1097. __SCX_DSQ_ITER_HAS_VTIME = 1U << 31,
  1098. __SCX_DSQ_ITER_USER_FLAGS = SCX_DSQ_ITER_REV,
  1099. __SCX_DSQ_ITER_ALL_FLAGS = __SCX_DSQ_ITER_USER_FLAGS |
  1100. __SCX_DSQ_ITER_HAS_SLICE |
  1101. __SCX_DSQ_ITER_HAS_VTIME,
  1102. };
  1103. struct bpf_iter_scx_dsq_kern {
  1104. struct scx_dsq_list_node cursor;
  1105. struct scx_dispatch_q *dsq;
  1106. u64 slice;
  1107. u64 vtime;
  1108. } __attribute__((aligned(8)));
  1109. struct bpf_iter_scx_dsq {
  1110. u64 __opaque[6];
  1111. } __attribute__((aligned(8)));
  1112. /*
  1113. * SCX task iterator.
  1114. */
  1115. struct scx_task_iter {
  1116. struct sched_ext_entity cursor;
  1117. struct task_struct *locked;
  1118. struct rq *rq;
  1119. struct rq_flags rf;
  1120. u32 cnt;
  1121. };
  1122. /**
  1123. * scx_task_iter_start - Lock scx_tasks_lock and start a task iteration
  1124. * @iter: iterator to init
  1125. *
  1126. * Initialize @iter and return with scx_tasks_lock held. Once initialized, @iter
  1127. * must eventually be stopped with scx_task_iter_stop().
  1128. *
  1129. * scx_tasks_lock and the rq lock may be released using scx_task_iter_unlock()
  1130. * between this and the first next() call or between any two next() calls. If
  1131. * the locks are released between two next() calls, the caller is responsible
  1132. * for ensuring that the task being iterated remains accessible either through
  1133. * RCU read lock or obtaining a reference count.
  1134. *
  1135. * All tasks which existed when the iteration started are guaranteed to be
  1136. * visited as long as they still exist.
  1137. */
  1138. static void scx_task_iter_start(struct scx_task_iter *iter)
  1139. {
  1140. BUILD_BUG_ON(__SCX_DSQ_ITER_ALL_FLAGS &
  1141. ((1U << __SCX_DSQ_LNODE_PRIV_SHIFT) - 1));
  1142. spin_lock_irq(&scx_tasks_lock);
  1143. iter->cursor = (struct sched_ext_entity){ .flags = SCX_TASK_CURSOR };
  1144. list_add(&iter->cursor.tasks_node, &scx_tasks);
  1145. iter->locked = NULL;
  1146. iter->cnt = 0;
  1147. }
  1148. static void __scx_task_iter_rq_unlock(struct scx_task_iter *iter)
  1149. {
  1150. if (iter->locked) {
  1151. task_rq_unlock(iter->rq, iter->locked, &iter->rf);
  1152. iter->locked = NULL;
  1153. }
  1154. }
  1155. /**
  1156. * scx_task_iter_unlock - Unlock rq and scx_tasks_lock held by a task iterator
  1157. * @iter: iterator to unlock
  1158. *
  1159. * If @iter is in the middle of a locked iteration, it may be locking the rq of
  1160. * the task currently being visited in addition to scx_tasks_lock. Unlock both.
  1161. * This function can be safely called anytime during an iteration.
  1162. */
  1163. static void scx_task_iter_unlock(struct scx_task_iter *iter)
  1164. {
  1165. __scx_task_iter_rq_unlock(iter);
  1166. spin_unlock_irq(&scx_tasks_lock);
  1167. }
  1168. /**
  1169. * scx_task_iter_relock - Lock scx_tasks_lock released by scx_task_iter_unlock()
  1170. * @iter: iterator to re-lock
  1171. *
  1172. * Re-lock scx_tasks_lock unlocked by scx_task_iter_unlock(). Note that it
  1173. * doesn't re-lock the rq lock. Must be called before other iterator operations.
  1174. */
  1175. static void scx_task_iter_relock(struct scx_task_iter *iter)
  1176. {
  1177. spin_lock_irq(&scx_tasks_lock);
  1178. }
  1179. /**
  1180. * scx_task_iter_stop - Stop a task iteration and unlock scx_tasks_lock
  1181. * @iter: iterator to exit
  1182. *
  1183. * Exit a previously initialized @iter. Must be called with scx_tasks_lock held
  1184. * which is released on return. If the iterator holds a task's rq lock, that rq
  1185. * lock is also released. See scx_task_iter_start() for details.
  1186. */
  1187. static void scx_task_iter_stop(struct scx_task_iter *iter)
  1188. {
  1189. list_del_init(&iter->cursor.tasks_node);
  1190. scx_task_iter_unlock(iter);
  1191. }
  1192. /**
  1193. * scx_task_iter_next - Next task
  1194. * @iter: iterator to walk
  1195. *
  1196. * Visit the next task. See scx_task_iter_start() for details. Locks are dropped
  1197. * and re-acquired every %SCX_OPS_TASK_ITER_BATCH iterations to avoid causing
  1198. * stalls by holding scx_tasks_lock for too long.
  1199. */
  1200. static struct task_struct *scx_task_iter_next(struct scx_task_iter *iter)
  1201. {
  1202. struct list_head *cursor = &iter->cursor.tasks_node;
  1203. struct sched_ext_entity *pos;
  1204. if (!(++iter->cnt % SCX_OPS_TASK_ITER_BATCH)) {
  1205. scx_task_iter_unlock(iter);
  1206. cond_resched();
  1207. scx_task_iter_relock(iter);
  1208. }
  1209. list_for_each_entry(pos, cursor, tasks_node) {
  1210. if (&pos->tasks_node == &scx_tasks)
  1211. return NULL;
  1212. if (!(pos->flags & SCX_TASK_CURSOR)) {
  1213. list_move(cursor, &pos->tasks_node);
  1214. return container_of(pos, struct task_struct, scx);
  1215. }
  1216. }
  1217. /* can't happen, should always terminate at scx_tasks above */
  1218. BUG();
  1219. }
  1220. /**
  1221. * scx_task_iter_next_locked - Next non-idle task with its rq locked
  1222. * @iter: iterator to walk
  1223. * @include_dead: Whether we should include dead tasks in the iteration
  1224. *
  1225. * Visit the non-idle task with its rq lock held. Allows callers to specify
  1226. * whether they would like to filter out dead tasks. See scx_task_iter_start()
  1227. * for details.
  1228. */
  1229. static struct task_struct *scx_task_iter_next_locked(struct scx_task_iter *iter)
  1230. {
  1231. struct task_struct *p;
  1232. __scx_task_iter_rq_unlock(iter);
  1233. while ((p = scx_task_iter_next(iter))) {
  1234. /*
  1235. * scx_task_iter is used to prepare and move tasks into SCX
  1236. * while loading the BPF scheduler and vice-versa while
  1237. * unloading. The init_tasks ("swappers") should be excluded
  1238. * from the iteration because:
  1239. *
  1240. * - It's unsafe to use __setschduler_prio() on an init_task to
  1241. * determine the sched_class to use as it won't preserve its
  1242. * idle_sched_class.
  1243. *
  1244. * - ops.init/exit_task() can easily be confused if called with
  1245. * init_tasks as they, e.g., share PID 0.
  1246. *
  1247. * As init_tasks are never scheduled through SCX, they can be
  1248. * skipped safely. Note that is_idle_task() which tests %PF_IDLE
  1249. * doesn't work here:
  1250. *
  1251. * - %PF_IDLE may not be set for an init_task whose CPU hasn't
  1252. * yet been onlined.
  1253. *
  1254. * - %PF_IDLE can be set on tasks that are not init_tasks. See
  1255. * play_idle_precise() used by CONFIG_IDLE_INJECT.
  1256. *
  1257. * Test for idle_sched_class as only init_tasks are on it.
  1258. */
  1259. if (p->sched_class != &idle_sched_class)
  1260. break;
  1261. }
  1262. if (!p)
  1263. return NULL;
  1264. iter->rq = task_rq_lock(p, &iter->rf);
  1265. iter->locked = p;
  1266. return p;
  1267. }
  1268. static enum scx_ops_enable_state scx_ops_enable_state(void)
  1269. {
  1270. return atomic_read(&scx_ops_enable_state_var);
  1271. }
  1272. static enum scx_ops_enable_state
  1273. scx_ops_set_enable_state(enum scx_ops_enable_state to)
  1274. {
  1275. return atomic_xchg(&scx_ops_enable_state_var, to);
  1276. }
  1277. static bool scx_ops_tryset_enable_state(enum scx_ops_enable_state to,
  1278. enum scx_ops_enable_state from)
  1279. {
  1280. int from_v = from;
  1281. return atomic_try_cmpxchg(&scx_ops_enable_state_var, &from_v, to);
  1282. }
  1283. static bool scx_rq_bypassing(struct rq *rq)
  1284. {
  1285. return unlikely(rq->scx.flags & SCX_RQ_BYPASSING);
  1286. }
  1287. /**
  1288. * wait_ops_state - Busy-wait the specified ops state to end
  1289. * @p: target task
  1290. * @opss: state to wait the end of
  1291. *
  1292. * Busy-wait for @p to transition out of @opss. This can only be used when the
  1293. * state part of @opss is %SCX_QUEUEING or %SCX_DISPATCHING. This function also
  1294. * has load_acquire semantics to ensure that the caller can see the updates made
  1295. * in the enqueueing and dispatching paths.
  1296. */
  1297. static void wait_ops_state(struct task_struct *p, unsigned long opss)
  1298. {
  1299. do {
  1300. cpu_relax();
  1301. } while (atomic_long_read_acquire(&p->scx.ops_state) == opss);
  1302. }
  1303. /**
  1304. * ops_cpu_valid - Verify a cpu number
  1305. * @cpu: cpu number which came from a BPF ops
  1306. * @where: extra information reported on error
  1307. *
  1308. * @cpu is a cpu number which came from the BPF scheduler and can be any value.
  1309. * Verify that it is in range and one of the possible cpus. If invalid, trigger
  1310. * an ops error.
  1311. */
  1312. static bool ops_cpu_valid(s32 cpu, const char *where)
  1313. {
  1314. if (likely(cpu >= 0 && cpu < nr_cpu_ids && cpu_possible(cpu))) {
  1315. return true;
  1316. } else {
  1317. scx_ops_error("invalid CPU %d%s%s", cpu,
  1318. where ? " " : "", where ?: "");
  1319. return false;
  1320. }
  1321. }
  1322. /**
  1323. * ops_sanitize_err - Sanitize a -errno value
  1324. * @ops_name: operation to blame on failure
  1325. * @err: -errno value to sanitize
  1326. *
  1327. * Verify @err is a valid -errno. If not, trigger scx_ops_error() and return
  1328. * -%EPROTO. This is necessary because returning a rogue -errno up the chain can
  1329. * cause misbehaviors. For an example, a large negative return from
  1330. * ops.init_task() triggers an oops when passed up the call chain because the
  1331. * value fails IS_ERR() test after being encoded with ERR_PTR() and then is
  1332. * handled as a pointer.
  1333. */
  1334. static int ops_sanitize_err(const char *ops_name, s32 err)
  1335. {
  1336. if (err < 0 && err >= -MAX_ERRNO)
  1337. return err;
  1338. scx_ops_error("ops.%s() returned an invalid errno %d", ops_name, err);
  1339. return -EPROTO;
  1340. }
  1341. static void run_deferred(struct rq *rq)
  1342. {
  1343. process_ddsp_deferred_locals(rq);
  1344. }
  1345. #ifdef CONFIG_SMP
  1346. static void deferred_bal_cb_workfn(struct rq *rq)
  1347. {
  1348. run_deferred(rq);
  1349. }
  1350. #endif
  1351. static void deferred_irq_workfn(struct irq_work *irq_work)
  1352. {
  1353. struct rq *rq = container_of(irq_work, struct rq, scx.deferred_irq_work);
  1354. raw_spin_rq_lock(rq);
  1355. run_deferred(rq);
  1356. raw_spin_rq_unlock(rq);
  1357. }
  1358. /**
  1359. * schedule_deferred - Schedule execution of deferred actions on an rq
  1360. * @rq: target rq
  1361. *
  1362. * Schedule execution of deferred actions on @rq. Must be called with @rq
  1363. * locked. Deferred actions are executed with @rq locked but unpinned, and thus
  1364. * can unlock @rq to e.g. migrate tasks to other rqs.
  1365. */
  1366. static void schedule_deferred(struct rq *rq)
  1367. {
  1368. lockdep_assert_rq_held(rq);
  1369. #ifdef CONFIG_SMP
  1370. /*
  1371. * If in the middle of waking up a task, task_woken_scx() will be called
  1372. * afterwards which will then run the deferred actions, no need to
  1373. * schedule anything.
  1374. */
  1375. if (rq->scx.flags & SCX_RQ_IN_WAKEUP)
  1376. return;
  1377. /*
  1378. * If in balance, the balance callbacks will be called before rq lock is
  1379. * released. Schedule one.
  1380. */
  1381. if (rq->scx.flags & SCX_RQ_IN_BALANCE) {
  1382. queue_balance_callback(rq, &rq->scx.deferred_bal_cb,
  1383. deferred_bal_cb_workfn);
  1384. return;
  1385. }
  1386. #endif
  1387. /*
  1388. * No scheduler hooks available. Queue an irq work. They are executed on
  1389. * IRQ re-enable which may take a bit longer than the scheduler hooks.
  1390. * The above WAKEUP and BALANCE paths should cover most of the cases and
  1391. * the time to IRQ re-enable shouldn't be long.
  1392. */
  1393. irq_work_queue(&rq->scx.deferred_irq_work);
  1394. }
  1395. /**
  1396. * touch_core_sched - Update timestamp used for core-sched task ordering
  1397. * @rq: rq to read clock from, must be locked
  1398. * @p: task to update the timestamp for
  1399. *
  1400. * Update @p->scx.core_sched_at timestamp. This is used by scx_prio_less() to
  1401. * implement global or local-DSQ FIFO ordering for core-sched. Should be called
  1402. * when a task becomes runnable and its turn on the CPU ends (e.g. slice
  1403. * exhaustion).
  1404. */
  1405. static void touch_core_sched(struct rq *rq, struct task_struct *p)
  1406. {
  1407. lockdep_assert_rq_held(rq);
  1408. #ifdef CONFIG_SCHED_CORE
  1409. /*
  1410. * It's okay to update the timestamp spuriously. Use
  1411. * sched_core_disabled() which is cheaper than enabled().
  1412. *
  1413. * As this is used to determine ordering between tasks of sibling CPUs,
  1414. * it may be better to use per-core dispatch sequence instead.
  1415. */
  1416. if (!sched_core_disabled())
  1417. p->scx.core_sched_at = sched_clock_cpu(cpu_of(rq));
  1418. #endif
  1419. }
  1420. /**
  1421. * touch_core_sched_dispatch - Update core-sched timestamp on dispatch
  1422. * @rq: rq to read clock from, must be locked
  1423. * @p: task being dispatched
  1424. *
  1425. * If the BPF scheduler implements custom core-sched ordering via
  1426. * ops.core_sched_before(), @p->scx.core_sched_at is used to implement FIFO
  1427. * ordering within each local DSQ. This function is called from dispatch paths
  1428. * and updates @p->scx.core_sched_at if custom core-sched ordering is in effect.
  1429. */
  1430. static void touch_core_sched_dispatch(struct rq *rq, struct task_struct *p)
  1431. {
  1432. lockdep_assert_rq_held(rq);
  1433. #ifdef CONFIG_SCHED_CORE
  1434. if (SCX_HAS_OP(core_sched_before))
  1435. touch_core_sched(rq, p);
  1436. #endif
  1437. }
  1438. static void update_curr_scx(struct rq *rq)
  1439. {
  1440. struct task_struct *curr = rq->curr;
  1441. s64 delta_exec;
  1442. delta_exec = update_curr_common(rq);
  1443. if (unlikely(delta_exec <= 0))
  1444. return;
  1445. if (curr->scx.slice != SCX_SLICE_INF) {
  1446. curr->scx.slice -= min_t(u64, curr->scx.slice, delta_exec);
  1447. if (!curr->scx.slice)
  1448. touch_core_sched(rq, curr);
  1449. }
  1450. }
  1451. static bool scx_dsq_priq_less(struct rb_node *node_a,
  1452. const struct rb_node *node_b)
  1453. {
  1454. const struct task_struct *a =
  1455. container_of(node_a, struct task_struct, scx.dsq_priq);
  1456. const struct task_struct *b =
  1457. container_of(node_b, struct task_struct, scx.dsq_priq);
  1458. return time_before64(a->scx.dsq_vtime, b->scx.dsq_vtime);
  1459. }
  1460. static void dsq_mod_nr(struct scx_dispatch_q *dsq, s32 delta)
  1461. {
  1462. /* scx_bpf_dsq_nr_queued() reads ->nr without locking, use WRITE_ONCE() */
  1463. WRITE_ONCE(dsq->nr, dsq->nr + delta);
  1464. }
  1465. static void dispatch_enqueue(struct scx_dispatch_q *dsq, struct task_struct *p,
  1466. u64 enq_flags)
  1467. {
  1468. bool is_local = dsq->id == SCX_DSQ_LOCAL;
  1469. WARN_ON_ONCE(p->scx.dsq || !list_empty(&p->scx.dsq_list.node));
  1470. WARN_ON_ONCE((p->scx.dsq_flags & SCX_TASK_DSQ_ON_PRIQ) ||
  1471. !RB_EMPTY_NODE(&p->scx.dsq_priq));
  1472. if (!is_local) {
  1473. raw_spin_lock(&dsq->lock);
  1474. if (unlikely(dsq->id == SCX_DSQ_INVALID)) {
  1475. scx_ops_error("attempting to dispatch to a destroyed dsq");
  1476. /* fall back to the global dsq */
  1477. raw_spin_unlock(&dsq->lock);
  1478. dsq = find_global_dsq(p);
  1479. raw_spin_lock(&dsq->lock);
  1480. }
  1481. }
  1482. if (unlikely((dsq->id & SCX_DSQ_FLAG_BUILTIN) &&
  1483. (enq_flags & SCX_ENQ_DSQ_PRIQ))) {
  1484. /*
  1485. * SCX_DSQ_LOCAL and SCX_DSQ_GLOBAL DSQs always consume from
  1486. * their FIFO queues. To avoid confusion and accidentally
  1487. * starving vtime-dispatched tasks by FIFO-dispatched tasks, we
  1488. * disallow any internal DSQ from doing vtime ordering of
  1489. * tasks.
  1490. */
  1491. scx_ops_error("cannot use vtime ordering for built-in DSQs");
  1492. enq_flags &= ~SCX_ENQ_DSQ_PRIQ;
  1493. }
  1494. if (enq_flags & SCX_ENQ_DSQ_PRIQ) {
  1495. struct rb_node *rbp;
  1496. /*
  1497. * A PRIQ DSQ shouldn't be using FIFO enqueueing. As tasks are
  1498. * linked to both the rbtree and list on PRIQs, this can only be
  1499. * tested easily when adding the first task.
  1500. */
  1501. if (unlikely(RB_EMPTY_ROOT(&dsq->priq) &&
  1502. nldsq_next_task(dsq, NULL, false)))
  1503. scx_ops_error("DSQ ID 0x%016llx already had FIFO-enqueued tasks",
  1504. dsq->id);
  1505. p->scx.dsq_flags |= SCX_TASK_DSQ_ON_PRIQ;
  1506. rb_add(&p->scx.dsq_priq, &dsq->priq, scx_dsq_priq_less);
  1507. /*
  1508. * Find the previous task and insert after it on the list so
  1509. * that @dsq->list is vtime ordered.
  1510. */
  1511. rbp = rb_prev(&p->scx.dsq_priq);
  1512. if (rbp) {
  1513. struct task_struct *prev =
  1514. container_of(rbp, struct task_struct,
  1515. scx.dsq_priq);
  1516. list_add(&p->scx.dsq_list.node, &prev->scx.dsq_list.node);
  1517. } else {
  1518. list_add(&p->scx.dsq_list.node, &dsq->list);
  1519. }
  1520. } else {
  1521. /* a FIFO DSQ shouldn't be using PRIQ enqueuing */
  1522. if (unlikely(!RB_EMPTY_ROOT(&dsq->priq)))
  1523. scx_ops_error("DSQ ID 0x%016llx already had PRIQ-enqueued tasks",
  1524. dsq->id);
  1525. if (enq_flags & (SCX_ENQ_HEAD | SCX_ENQ_PREEMPT))
  1526. list_add(&p->scx.dsq_list.node, &dsq->list);
  1527. else
  1528. list_add_tail(&p->scx.dsq_list.node, &dsq->list);
  1529. }
  1530. /* seq records the order tasks are queued, used by BPF DSQ iterator */
  1531. dsq->seq++;
  1532. p->scx.dsq_seq = dsq->seq;
  1533. dsq_mod_nr(dsq, 1);
  1534. p->scx.dsq = dsq;
  1535. /*
  1536. * scx.ddsp_dsq_id and scx.ddsp_enq_flags are only relevant on the
  1537. * direct dispatch path, but we clear them here because the direct
  1538. * dispatch verdict may be overridden on the enqueue path during e.g.
  1539. * bypass.
  1540. */
  1541. p->scx.ddsp_dsq_id = SCX_DSQ_INVALID;
  1542. p->scx.ddsp_enq_flags = 0;
  1543. /*
  1544. * We're transitioning out of QUEUEING or DISPATCHING. store_release to
  1545. * match waiters' load_acquire.
  1546. */
  1547. if (enq_flags & SCX_ENQ_CLEAR_OPSS)
  1548. atomic_long_set_release(&p->scx.ops_state, SCX_OPSS_NONE);
  1549. if (is_local) {
  1550. struct rq *rq = container_of(dsq, struct rq, scx.local_dsq);
  1551. bool preempt = false;
  1552. if ((enq_flags & SCX_ENQ_PREEMPT) && p != rq->curr &&
  1553. rq->curr->sched_class == &ext_sched_class) {
  1554. rq->curr->scx.slice = 0;
  1555. preempt = true;
  1556. }
  1557. if (preempt || sched_class_above(&ext_sched_class,
  1558. rq->curr->sched_class))
  1559. resched_curr(rq);
  1560. } else {
  1561. raw_spin_unlock(&dsq->lock);
  1562. }
  1563. }
  1564. static void task_unlink_from_dsq(struct task_struct *p,
  1565. struct scx_dispatch_q *dsq)
  1566. {
  1567. WARN_ON_ONCE(list_empty(&p->scx.dsq_list.node));
  1568. if (p->scx.dsq_flags & SCX_TASK_DSQ_ON_PRIQ) {
  1569. rb_erase(&p->scx.dsq_priq, &dsq->priq);
  1570. RB_CLEAR_NODE(&p->scx.dsq_priq);
  1571. p->scx.dsq_flags &= ~SCX_TASK_DSQ_ON_PRIQ;
  1572. }
  1573. list_del_init(&p->scx.dsq_list.node);
  1574. dsq_mod_nr(dsq, -1);
  1575. }
  1576. static void dispatch_dequeue(struct rq *rq, struct task_struct *p)
  1577. {
  1578. struct scx_dispatch_q *dsq = p->scx.dsq;
  1579. bool is_local = dsq == &rq->scx.local_dsq;
  1580. if (!dsq) {
  1581. /*
  1582. * If !dsq && on-list, @p is on @rq's ddsp_deferred_locals.
  1583. * Unlinking is all that's needed to cancel.
  1584. */
  1585. if (unlikely(!list_empty(&p->scx.dsq_list.node)))
  1586. list_del_init(&p->scx.dsq_list.node);
  1587. /*
  1588. * When dispatching directly from the BPF scheduler to a local
  1589. * DSQ, the task isn't associated with any DSQ but
  1590. * @p->scx.holding_cpu may be set under the protection of
  1591. * %SCX_OPSS_DISPATCHING.
  1592. */
  1593. if (p->scx.holding_cpu >= 0)
  1594. p->scx.holding_cpu = -1;
  1595. return;
  1596. }
  1597. if (!is_local)
  1598. raw_spin_lock(&dsq->lock);
  1599. /*
  1600. * Now that we hold @dsq->lock, @p->holding_cpu and @p->scx.dsq_* can't
  1601. * change underneath us.
  1602. */
  1603. if (p->scx.holding_cpu < 0) {
  1604. /* @p must still be on @dsq, dequeue */
  1605. task_unlink_from_dsq(p, dsq);
  1606. } else {
  1607. /*
  1608. * We're racing against dispatch_to_local_dsq() which already
  1609. * removed @p from @dsq and set @p->scx.holding_cpu. Clear the
  1610. * holding_cpu which tells dispatch_to_local_dsq() that it lost
  1611. * the race.
  1612. */
  1613. WARN_ON_ONCE(!list_empty(&p->scx.dsq_list.node));
  1614. p->scx.holding_cpu = -1;
  1615. }
  1616. p->scx.dsq = NULL;
  1617. if (!is_local)
  1618. raw_spin_unlock(&dsq->lock);
  1619. }
  1620. static struct scx_dispatch_q *find_dsq_for_dispatch(struct rq *rq, u64 dsq_id,
  1621. struct task_struct *p)
  1622. {
  1623. struct scx_dispatch_q *dsq;
  1624. if (dsq_id == SCX_DSQ_LOCAL)
  1625. return &rq->scx.local_dsq;
  1626. if ((dsq_id & SCX_DSQ_LOCAL_ON) == SCX_DSQ_LOCAL_ON) {
  1627. s32 cpu = dsq_id & SCX_DSQ_LOCAL_CPU_MASK;
  1628. if (!ops_cpu_valid(cpu, "in SCX_DSQ_LOCAL_ON dispatch verdict"))
  1629. return find_global_dsq(p);
  1630. return &cpu_rq(cpu)->scx.local_dsq;
  1631. }
  1632. if (dsq_id == SCX_DSQ_GLOBAL)
  1633. dsq = find_global_dsq(p);
  1634. else
  1635. dsq = find_user_dsq(dsq_id);
  1636. if (unlikely(!dsq)) {
  1637. scx_ops_error("non-existent DSQ 0x%llx for %s[%d]",
  1638. dsq_id, p->comm, p->pid);
  1639. return find_global_dsq(p);
  1640. }
  1641. return dsq;
  1642. }
  1643. static void mark_direct_dispatch(struct task_struct *ddsp_task,
  1644. struct task_struct *p, u64 dsq_id,
  1645. u64 enq_flags)
  1646. {
  1647. /*
  1648. * Mark that dispatch already happened from ops.select_cpu() or
  1649. * ops.enqueue() by spoiling direct_dispatch_task with a non-NULL value
  1650. * which can never match a valid task pointer.
  1651. */
  1652. __this_cpu_write(direct_dispatch_task, ERR_PTR(-ESRCH));
  1653. /* @p must match the task on the enqueue path */
  1654. if (unlikely(p != ddsp_task)) {
  1655. if (IS_ERR(ddsp_task))
  1656. scx_ops_error("%s[%d] already direct-dispatched",
  1657. p->comm, p->pid);
  1658. else
  1659. scx_ops_error("scheduling for %s[%d] but trying to direct-dispatch %s[%d]",
  1660. ddsp_task->comm, ddsp_task->pid,
  1661. p->comm, p->pid);
  1662. return;
  1663. }
  1664. WARN_ON_ONCE(p->scx.ddsp_dsq_id != SCX_DSQ_INVALID);
  1665. WARN_ON_ONCE(p->scx.ddsp_enq_flags);
  1666. p->scx.ddsp_dsq_id = dsq_id;
  1667. p->scx.ddsp_enq_flags = enq_flags;
  1668. }
  1669. static void direct_dispatch(struct task_struct *p, u64 enq_flags)
  1670. {
  1671. struct rq *rq = task_rq(p);
  1672. struct scx_dispatch_q *dsq =
  1673. find_dsq_for_dispatch(rq, p->scx.ddsp_dsq_id, p);
  1674. touch_core_sched_dispatch(rq, p);
  1675. p->scx.ddsp_enq_flags |= enq_flags;
  1676. /*
  1677. * We are in the enqueue path with @rq locked and pinned, and thus can't
  1678. * double lock a remote rq and enqueue to its local DSQ. For
  1679. * DSQ_LOCAL_ON verdicts targeting the local DSQ of a remote CPU, defer
  1680. * the enqueue so that it's executed when @rq can be unlocked.
  1681. */
  1682. if (dsq->id == SCX_DSQ_LOCAL && dsq != &rq->scx.local_dsq) {
  1683. unsigned long opss;
  1684. opss = atomic_long_read(&p->scx.ops_state) & SCX_OPSS_STATE_MASK;
  1685. switch (opss & SCX_OPSS_STATE_MASK) {
  1686. case SCX_OPSS_NONE:
  1687. break;
  1688. case SCX_OPSS_QUEUEING:
  1689. /*
  1690. * As @p was never passed to the BPF side, _release is
  1691. * not strictly necessary. Still do it for consistency.
  1692. */
  1693. atomic_long_set_release(&p->scx.ops_state, SCX_OPSS_NONE);
  1694. break;
  1695. default:
  1696. WARN_ONCE(true, "sched_ext: %s[%d] has invalid ops state 0x%lx in direct_dispatch()",
  1697. p->comm, p->pid, opss);
  1698. atomic_long_set_release(&p->scx.ops_state, SCX_OPSS_NONE);
  1699. break;
  1700. }
  1701. WARN_ON_ONCE(p->scx.dsq || !list_empty(&p->scx.dsq_list.node));
  1702. list_add_tail(&p->scx.dsq_list.node,
  1703. &rq->scx.ddsp_deferred_locals);
  1704. schedule_deferred(rq);
  1705. return;
  1706. }
  1707. dispatch_enqueue(dsq, p, p->scx.ddsp_enq_flags | SCX_ENQ_CLEAR_OPSS);
  1708. }
  1709. static bool scx_rq_online(struct rq *rq)
  1710. {
  1711. /*
  1712. * Test both cpu_active() and %SCX_RQ_ONLINE. %SCX_RQ_ONLINE indicates
  1713. * the online state as seen from the BPF scheduler. cpu_active() test
  1714. * guarantees that, if this function returns %true, %SCX_RQ_ONLINE will
  1715. * stay set until the current scheduling operation is complete even if
  1716. * we aren't locking @rq.
  1717. */
  1718. return likely((rq->scx.flags & SCX_RQ_ONLINE) && cpu_active(cpu_of(rq)));
  1719. }
  1720. static void do_enqueue_task(struct rq *rq, struct task_struct *p, u64 enq_flags,
  1721. int sticky_cpu)
  1722. {
  1723. struct task_struct **ddsp_taskp;
  1724. unsigned long qseq;
  1725. WARN_ON_ONCE(!(p->scx.flags & SCX_TASK_QUEUED));
  1726. /* rq migration */
  1727. if (sticky_cpu == cpu_of(rq))
  1728. goto local_norefill;
  1729. /*
  1730. * If !scx_rq_online(), we already told the BPF scheduler that the CPU
  1731. * is offline and are just running the hotplug path. Don't bother the
  1732. * BPF scheduler.
  1733. */
  1734. if (!scx_rq_online(rq))
  1735. goto local;
  1736. if (scx_rq_bypassing(rq))
  1737. goto global;
  1738. if (p->scx.ddsp_dsq_id != SCX_DSQ_INVALID)
  1739. goto direct;
  1740. /* see %SCX_OPS_ENQ_EXITING */
  1741. if (!static_branch_unlikely(&scx_ops_enq_exiting) &&
  1742. unlikely(p->flags & PF_EXITING))
  1743. goto local;
  1744. if (!SCX_HAS_OP(enqueue))
  1745. goto global;
  1746. /* DSQ bypass didn't trigger, enqueue on the BPF scheduler */
  1747. qseq = rq->scx.ops_qseq++ << SCX_OPSS_QSEQ_SHIFT;
  1748. WARN_ON_ONCE(atomic_long_read(&p->scx.ops_state) != SCX_OPSS_NONE);
  1749. atomic_long_set(&p->scx.ops_state, SCX_OPSS_QUEUEING | qseq);
  1750. ddsp_taskp = this_cpu_ptr(&direct_dispatch_task);
  1751. WARN_ON_ONCE(*ddsp_taskp);
  1752. *ddsp_taskp = p;
  1753. SCX_CALL_OP_TASK(SCX_KF_ENQUEUE, enqueue, p, enq_flags);
  1754. *ddsp_taskp = NULL;
  1755. if (p->scx.ddsp_dsq_id != SCX_DSQ_INVALID)
  1756. goto direct;
  1757. /*
  1758. * If not directly dispatched, QUEUEING isn't clear yet and dispatch or
  1759. * dequeue may be waiting. The store_release matches their load_acquire.
  1760. */
  1761. atomic_long_set_release(&p->scx.ops_state, SCX_OPSS_QUEUED | qseq);
  1762. return;
  1763. direct:
  1764. direct_dispatch(p, enq_flags);
  1765. return;
  1766. local:
  1767. /*
  1768. * For task-ordering, slice refill must be treated as implying the end
  1769. * of the current slice. Otherwise, the longer @p stays on the CPU, the
  1770. * higher priority it becomes from scx_prio_less()'s POV.
  1771. */
  1772. touch_core_sched(rq, p);
  1773. p->scx.slice = SCX_SLICE_DFL;
  1774. local_norefill:
  1775. dispatch_enqueue(&rq->scx.local_dsq, p, enq_flags);
  1776. return;
  1777. global:
  1778. touch_core_sched(rq, p); /* see the comment in local: */
  1779. p->scx.slice = SCX_SLICE_DFL;
  1780. dispatch_enqueue(find_global_dsq(p), p, enq_flags);
  1781. }
  1782. static bool task_runnable(const struct task_struct *p)
  1783. {
  1784. return !list_empty(&p->scx.runnable_node);
  1785. }
  1786. static void set_task_runnable(struct rq *rq, struct task_struct *p)
  1787. {
  1788. lockdep_assert_rq_held(rq);
  1789. if (p->scx.flags & SCX_TASK_RESET_RUNNABLE_AT) {
  1790. p->scx.runnable_at = jiffies;
  1791. p->scx.flags &= ~SCX_TASK_RESET_RUNNABLE_AT;
  1792. }
  1793. /*
  1794. * list_add_tail() must be used. scx_ops_bypass() depends on tasks being
  1795. * appened to the runnable_list.
  1796. */
  1797. list_add_tail(&p->scx.runnable_node, &rq->scx.runnable_list);
  1798. }
  1799. static void clr_task_runnable(struct task_struct *p, bool reset_runnable_at)
  1800. {
  1801. list_del_init(&p->scx.runnable_node);
  1802. if (reset_runnable_at)
  1803. p->scx.flags |= SCX_TASK_RESET_RUNNABLE_AT;
  1804. }
  1805. static void enqueue_task_scx(struct rq *rq, struct task_struct *p, int enq_flags)
  1806. {
  1807. int sticky_cpu = p->scx.sticky_cpu;
  1808. if (enq_flags & ENQUEUE_WAKEUP)
  1809. rq->scx.flags |= SCX_RQ_IN_WAKEUP;
  1810. enq_flags |= rq->scx.extra_enq_flags;
  1811. if (sticky_cpu >= 0)
  1812. p->scx.sticky_cpu = -1;
  1813. /*
  1814. * Restoring a running task will be immediately followed by
  1815. * set_next_task_scx() which expects the task to not be on the BPF
  1816. * scheduler as tasks can only start running through local DSQs. Force
  1817. * direct-dispatch into the local DSQ by setting the sticky_cpu.
  1818. */
  1819. if (unlikely(enq_flags & ENQUEUE_RESTORE) && task_current(rq, p))
  1820. sticky_cpu = cpu_of(rq);
  1821. if (p->scx.flags & SCX_TASK_QUEUED) {
  1822. WARN_ON_ONCE(!task_runnable(p));
  1823. goto out;
  1824. }
  1825. set_task_runnable(rq, p);
  1826. p->scx.flags |= SCX_TASK_QUEUED;
  1827. rq->scx.nr_running++;
  1828. add_nr_running(rq, 1);
  1829. if (SCX_HAS_OP(runnable) && !task_on_rq_migrating(p))
  1830. SCX_CALL_OP_TASK(SCX_KF_REST, runnable, p, enq_flags);
  1831. if (enq_flags & SCX_ENQ_WAKEUP)
  1832. touch_core_sched(rq, p);
  1833. do_enqueue_task(rq, p, enq_flags, sticky_cpu);
  1834. out:
  1835. rq->scx.flags &= ~SCX_RQ_IN_WAKEUP;
  1836. }
  1837. static void ops_dequeue(struct task_struct *p, u64 deq_flags)
  1838. {
  1839. unsigned long opss;
  1840. /* dequeue is always temporary, don't reset runnable_at */
  1841. clr_task_runnable(p, false);
  1842. /* acquire ensures that we see the preceding updates on QUEUED */
  1843. opss = atomic_long_read_acquire(&p->scx.ops_state);
  1844. switch (opss & SCX_OPSS_STATE_MASK) {
  1845. case SCX_OPSS_NONE:
  1846. break;
  1847. case SCX_OPSS_QUEUEING:
  1848. /*
  1849. * QUEUEING is started and finished while holding @p's rq lock.
  1850. * As we're holding the rq lock now, we shouldn't see QUEUEING.
  1851. */
  1852. BUG();
  1853. case SCX_OPSS_QUEUED:
  1854. if (SCX_HAS_OP(dequeue))
  1855. SCX_CALL_OP_TASK(SCX_KF_REST, dequeue, p, deq_flags);
  1856. if (atomic_long_try_cmpxchg(&p->scx.ops_state, &opss,
  1857. SCX_OPSS_NONE))
  1858. break;
  1859. fallthrough;
  1860. case SCX_OPSS_DISPATCHING:
  1861. /*
  1862. * If @p is being dispatched from the BPF scheduler to a DSQ,
  1863. * wait for the transfer to complete so that @p doesn't get
  1864. * added to its DSQ after dequeueing is complete.
  1865. *
  1866. * As we're waiting on DISPATCHING with the rq locked, the
  1867. * dispatching side shouldn't try to lock the rq while
  1868. * DISPATCHING is set. See dispatch_to_local_dsq().
  1869. *
  1870. * DISPATCHING shouldn't have qseq set and control can reach
  1871. * here with NONE @opss from the above QUEUED case block.
  1872. * Explicitly wait on %SCX_OPSS_DISPATCHING instead of @opss.
  1873. */
  1874. wait_ops_state(p, SCX_OPSS_DISPATCHING);
  1875. BUG_ON(atomic_long_read(&p->scx.ops_state) != SCX_OPSS_NONE);
  1876. break;
  1877. }
  1878. }
  1879. static bool dequeue_task_scx(struct rq *rq, struct task_struct *p, int deq_flags)
  1880. {
  1881. if (!(p->scx.flags & SCX_TASK_QUEUED)) {
  1882. WARN_ON_ONCE(task_runnable(p));
  1883. return true;
  1884. }
  1885. ops_dequeue(p, deq_flags);
  1886. /*
  1887. * A currently running task which is going off @rq first gets dequeued
  1888. * and then stops running. As we want running <-> stopping transitions
  1889. * to be contained within runnable <-> quiescent transitions, trigger
  1890. * ->stopping() early here instead of in put_prev_task_scx().
  1891. *
  1892. * @p may go through multiple stopping <-> running transitions between
  1893. * here and put_prev_task_scx() if task attribute changes occur while
  1894. * balance_scx() leaves @rq unlocked. However, they don't contain any
  1895. * information meaningful to the BPF scheduler and can be suppressed by
  1896. * skipping the callbacks if the task is !QUEUED.
  1897. */
  1898. if (SCX_HAS_OP(stopping) && task_current(rq, p)) {
  1899. update_curr_scx(rq);
  1900. SCX_CALL_OP_TASK(SCX_KF_REST, stopping, p, false);
  1901. }
  1902. if (SCX_HAS_OP(quiescent) && !task_on_rq_migrating(p))
  1903. SCX_CALL_OP_TASK(SCX_KF_REST, quiescent, p, deq_flags);
  1904. if (deq_flags & SCX_DEQ_SLEEP)
  1905. p->scx.flags |= SCX_TASK_DEQD_FOR_SLEEP;
  1906. else
  1907. p->scx.flags &= ~SCX_TASK_DEQD_FOR_SLEEP;
  1908. p->scx.flags &= ~SCX_TASK_QUEUED;
  1909. rq->scx.nr_running--;
  1910. sub_nr_running(rq, 1);
  1911. dispatch_dequeue(rq, p);
  1912. return true;
  1913. }
  1914. static void yield_task_scx(struct rq *rq)
  1915. {
  1916. struct task_struct *p = rq->curr;
  1917. if (SCX_HAS_OP(yield))
  1918. SCX_CALL_OP_2TASKS_RET(SCX_KF_REST, yield, p, NULL);
  1919. else
  1920. p->scx.slice = 0;
  1921. }
  1922. static bool yield_to_task_scx(struct rq *rq, struct task_struct *to)
  1923. {
  1924. struct task_struct *from = rq->curr;
  1925. if (SCX_HAS_OP(yield))
  1926. return SCX_CALL_OP_2TASKS_RET(SCX_KF_REST, yield, from, to);
  1927. else
  1928. return false;
  1929. }
  1930. static void move_local_task_to_local_dsq(struct task_struct *p, u64 enq_flags,
  1931. struct scx_dispatch_q *src_dsq,
  1932. struct rq *dst_rq)
  1933. {
  1934. struct scx_dispatch_q *dst_dsq = &dst_rq->scx.local_dsq;
  1935. /* @dsq is locked and @p is on @dst_rq */
  1936. lockdep_assert_held(&src_dsq->lock);
  1937. lockdep_assert_rq_held(dst_rq);
  1938. WARN_ON_ONCE(p->scx.holding_cpu >= 0);
  1939. if (enq_flags & (SCX_ENQ_HEAD | SCX_ENQ_PREEMPT))
  1940. list_add(&p->scx.dsq_list.node, &dst_dsq->list);
  1941. else
  1942. list_add_tail(&p->scx.dsq_list.node, &dst_dsq->list);
  1943. dsq_mod_nr(dst_dsq, 1);
  1944. p->scx.dsq = dst_dsq;
  1945. }
  1946. #ifdef CONFIG_SMP
  1947. /**
  1948. * move_remote_task_to_local_dsq - Move a task from a foreign rq to a local DSQ
  1949. * @p: task to move
  1950. * @enq_flags: %SCX_ENQ_*
  1951. * @src_rq: rq to move the task from, locked on entry, released on return
  1952. * @dst_rq: rq to move the task into, locked on return
  1953. *
  1954. * Move @p which is currently on @src_rq to @dst_rq's local DSQ.
  1955. */
  1956. static void move_remote_task_to_local_dsq(struct task_struct *p, u64 enq_flags,
  1957. struct rq *src_rq, struct rq *dst_rq)
  1958. {
  1959. lockdep_assert_rq_held(src_rq);
  1960. /* the following marks @p MIGRATING which excludes dequeue */
  1961. deactivate_task(src_rq, p, 0);
  1962. set_task_cpu(p, cpu_of(dst_rq));
  1963. p->scx.sticky_cpu = cpu_of(dst_rq);
  1964. raw_spin_rq_unlock(src_rq);
  1965. raw_spin_rq_lock(dst_rq);
  1966. /*
  1967. * We want to pass scx-specific enq_flags but activate_task() will
  1968. * truncate the upper 32 bit. As we own @rq, we can pass them through
  1969. * @rq->scx.extra_enq_flags instead.
  1970. */
  1971. WARN_ON_ONCE(!cpumask_test_cpu(cpu_of(dst_rq), p->cpus_ptr));
  1972. WARN_ON_ONCE(dst_rq->scx.extra_enq_flags);
  1973. dst_rq->scx.extra_enq_flags = enq_flags;
  1974. activate_task(dst_rq, p, 0);
  1975. dst_rq->scx.extra_enq_flags = 0;
  1976. }
  1977. /*
  1978. * Similar to kernel/sched/core.c::is_cpu_allowed(). However, there are two
  1979. * differences:
  1980. *
  1981. * - is_cpu_allowed() asks "Can this task run on this CPU?" while
  1982. * task_can_run_on_remote_rq() asks "Can the BPF scheduler migrate the task to
  1983. * this CPU?".
  1984. *
  1985. * While migration is disabled, is_cpu_allowed() has to say "yes" as the task
  1986. * must be allowed to finish on the CPU that it's currently on regardless of
  1987. * the CPU state. However, task_can_run_on_remote_rq() must say "no" as the
  1988. * BPF scheduler shouldn't attempt to migrate a task which has migration
  1989. * disabled.
  1990. *
  1991. * - The BPF scheduler is bypassed while the rq is offline and we can always say
  1992. * no to the BPF scheduler initiated migrations while offline.
  1993. *
  1994. * The caller must ensure that @p and @rq are on different CPUs.
  1995. */
  1996. static bool task_can_run_on_remote_rq(struct task_struct *p, struct rq *rq,
  1997. bool trigger_error)
  1998. {
  1999. int cpu = cpu_of(rq);
  2000. SCHED_WARN_ON(task_cpu(p) == cpu);
  2001. /*
  2002. * If @p has migration disabled, @p->cpus_ptr is updated to contain only
  2003. * the pinned CPU in migrate_disable_switch() while @p is being switched
  2004. * out. However, put_prev_task_scx() is called before @p->cpus_ptr is
  2005. * updated and thus another CPU may see @p on a DSQ inbetween leading to
  2006. * @p passing the below task_allowed_on_cpu() check while migration is
  2007. * disabled.
  2008. *
  2009. * Test the migration disabled state first as the race window is narrow
  2010. * and the BPF scheduler failing to check migration disabled state can
  2011. * easily be masked if task_allowed_on_cpu() is done first.
  2012. */
  2013. if (unlikely(is_migration_disabled(p))) {
  2014. if (trigger_error)
  2015. scx_ops_error("SCX_DSQ_LOCAL[_ON] cannot move migration disabled %s[%d] from CPU %d to %d",
  2016. p->comm, p->pid, task_cpu(p), cpu);
  2017. return false;
  2018. }
  2019. /*
  2020. * We don't require the BPF scheduler to avoid dispatching to offline
  2021. * CPUs mostly for convenience but also because CPUs can go offline
  2022. * between scx_bpf_dispatch() calls and here. Trigger error iff the
  2023. * picked CPU is outside the allowed mask.
  2024. */
  2025. if (!task_allowed_on_cpu(p, cpu)) {
  2026. if (trigger_error)
  2027. scx_ops_error("SCX_DSQ_LOCAL[_ON] target CPU %d not allowed for %s[%d]",
  2028. cpu, p->comm, p->pid);
  2029. return false;
  2030. }
  2031. if (!scx_rq_online(rq))
  2032. return false;
  2033. return true;
  2034. }
  2035. /**
  2036. * unlink_dsq_and_lock_src_rq() - Unlink task from its DSQ and lock its task_rq
  2037. * @p: target task
  2038. * @dsq: locked DSQ @p is currently on
  2039. * @src_rq: rq @p is currently on, stable with @dsq locked
  2040. *
  2041. * Called with @dsq locked but no rq's locked. We want to move @p to a different
  2042. * DSQ, including any local DSQ, but are not locking @src_rq. Locking @src_rq is
  2043. * required when transferring into a local DSQ. Even when transferring into a
  2044. * non-local DSQ, it's better to use the same mechanism to protect against
  2045. * dequeues and maintain the invariant that @p->scx.dsq can only change while
  2046. * @src_rq is locked, which e.g. scx_dump_task() depends on.
  2047. *
  2048. * We want to grab @src_rq but that can deadlock if we try while locking @dsq,
  2049. * so we want to unlink @p from @dsq, drop its lock and then lock @src_rq. As
  2050. * this may race with dequeue, which can't drop the rq lock or fail, do a little
  2051. * dancing from our side.
  2052. *
  2053. * @p->scx.holding_cpu is set to this CPU before @dsq is unlocked. If @p gets
  2054. * dequeued after we unlock @dsq but before locking @src_rq, the holding_cpu
  2055. * would be cleared to -1. While other cpus may have updated it to different
  2056. * values afterwards, as this operation can't be preempted or recurse, the
  2057. * holding_cpu can never become this CPU again before we're done. Thus, we can
  2058. * tell whether we lost to dequeue by testing whether the holding_cpu still
  2059. * points to this CPU. See dispatch_dequeue() for the counterpart.
  2060. *
  2061. * On return, @dsq is unlocked and @src_rq is locked. Returns %true if @p is
  2062. * still valid. %false if lost to dequeue.
  2063. */
  2064. static bool unlink_dsq_and_lock_src_rq(struct task_struct *p,
  2065. struct scx_dispatch_q *dsq,
  2066. struct rq *src_rq)
  2067. {
  2068. s32 cpu = raw_smp_processor_id();
  2069. lockdep_assert_held(&dsq->lock);
  2070. WARN_ON_ONCE(p->scx.holding_cpu >= 0);
  2071. task_unlink_from_dsq(p, dsq);
  2072. p->scx.holding_cpu = cpu;
  2073. raw_spin_unlock(&dsq->lock);
  2074. raw_spin_rq_lock(src_rq);
  2075. /* task_rq couldn't have changed if we're still the holding cpu */
  2076. return likely(p->scx.holding_cpu == cpu) &&
  2077. !WARN_ON_ONCE(src_rq != task_rq(p));
  2078. }
  2079. static bool consume_remote_task(struct rq *this_rq, struct task_struct *p,
  2080. struct scx_dispatch_q *dsq, struct rq *src_rq)
  2081. {
  2082. raw_spin_rq_unlock(this_rq);
  2083. if (unlink_dsq_and_lock_src_rq(p, dsq, src_rq)) {
  2084. move_remote_task_to_local_dsq(p, 0, src_rq, this_rq);
  2085. return true;
  2086. } else {
  2087. raw_spin_rq_unlock(src_rq);
  2088. raw_spin_rq_lock(this_rq);
  2089. return false;
  2090. }
  2091. }
  2092. #else /* CONFIG_SMP */
  2093. static inline void move_remote_task_to_local_dsq(struct task_struct *p, u64 enq_flags, struct rq *src_rq, struct rq *dst_rq) { WARN_ON_ONCE(1); }
  2094. static inline bool task_can_run_on_remote_rq(struct task_struct *p, struct rq *rq, bool trigger_error) { return false; }
  2095. static inline bool consume_remote_task(struct rq *this_rq, struct task_struct *p, struct scx_dispatch_q *dsq, struct rq *task_rq) { return false; }
  2096. #endif /* CONFIG_SMP */
  2097. /**
  2098. * move_task_between_dsqs() - Move a task from one DSQ to another
  2099. * @p: target task
  2100. * @enq_flags: %SCX_ENQ_*
  2101. * @src_dsq: DSQ @p is currently on, must not be a local DSQ
  2102. * @dst_dsq: DSQ @p is being moved to, can be any DSQ
  2103. *
  2104. * Must be called with @p's task_rq and @src_dsq locked. If @dst_dsq is a local
  2105. * DSQ and @p is on a different CPU, @p will be migrated and thus its task_rq
  2106. * will change. As @p's task_rq is locked, this function doesn't need to use the
  2107. * holding_cpu mechanism.
  2108. *
  2109. * On return, @src_dsq is unlocked and only @p's new task_rq, which is the
  2110. * return value, is locked.
  2111. */
  2112. static struct rq *move_task_between_dsqs(struct task_struct *p, u64 enq_flags,
  2113. struct scx_dispatch_q *src_dsq,
  2114. struct scx_dispatch_q *dst_dsq)
  2115. {
  2116. struct rq *src_rq = task_rq(p), *dst_rq;
  2117. BUG_ON(src_dsq->id == SCX_DSQ_LOCAL);
  2118. lockdep_assert_held(&src_dsq->lock);
  2119. lockdep_assert_rq_held(src_rq);
  2120. if (dst_dsq->id == SCX_DSQ_LOCAL) {
  2121. dst_rq = container_of(dst_dsq, struct rq, scx.local_dsq);
  2122. if (src_rq != dst_rq &&
  2123. unlikely(!task_can_run_on_remote_rq(p, dst_rq, true))) {
  2124. dst_dsq = find_global_dsq(p);
  2125. dst_rq = src_rq;
  2126. }
  2127. } else {
  2128. /* no need to migrate if destination is a non-local DSQ */
  2129. dst_rq = src_rq;
  2130. }
  2131. /*
  2132. * Move @p into $dst_dsq. If $dst_dsq is the local DSQ of a different
  2133. * CPU, @p will be migrated.
  2134. */
  2135. if (dst_dsq->id == SCX_DSQ_LOCAL) {
  2136. /* @p is going from a non-local DSQ to a local DSQ */
  2137. if (src_rq == dst_rq) {
  2138. task_unlink_from_dsq(p, src_dsq);
  2139. move_local_task_to_local_dsq(p, enq_flags,
  2140. src_dsq, dst_rq);
  2141. raw_spin_unlock(&src_dsq->lock);
  2142. } else {
  2143. raw_spin_unlock(&src_dsq->lock);
  2144. move_remote_task_to_local_dsq(p, enq_flags,
  2145. src_rq, dst_rq);
  2146. }
  2147. } else {
  2148. /*
  2149. * @p is going from a non-local DSQ to a non-local DSQ. As
  2150. * $src_dsq is already locked, do an abbreviated dequeue.
  2151. */
  2152. task_unlink_from_dsq(p, src_dsq);
  2153. p->scx.dsq = NULL;
  2154. raw_spin_unlock(&src_dsq->lock);
  2155. dispatch_enqueue(dst_dsq, p, enq_flags);
  2156. }
  2157. return dst_rq;
  2158. }
  2159. static bool consume_dispatch_q(struct rq *rq, struct scx_dispatch_q *dsq)
  2160. {
  2161. struct task_struct *p;
  2162. retry:
  2163. /*
  2164. * The caller can't expect to successfully consume a task if the task's
  2165. * addition to @dsq isn't guaranteed to be visible somehow. Test
  2166. * @dsq->list without locking and skip if it seems empty.
  2167. */
  2168. if (list_empty(&dsq->list))
  2169. return false;
  2170. raw_spin_lock(&dsq->lock);
  2171. nldsq_for_each_task(p, dsq) {
  2172. struct rq *task_rq = task_rq(p);
  2173. if (rq == task_rq) {
  2174. task_unlink_from_dsq(p, dsq);
  2175. move_local_task_to_local_dsq(p, 0, dsq, rq);
  2176. raw_spin_unlock(&dsq->lock);
  2177. return true;
  2178. }
  2179. if (task_can_run_on_remote_rq(p, rq, false)) {
  2180. if (likely(consume_remote_task(rq, p, dsq, task_rq)))
  2181. return true;
  2182. goto retry;
  2183. }
  2184. }
  2185. raw_spin_unlock(&dsq->lock);
  2186. return false;
  2187. }
  2188. static bool consume_global_dsq(struct rq *rq)
  2189. {
  2190. int node = cpu_to_node(cpu_of(rq));
  2191. return consume_dispatch_q(rq, global_dsqs[node]);
  2192. }
  2193. /**
  2194. * dispatch_to_local_dsq - Dispatch a task to a local dsq
  2195. * @rq: current rq which is locked
  2196. * @dst_dsq: destination DSQ
  2197. * @p: task to dispatch
  2198. * @enq_flags: %SCX_ENQ_*
  2199. *
  2200. * We're holding @rq lock and want to dispatch @p to @dst_dsq which is a local
  2201. * DSQ. This function performs all the synchronization dancing needed because
  2202. * local DSQs are protected with rq locks.
  2203. *
  2204. * The caller must have exclusive ownership of @p (e.g. through
  2205. * %SCX_OPSS_DISPATCHING).
  2206. */
  2207. static void dispatch_to_local_dsq(struct rq *rq, struct scx_dispatch_q *dst_dsq,
  2208. struct task_struct *p, u64 enq_flags)
  2209. {
  2210. struct rq *src_rq = task_rq(p);
  2211. struct rq *dst_rq = container_of(dst_dsq, struct rq, scx.local_dsq);
  2212. #ifdef CONFIG_SMP
  2213. struct rq *locked_rq = rq;
  2214. #endif
  2215. /*
  2216. * We're synchronized against dequeue through DISPATCHING. As @p can't
  2217. * be dequeued, its task_rq and cpus_allowed are stable too.
  2218. *
  2219. * If dispatching to @rq that @p is already on, no lock dancing needed.
  2220. */
  2221. if (rq == src_rq && rq == dst_rq) {
  2222. dispatch_enqueue(dst_dsq, p, enq_flags | SCX_ENQ_CLEAR_OPSS);
  2223. return;
  2224. }
  2225. #ifdef CONFIG_SMP
  2226. if (src_rq != dst_rq &&
  2227. unlikely(!task_can_run_on_remote_rq(p, dst_rq, true))) {
  2228. dispatch_enqueue(find_global_dsq(p), p,
  2229. enq_flags | SCX_ENQ_CLEAR_OPSS);
  2230. return;
  2231. }
  2232. /*
  2233. * @p is on a possibly remote @src_rq which we need to lock to move the
  2234. * task. If dequeue is in progress, it'd be locking @src_rq and waiting
  2235. * on DISPATCHING, so we can't grab @src_rq lock while holding
  2236. * DISPATCHING.
  2237. *
  2238. * As DISPATCHING guarantees that @p is wholly ours, we can pretend that
  2239. * we're moving from a DSQ and use the same mechanism - mark the task
  2240. * under transfer with holding_cpu, release DISPATCHING and then follow
  2241. * the same protocol. See unlink_dsq_and_lock_src_rq().
  2242. */
  2243. p->scx.holding_cpu = raw_smp_processor_id();
  2244. /* store_release ensures that dequeue sees the above */
  2245. atomic_long_set_release(&p->scx.ops_state, SCX_OPSS_NONE);
  2246. /* switch to @src_rq lock */
  2247. if (locked_rq != src_rq) {
  2248. raw_spin_rq_unlock(locked_rq);
  2249. locked_rq = src_rq;
  2250. raw_spin_rq_lock(src_rq);
  2251. }
  2252. /* task_rq couldn't have changed if we're still the holding cpu */
  2253. if (likely(p->scx.holding_cpu == raw_smp_processor_id()) &&
  2254. !WARN_ON_ONCE(src_rq != task_rq(p))) {
  2255. /*
  2256. * If @p is staying on the same rq, there's no need to go
  2257. * through the full deactivate/activate cycle. Optimize by
  2258. * abbreviating move_remote_task_to_local_dsq().
  2259. */
  2260. if (src_rq == dst_rq) {
  2261. p->scx.holding_cpu = -1;
  2262. dispatch_enqueue(&dst_rq->scx.local_dsq, p, enq_flags);
  2263. } else {
  2264. move_remote_task_to_local_dsq(p, enq_flags,
  2265. src_rq, dst_rq);
  2266. /* task has been moved to dst_rq, which is now locked */
  2267. locked_rq = dst_rq;
  2268. }
  2269. /* if the destination CPU is idle, wake it up */
  2270. if (sched_class_above(p->sched_class, dst_rq->curr->sched_class))
  2271. resched_curr(dst_rq);
  2272. }
  2273. /* switch back to @rq lock */
  2274. if (locked_rq != rq) {
  2275. raw_spin_rq_unlock(locked_rq);
  2276. raw_spin_rq_lock(rq);
  2277. }
  2278. #else /* CONFIG_SMP */
  2279. BUG(); /* control can not reach here on UP */
  2280. #endif /* CONFIG_SMP */
  2281. }
  2282. /**
  2283. * finish_dispatch - Asynchronously finish dispatching a task
  2284. * @rq: current rq which is locked
  2285. * @p: task to finish dispatching
  2286. * @qseq_at_dispatch: qseq when @p started getting dispatched
  2287. * @dsq_id: destination DSQ ID
  2288. * @enq_flags: %SCX_ENQ_*
  2289. *
  2290. * Dispatching to local DSQs may need to wait for queueing to complete or
  2291. * require rq lock dancing. As we don't wanna do either while inside
  2292. * ops.dispatch() to avoid locking order inversion, we split dispatching into
  2293. * two parts. scx_bpf_dispatch() which is called by ops.dispatch() records the
  2294. * task and its qseq. Once ops.dispatch() returns, this function is called to
  2295. * finish up.
  2296. *
  2297. * There is no guarantee that @p is still valid for dispatching or even that it
  2298. * was valid in the first place. Make sure that the task is still owned by the
  2299. * BPF scheduler and claim the ownership before dispatching.
  2300. */
  2301. static void finish_dispatch(struct rq *rq, struct task_struct *p,
  2302. unsigned long qseq_at_dispatch,
  2303. u64 dsq_id, u64 enq_flags)
  2304. {
  2305. struct scx_dispatch_q *dsq;
  2306. unsigned long opss;
  2307. touch_core_sched_dispatch(rq, p);
  2308. retry:
  2309. /*
  2310. * No need for _acquire here. @p is accessed only after a successful
  2311. * try_cmpxchg to DISPATCHING.
  2312. */
  2313. opss = atomic_long_read(&p->scx.ops_state);
  2314. switch (opss & SCX_OPSS_STATE_MASK) {
  2315. case SCX_OPSS_DISPATCHING:
  2316. case SCX_OPSS_NONE:
  2317. /* someone else already got to it */
  2318. return;
  2319. case SCX_OPSS_QUEUED:
  2320. /*
  2321. * If qseq doesn't match, @p has gone through at least one
  2322. * dispatch/dequeue and re-enqueue cycle between
  2323. * scx_bpf_dispatch() and here and we have no claim on it.
  2324. */
  2325. if ((opss & SCX_OPSS_QSEQ_MASK) != qseq_at_dispatch)
  2326. return;
  2327. /*
  2328. * While we know @p is accessible, we don't yet have a claim on
  2329. * it - the BPF scheduler is allowed to dispatch tasks
  2330. * spuriously and there can be a racing dequeue attempt. Let's
  2331. * claim @p by atomically transitioning it from QUEUED to
  2332. * DISPATCHING.
  2333. */
  2334. if (likely(atomic_long_try_cmpxchg(&p->scx.ops_state, &opss,
  2335. SCX_OPSS_DISPATCHING)))
  2336. break;
  2337. goto retry;
  2338. case SCX_OPSS_QUEUEING:
  2339. /*
  2340. * do_enqueue_task() is in the process of transferring the task
  2341. * to the BPF scheduler while holding @p's rq lock. As we aren't
  2342. * holding any kernel or BPF resource that the enqueue path may
  2343. * depend upon, it's safe to wait.
  2344. */
  2345. wait_ops_state(p, opss);
  2346. goto retry;
  2347. }
  2348. BUG_ON(!(p->scx.flags & SCX_TASK_QUEUED));
  2349. dsq = find_dsq_for_dispatch(this_rq(), dsq_id, p);
  2350. if (dsq->id == SCX_DSQ_LOCAL)
  2351. dispatch_to_local_dsq(rq, dsq, p, enq_flags);
  2352. else
  2353. dispatch_enqueue(dsq, p, enq_flags | SCX_ENQ_CLEAR_OPSS);
  2354. }
  2355. static void flush_dispatch_buf(struct rq *rq)
  2356. {
  2357. struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
  2358. u32 u;
  2359. for (u = 0; u < dspc->cursor; u++) {
  2360. struct scx_dsp_buf_ent *ent = &dspc->buf[u];
  2361. finish_dispatch(rq, ent->task, ent->qseq, ent->dsq_id,
  2362. ent->enq_flags);
  2363. }
  2364. dspc->nr_tasks += dspc->cursor;
  2365. dspc->cursor = 0;
  2366. }
  2367. static int balance_one(struct rq *rq, struct task_struct *prev)
  2368. {
  2369. struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
  2370. bool prev_on_scx = prev->sched_class == &ext_sched_class;
  2371. bool prev_on_rq = prev->scx.flags & SCX_TASK_QUEUED;
  2372. int nr_loops = SCX_DSP_MAX_LOOPS;
  2373. lockdep_assert_rq_held(rq);
  2374. rq->scx.flags |= SCX_RQ_IN_BALANCE;
  2375. rq->scx.flags &= ~(SCX_RQ_BAL_PENDING | SCX_RQ_BAL_KEEP);
  2376. if (static_branch_unlikely(&scx_ops_cpu_preempt) &&
  2377. unlikely(rq->scx.cpu_released)) {
  2378. /*
  2379. * If the previous sched_class for the current CPU was not SCX,
  2380. * notify the BPF scheduler that it again has control of the
  2381. * core. This callback complements ->cpu_release(), which is
  2382. * emitted in scx_next_task_picked().
  2383. */
  2384. if (SCX_HAS_OP(cpu_acquire))
  2385. SCX_CALL_OP(SCX_KF_REST, cpu_acquire, cpu_of(rq), NULL);
  2386. rq->scx.cpu_released = false;
  2387. }
  2388. if (prev_on_scx) {
  2389. update_curr_scx(rq);
  2390. /*
  2391. * If @prev is runnable & has slice left, it has priority and
  2392. * fetching more just increases latency for the fetched tasks.
  2393. * Tell pick_task_scx() to keep running @prev. If the BPF
  2394. * scheduler wants to handle this explicitly, it should
  2395. * implement ->cpu_release().
  2396. *
  2397. * See scx_ops_disable_workfn() for the explanation on the
  2398. * bypassing test.
  2399. */
  2400. if (prev_on_rq && prev->scx.slice && !scx_rq_bypassing(rq)) {
  2401. rq->scx.flags |= SCX_RQ_BAL_KEEP;
  2402. goto has_tasks;
  2403. }
  2404. }
  2405. /* if there already are tasks to run, nothing to do */
  2406. if (rq->scx.local_dsq.nr)
  2407. goto has_tasks;
  2408. if (consume_global_dsq(rq))
  2409. goto has_tasks;
  2410. if (!SCX_HAS_OP(dispatch) || scx_rq_bypassing(rq) || !scx_rq_online(rq))
  2411. goto no_tasks;
  2412. dspc->rq = rq;
  2413. /*
  2414. * The dispatch loop. Because flush_dispatch_buf() may drop the rq lock,
  2415. * the local DSQ might still end up empty after a successful
  2416. * ops.dispatch(). If the local DSQ is empty even after ops.dispatch()
  2417. * produced some tasks, retry. The BPF scheduler may depend on this
  2418. * looping behavior to simplify its implementation.
  2419. */
  2420. do {
  2421. dspc->nr_tasks = 0;
  2422. SCX_CALL_OP(SCX_KF_DISPATCH, dispatch, cpu_of(rq),
  2423. prev_on_scx ? prev : NULL);
  2424. flush_dispatch_buf(rq);
  2425. if (prev_on_rq && prev->scx.slice) {
  2426. rq->scx.flags |= SCX_RQ_BAL_KEEP;
  2427. goto has_tasks;
  2428. }
  2429. if (rq->scx.local_dsq.nr)
  2430. goto has_tasks;
  2431. if (consume_global_dsq(rq))
  2432. goto has_tasks;
  2433. /*
  2434. * ops.dispatch() can trap us in this loop by repeatedly
  2435. * dispatching ineligible tasks. Break out once in a while to
  2436. * allow the watchdog to run. As IRQ can't be enabled in
  2437. * balance(), we want to complete this scheduling cycle and then
  2438. * start a new one. IOW, we want to call resched_curr() on the
  2439. * next, most likely idle, task, not the current one. Use
  2440. * scx_bpf_kick_cpu() for deferred kicking.
  2441. */
  2442. if (unlikely(!--nr_loops)) {
  2443. scx_bpf_kick_cpu(cpu_of(rq), 0);
  2444. break;
  2445. }
  2446. } while (dspc->nr_tasks);
  2447. no_tasks:
  2448. /*
  2449. * Didn't find another task to run. Keep running @prev unless
  2450. * %SCX_OPS_ENQ_LAST is in effect.
  2451. */
  2452. if (prev_on_rq && (!static_branch_unlikely(&scx_ops_enq_last) ||
  2453. scx_rq_bypassing(rq))) {
  2454. rq->scx.flags |= SCX_RQ_BAL_KEEP;
  2455. goto has_tasks;
  2456. }
  2457. rq->scx.flags &= ~SCX_RQ_IN_BALANCE;
  2458. return false;
  2459. has_tasks:
  2460. rq->scx.flags &= ~SCX_RQ_IN_BALANCE;
  2461. return true;
  2462. }
  2463. static int balance_scx(struct rq *rq, struct task_struct *prev,
  2464. struct rq_flags *rf)
  2465. {
  2466. int ret;
  2467. rq_unpin_lock(rq, rf);
  2468. ret = balance_one(rq, prev);
  2469. #ifdef CONFIG_SCHED_SMT
  2470. /*
  2471. * When core-sched is enabled, this ops.balance() call will be followed
  2472. * by pick_task_scx() on this CPU and the SMT siblings. Balance the
  2473. * siblings too.
  2474. */
  2475. if (sched_core_enabled(rq)) {
  2476. const struct cpumask *smt_mask = cpu_smt_mask(cpu_of(rq));
  2477. int scpu;
  2478. for_each_cpu_andnot(scpu, smt_mask, cpumask_of(cpu_of(rq))) {
  2479. struct rq *srq = cpu_rq(scpu);
  2480. struct task_struct *sprev = srq->curr;
  2481. WARN_ON_ONCE(__rq_lockp(rq) != __rq_lockp(srq));
  2482. update_rq_clock(srq);
  2483. balance_one(srq, sprev);
  2484. }
  2485. }
  2486. #endif
  2487. rq_repin_lock(rq, rf);
  2488. return ret;
  2489. }
  2490. static void process_ddsp_deferred_locals(struct rq *rq)
  2491. {
  2492. struct task_struct *p;
  2493. lockdep_assert_rq_held(rq);
  2494. /*
  2495. * Now that @rq can be unlocked, execute the deferred enqueueing of
  2496. * tasks directly dispatched to the local DSQs of other CPUs. See
  2497. * direct_dispatch(). Keep popping from the head instead of using
  2498. * list_for_each_entry_safe() as dispatch_local_dsq() may unlock @rq
  2499. * temporarily.
  2500. */
  2501. while ((p = list_first_entry_or_null(&rq->scx.ddsp_deferred_locals,
  2502. struct task_struct, scx.dsq_list.node))) {
  2503. struct scx_dispatch_q *dsq;
  2504. list_del_init(&p->scx.dsq_list.node);
  2505. dsq = find_dsq_for_dispatch(rq, p->scx.ddsp_dsq_id, p);
  2506. if (!WARN_ON_ONCE(dsq->id != SCX_DSQ_LOCAL))
  2507. dispatch_to_local_dsq(rq, dsq, p, p->scx.ddsp_enq_flags);
  2508. }
  2509. }
  2510. static void set_next_task_scx(struct rq *rq, struct task_struct *p, bool first)
  2511. {
  2512. if (p->scx.flags & SCX_TASK_QUEUED) {
  2513. /*
  2514. * Core-sched might decide to execute @p before it is
  2515. * dispatched. Call ops_dequeue() to notify the BPF scheduler.
  2516. */
  2517. ops_dequeue(p, SCX_DEQ_CORE_SCHED_EXEC);
  2518. dispatch_dequeue(rq, p);
  2519. }
  2520. p->se.exec_start = rq_clock_task(rq);
  2521. /* see dequeue_task_scx() on why we skip when !QUEUED */
  2522. if (SCX_HAS_OP(running) && (p->scx.flags & SCX_TASK_QUEUED))
  2523. SCX_CALL_OP_TASK(SCX_KF_REST, running, p);
  2524. clr_task_runnable(p, true);
  2525. /*
  2526. * @p is getting newly scheduled or got kicked after someone updated its
  2527. * slice. Refresh whether tick can be stopped. See scx_can_stop_tick().
  2528. */
  2529. if ((p->scx.slice == SCX_SLICE_INF) !=
  2530. (bool)(rq->scx.flags & SCX_RQ_CAN_STOP_TICK)) {
  2531. if (p->scx.slice == SCX_SLICE_INF)
  2532. rq->scx.flags |= SCX_RQ_CAN_STOP_TICK;
  2533. else
  2534. rq->scx.flags &= ~SCX_RQ_CAN_STOP_TICK;
  2535. sched_update_tick_dependency(rq);
  2536. /*
  2537. * For now, let's refresh the load_avgs just when transitioning
  2538. * in and out of nohz. In the future, we might want to add a
  2539. * mechanism which calls the following periodically on
  2540. * tick-stopped CPUs.
  2541. */
  2542. update_other_load_avgs(rq);
  2543. }
  2544. }
  2545. static enum scx_cpu_preempt_reason
  2546. preempt_reason_from_class(const struct sched_class *class)
  2547. {
  2548. #ifdef CONFIG_SMP
  2549. if (class == &stop_sched_class)
  2550. return SCX_CPU_PREEMPT_STOP;
  2551. #endif
  2552. if (class == &dl_sched_class)
  2553. return SCX_CPU_PREEMPT_DL;
  2554. if (class == &rt_sched_class)
  2555. return SCX_CPU_PREEMPT_RT;
  2556. return SCX_CPU_PREEMPT_UNKNOWN;
  2557. }
  2558. static void switch_class(struct rq *rq, struct task_struct *next)
  2559. {
  2560. const struct sched_class *next_class = next->sched_class;
  2561. #ifdef CONFIG_SMP
  2562. /*
  2563. * Pairs with the smp_load_acquire() issued by a CPU in
  2564. * kick_cpus_irq_workfn() who is waiting for this CPU to perform a
  2565. * resched.
  2566. */
  2567. smp_store_release(&rq->scx.pnt_seq, rq->scx.pnt_seq + 1);
  2568. #endif
  2569. if (!static_branch_unlikely(&scx_ops_cpu_preempt))
  2570. return;
  2571. /*
  2572. * The callback is conceptually meant to convey that the CPU is no
  2573. * longer under the control of SCX. Therefore, don't invoke the callback
  2574. * if the next class is below SCX (in which case the BPF scheduler has
  2575. * actively decided not to schedule any tasks on the CPU).
  2576. */
  2577. if (sched_class_above(&ext_sched_class, next_class))
  2578. return;
  2579. /*
  2580. * At this point we know that SCX was preempted by a higher priority
  2581. * sched_class, so invoke the ->cpu_release() callback if we have not
  2582. * done so already. We only send the callback once between SCX being
  2583. * preempted, and it regaining control of the CPU.
  2584. *
  2585. * ->cpu_release() complements ->cpu_acquire(), which is emitted the
  2586. * next time that balance_scx() is invoked.
  2587. */
  2588. if (!rq->scx.cpu_released) {
  2589. if (SCX_HAS_OP(cpu_release)) {
  2590. struct scx_cpu_release_args args = {
  2591. .reason = preempt_reason_from_class(next_class),
  2592. .task = next,
  2593. };
  2594. SCX_CALL_OP(SCX_KF_CPU_RELEASE,
  2595. cpu_release, cpu_of(rq), &args);
  2596. }
  2597. rq->scx.cpu_released = true;
  2598. }
  2599. }
  2600. static void put_prev_task_scx(struct rq *rq, struct task_struct *p,
  2601. struct task_struct *next)
  2602. {
  2603. update_curr_scx(rq);
  2604. /* see dequeue_task_scx() on why we skip when !QUEUED */
  2605. if (SCX_HAS_OP(stopping) && (p->scx.flags & SCX_TASK_QUEUED))
  2606. SCX_CALL_OP_TASK(SCX_KF_REST, stopping, p, true);
  2607. if (p->scx.flags & SCX_TASK_QUEUED) {
  2608. set_task_runnable(rq, p);
  2609. /*
  2610. * If @p has slice left and is being put, @p is getting
  2611. * preempted by a higher priority scheduler class or core-sched
  2612. * forcing a different task. Leave it at the head of the local
  2613. * DSQ.
  2614. */
  2615. if (p->scx.slice && !scx_rq_bypassing(rq)) {
  2616. dispatch_enqueue(&rq->scx.local_dsq, p, SCX_ENQ_HEAD);
  2617. goto switch_class;
  2618. }
  2619. /*
  2620. * If @p is runnable but we're about to enter a lower
  2621. * sched_class, %SCX_OPS_ENQ_LAST must be set. Tell
  2622. * ops.enqueue() that @p is the only one available for this cpu,
  2623. * which should trigger an explicit follow-up scheduling event.
  2624. */
  2625. if (sched_class_above(&ext_sched_class, next->sched_class)) {
  2626. WARN_ON_ONCE(!static_branch_unlikely(&scx_ops_enq_last));
  2627. do_enqueue_task(rq, p, SCX_ENQ_LAST, -1);
  2628. } else {
  2629. do_enqueue_task(rq, p, 0, -1);
  2630. }
  2631. }
  2632. switch_class:
  2633. if (next && next->sched_class != &ext_sched_class)
  2634. switch_class(rq, next);
  2635. }
  2636. static struct task_struct *first_local_task(struct rq *rq)
  2637. {
  2638. return list_first_entry_or_null(&rq->scx.local_dsq.list,
  2639. struct task_struct, scx.dsq_list.node);
  2640. }
  2641. static struct task_struct *pick_task_scx(struct rq *rq)
  2642. {
  2643. struct task_struct *prev = rq->curr;
  2644. struct task_struct *p;
  2645. bool keep_prev = rq->scx.flags & SCX_RQ_BAL_KEEP;
  2646. bool kick_idle = false;
  2647. /*
  2648. * WORKAROUND:
  2649. *
  2650. * %SCX_RQ_BAL_KEEP should be set iff $prev is on SCX as it must just
  2651. * have gone through balance_scx(). Unfortunately, there currently is a
  2652. * bug where fair could say yes on balance() but no on pick_task(),
  2653. * which then ends up calling pick_task_scx() without preceding
  2654. * balance_scx().
  2655. *
  2656. * Keep running @prev if possible and avoid stalling from entering idle
  2657. * without balancing.
  2658. *
  2659. * Once fair is fixed, remove the workaround and trigger WARN_ON_ONCE()
  2660. * if pick_task_scx() is called without preceding balance_scx().
  2661. */
  2662. if (unlikely(rq->scx.flags & SCX_RQ_BAL_PENDING)) {
  2663. if (prev->scx.flags & SCX_TASK_QUEUED) {
  2664. keep_prev = true;
  2665. } else {
  2666. keep_prev = false;
  2667. kick_idle = true;
  2668. }
  2669. } else if (unlikely(keep_prev &&
  2670. prev->sched_class != &ext_sched_class)) {
  2671. /*
  2672. * Can happen while enabling as SCX_RQ_BAL_PENDING assertion is
  2673. * conditional on scx_enabled() and may have been skipped.
  2674. */
  2675. WARN_ON_ONCE(scx_ops_enable_state() == SCX_OPS_ENABLED);
  2676. keep_prev = false;
  2677. }
  2678. /*
  2679. * If balance_scx() is telling us to keep running @prev, replenish slice
  2680. * if necessary and keep running @prev. Otherwise, pop the first one
  2681. * from the local DSQ.
  2682. */
  2683. if (keep_prev) {
  2684. p = prev;
  2685. if (!p->scx.slice)
  2686. p->scx.slice = SCX_SLICE_DFL;
  2687. } else {
  2688. p = first_local_task(rq);
  2689. if (!p) {
  2690. if (kick_idle)
  2691. scx_bpf_kick_cpu(cpu_of(rq), SCX_KICK_IDLE);
  2692. return NULL;
  2693. }
  2694. if (unlikely(!p->scx.slice)) {
  2695. if (!scx_rq_bypassing(rq) && !scx_warned_zero_slice) {
  2696. printk_deferred(KERN_WARNING "sched_ext: %s[%d] has zero slice in %s()\n",
  2697. p->comm, p->pid, __func__);
  2698. scx_warned_zero_slice = true;
  2699. }
  2700. p->scx.slice = SCX_SLICE_DFL;
  2701. }
  2702. }
  2703. return p;
  2704. }
  2705. #ifdef CONFIG_SCHED_CORE
  2706. /**
  2707. * scx_prio_less - Task ordering for core-sched
  2708. * @a: task A
  2709. * @b: task B
  2710. *
  2711. * Core-sched is implemented as an additional scheduling layer on top of the
  2712. * usual sched_class'es and needs to find out the expected task ordering. For
  2713. * SCX, core-sched calls this function to interrogate the task ordering.
  2714. *
  2715. * Unless overridden by ops.core_sched_before(), @p->scx.core_sched_at is used
  2716. * to implement the default task ordering. The older the timestamp, the higher
  2717. * prority the task - the global FIFO ordering matching the default scheduling
  2718. * behavior.
  2719. *
  2720. * When ops.core_sched_before() is enabled, @p->scx.core_sched_at is used to
  2721. * implement FIFO ordering within each local DSQ. See pick_task_scx().
  2722. */
  2723. bool scx_prio_less(const struct task_struct *a, const struct task_struct *b,
  2724. bool in_fi)
  2725. {
  2726. /*
  2727. * The const qualifiers are dropped from task_struct pointers when
  2728. * calling ops.core_sched_before(). Accesses are controlled by the
  2729. * verifier.
  2730. */
  2731. if (SCX_HAS_OP(core_sched_before) && !scx_rq_bypassing(task_rq(a)))
  2732. return SCX_CALL_OP_2TASKS_RET(SCX_KF_REST, core_sched_before,
  2733. (struct task_struct *)a,
  2734. (struct task_struct *)b);
  2735. else
  2736. return time_after64(a->scx.core_sched_at, b->scx.core_sched_at);
  2737. }
  2738. #endif /* CONFIG_SCHED_CORE */
  2739. #ifdef CONFIG_SMP
  2740. static bool test_and_clear_cpu_idle(int cpu)
  2741. {
  2742. #ifdef CONFIG_SCHED_SMT
  2743. /*
  2744. * SMT mask should be cleared whether we can claim @cpu or not. The SMT
  2745. * cluster is not wholly idle either way. This also prevents
  2746. * scx_pick_idle_cpu() from getting caught in an infinite loop.
  2747. */
  2748. if (sched_smt_active()) {
  2749. const struct cpumask *smt = cpu_smt_mask(cpu);
  2750. /*
  2751. * If offline, @cpu is not its own sibling and
  2752. * scx_pick_idle_cpu() can get caught in an infinite loop as
  2753. * @cpu is never cleared from idle_masks.smt. Ensure that @cpu
  2754. * is eventually cleared.
  2755. */
  2756. if (cpumask_intersects(smt, idle_masks.smt))
  2757. cpumask_andnot(idle_masks.smt, idle_masks.smt, smt);
  2758. else if (cpumask_test_cpu(cpu, idle_masks.smt))
  2759. __cpumask_clear_cpu(cpu, idle_masks.smt);
  2760. }
  2761. #endif
  2762. return cpumask_test_and_clear_cpu(cpu, idle_masks.cpu);
  2763. }
  2764. static s32 scx_pick_idle_cpu(const struct cpumask *cpus_allowed, u64 flags)
  2765. {
  2766. int cpu;
  2767. retry:
  2768. if (sched_smt_active()) {
  2769. cpu = cpumask_any_and_distribute(idle_masks.smt, cpus_allowed);
  2770. if (cpu < nr_cpu_ids)
  2771. goto found;
  2772. if (flags & SCX_PICK_IDLE_CORE)
  2773. return -EBUSY;
  2774. }
  2775. cpu = cpumask_any_and_distribute(idle_masks.cpu, cpus_allowed);
  2776. if (cpu >= nr_cpu_ids)
  2777. return -EBUSY;
  2778. found:
  2779. if (test_and_clear_cpu_idle(cpu))
  2780. return cpu;
  2781. else
  2782. goto retry;
  2783. }
  2784. static s32 scx_select_cpu_dfl(struct task_struct *p, s32 prev_cpu,
  2785. u64 wake_flags, bool *found)
  2786. {
  2787. s32 cpu;
  2788. *found = false;
  2789. /*
  2790. * This is necessary to protect llc_cpus.
  2791. */
  2792. rcu_read_lock();
  2793. /*
  2794. * If WAKE_SYNC, the waker's local DSQ is empty, and the system is
  2795. * under utilized, wake up @p to the local DSQ of the waker. Checking
  2796. * only for an empty local DSQ is insufficient as it could give the
  2797. * wakee an unfair advantage when the system is oversaturated.
  2798. * Checking only for the presence of idle CPUs is also insufficient as
  2799. * the local DSQ of the waker could have tasks piled up on it even if
  2800. * there is an idle core elsewhere on the system.
  2801. */
  2802. cpu = smp_processor_id();
  2803. if ((wake_flags & SCX_WAKE_SYNC) &&
  2804. !cpumask_empty(idle_masks.cpu) && !(current->flags & PF_EXITING) &&
  2805. cpu_rq(cpu)->scx.local_dsq.nr == 0) {
  2806. if (cpumask_test_cpu(cpu, p->cpus_ptr))
  2807. goto cpu_found;
  2808. }
  2809. /*
  2810. * If CPU has SMT, any wholly idle CPU is likely a better pick than
  2811. * partially idle @prev_cpu.
  2812. */
  2813. if (sched_smt_active()) {
  2814. if (cpumask_test_cpu(prev_cpu, idle_masks.smt) &&
  2815. test_and_clear_cpu_idle(prev_cpu)) {
  2816. cpu = prev_cpu;
  2817. goto cpu_found;
  2818. }
  2819. cpu = scx_pick_idle_cpu(p->cpus_ptr, SCX_PICK_IDLE_CORE);
  2820. if (cpu >= 0)
  2821. goto cpu_found;
  2822. }
  2823. if (test_and_clear_cpu_idle(prev_cpu)) {
  2824. cpu = prev_cpu;
  2825. goto cpu_found;
  2826. }
  2827. cpu = scx_pick_idle_cpu(p->cpus_ptr, 0);
  2828. if (cpu >= 0)
  2829. goto cpu_found;
  2830. rcu_read_unlock();
  2831. return prev_cpu;
  2832. cpu_found:
  2833. rcu_read_unlock();
  2834. *found = true;
  2835. return cpu;
  2836. }
  2837. static int select_task_rq_scx(struct task_struct *p, int prev_cpu, int wake_flags)
  2838. {
  2839. /*
  2840. * sched_exec() calls with %WF_EXEC when @p is about to exec(2) as it
  2841. * can be a good migration opportunity with low cache and memory
  2842. * footprint. Returning a CPU different than @prev_cpu triggers
  2843. * immediate rq migration. However, for SCX, as the current rq
  2844. * association doesn't dictate where the task is going to run, this
  2845. * doesn't fit well. If necessary, we can later add a dedicated method
  2846. * which can decide to preempt self to force it through the regular
  2847. * scheduling path.
  2848. */
  2849. if (unlikely(wake_flags & WF_EXEC))
  2850. return prev_cpu;
  2851. if (SCX_HAS_OP(select_cpu) && !scx_rq_bypassing(task_rq(p))) {
  2852. s32 cpu;
  2853. struct task_struct **ddsp_taskp;
  2854. ddsp_taskp = this_cpu_ptr(&direct_dispatch_task);
  2855. WARN_ON_ONCE(*ddsp_taskp);
  2856. *ddsp_taskp = p;
  2857. cpu = SCX_CALL_OP_TASK_RET(SCX_KF_ENQUEUE | SCX_KF_SELECT_CPU,
  2858. select_cpu, p, prev_cpu, wake_flags);
  2859. *ddsp_taskp = NULL;
  2860. if (ops_cpu_valid(cpu, "from ops.select_cpu()"))
  2861. return cpu;
  2862. else
  2863. return prev_cpu;
  2864. } else {
  2865. bool found;
  2866. s32 cpu;
  2867. cpu = scx_select_cpu_dfl(p, prev_cpu, wake_flags, &found);
  2868. if (found) {
  2869. p->scx.slice = SCX_SLICE_DFL;
  2870. p->scx.ddsp_dsq_id = SCX_DSQ_LOCAL;
  2871. }
  2872. return cpu;
  2873. }
  2874. }
  2875. static void task_woken_scx(struct rq *rq, struct task_struct *p)
  2876. {
  2877. run_deferred(rq);
  2878. }
  2879. static void set_cpus_allowed_scx(struct task_struct *p,
  2880. struct affinity_context *ac)
  2881. {
  2882. set_cpus_allowed_common(p, ac);
  2883. /*
  2884. * The effective cpumask is stored in @p->cpus_ptr which may temporarily
  2885. * differ from the configured one in @p->cpus_mask. Always tell the bpf
  2886. * scheduler the effective one.
  2887. *
  2888. * Fine-grained memory write control is enforced by BPF making the const
  2889. * designation pointless. Cast it away when calling the operation.
  2890. */
  2891. if (SCX_HAS_OP(set_cpumask))
  2892. SCX_CALL_OP_TASK(SCX_KF_REST, set_cpumask, p,
  2893. (struct cpumask *)p->cpus_ptr);
  2894. }
  2895. static void reset_idle_masks(void)
  2896. {
  2897. /*
  2898. * Consider all online cpus idle. Should converge to the actual state
  2899. * quickly.
  2900. */
  2901. cpumask_copy(idle_masks.cpu, cpu_online_mask);
  2902. cpumask_copy(idle_masks.smt, cpu_online_mask);
  2903. }
  2904. static void update_builtin_idle(int cpu, bool idle)
  2905. {
  2906. if (idle)
  2907. cpumask_set_cpu(cpu, idle_masks.cpu);
  2908. else
  2909. cpumask_clear_cpu(cpu, idle_masks.cpu);
  2910. #ifdef CONFIG_SCHED_SMT
  2911. if (sched_smt_active()) {
  2912. const struct cpumask *smt = cpu_smt_mask(cpu);
  2913. if (idle) {
  2914. /*
  2915. * idle_masks.smt handling is racy but that's fine as
  2916. * it's only for optimization and self-correcting.
  2917. */
  2918. for_each_cpu(cpu, smt) {
  2919. if (!cpumask_test_cpu(cpu, idle_masks.cpu))
  2920. return;
  2921. }
  2922. cpumask_or(idle_masks.smt, idle_masks.smt, smt);
  2923. } else {
  2924. cpumask_andnot(idle_masks.smt, idle_masks.smt, smt);
  2925. }
  2926. }
  2927. #endif
  2928. }
  2929. /*
  2930. * Update the idle state of a CPU to @idle.
  2931. *
  2932. * If @do_notify is true, ops.update_idle() is invoked to notify the scx
  2933. * scheduler of an actual idle state transition (idle to busy or vice
  2934. * versa). If @do_notify is false, only the idle state in the idle masks is
  2935. * refreshed without invoking ops.update_idle().
  2936. *
  2937. * This distinction is necessary, because an idle CPU can be "reserved" and
  2938. * awakened via scx_bpf_pick_idle_cpu() + scx_bpf_kick_cpu(), marking it as
  2939. * busy even if no tasks are dispatched. In this case, the CPU may return
  2940. * to idle without a true state transition. Refreshing the idle masks
  2941. * without invoking ops.update_idle() ensures accurate idle state tracking
  2942. * while avoiding unnecessary updates and maintaining balanced state
  2943. * transitions.
  2944. */
  2945. void __scx_update_idle(struct rq *rq, bool idle, bool do_notify)
  2946. {
  2947. int cpu = cpu_of(rq);
  2948. lockdep_assert_rq_held(rq);
  2949. /*
  2950. * Trigger ops.update_idle() only when transitioning from a task to
  2951. * the idle thread and vice versa.
  2952. *
  2953. * Idle transitions are indicated by do_notify being set to true,
  2954. * managed by put_prev_task_idle()/set_next_task_idle().
  2955. */
  2956. if (SCX_HAS_OP(update_idle) && do_notify && !scx_rq_bypassing(rq))
  2957. SCX_CALL_OP(SCX_KF_REST, update_idle, cpu_of(rq), idle);
  2958. /*
  2959. * Update the idle masks:
  2960. * - for real idle transitions (do_notify == true)
  2961. * - for idle-to-idle transitions (indicated by the previous task
  2962. * being the idle thread, managed by pick_task_idle())
  2963. *
  2964. * Skip updating idle masks if the previous task is not the idle
  2965. * thread, since set_next_task_idle() has already handled it when
  2966. * transitioning from a task to the idle thread (calling this
  2967. * function with do_notify == true).
  2968. *
  2969. * In this way we can avoid updating the idle masks twice,
  2970. * unnecessarily.
  2971. */
  2972. if (static_branch_likely(&scx_builtin_idle_enabled))
  2973. if (do_notify || is_idle_task(rq->curr))
  2974. update_builtin_idle(cpu, idle);
  2975. }
  2976. static void handle_hotplug(struct rq *rq, bool online)
  2977. {
  2978. int cpu = cpu_of(rq);
  2979. atomic_long_inc(&scx_hotplug_seq);
  2980. if (online && SCX_HAS_OP(cpu_online))
  2981. SCX_CALL_OP(SCX_KF_UNLOCKED, cpu_online, cpu);
  2982. else if (!online && SCX_HAS_OP(cpu_offline))
  2983. SCX_CALL_OP(SCX_KF_UNLOCKED, cpu_offline, cpu);
  2984. else
  2985. scx_ops_exit(SCX_ECODE_ACT_RESTART | SCX_ECODE_RSN_HOTPLUG,
  2986. "cpu %d going %s, exiting scheduler", cpu,
  2987. online ? "online" : "offline");
  2988. }
  2989. void scx_rq_activate(struct rq *rq)
  2990. {
  2991. handle_hotplug(rq, true);
  2992. }
  2993. void scx_rq_deactivate(struct rq *rq)
  2994. {
  2995. handle_hotplug(rq, false);
  2996. }
  2997. static void rq_online_scx(struct rq *rq)
  2998. {
  2999. rq->scx.flags |= SCX_RQ_ONLINE;
  3000. }
  3001. static void rq_offline_scx(struct rq *rq)
  3002. {
  3003. rq->scx.flags &= ~SCX_RQ_ONLINE;
  3004. }
  3005. #else /* CONFIG_SMP */
  3006. static bool test_and_clear_cpu_idle(int cpu) { return false; }
  3007. static s32 scx_pick_idle_cpu(const struct cpumask *cpus_allowed, u64 flags) { return -EBUSY; }
  3008. static void reset_idle_masks(void) {}
  3009. #endif /* CONFIG_SMP */
  3010. static bool check_rq_for_timeouts(struct rq *rq)
  3011. {
  3012. struct task_struct *p;
  3013. struct rq_flags rf;
  3014. bool timed_out = false;
  3015. rq_lock_irqsave(rq, &rf);
  3016. list_for_each_entry(p, &rq->scx.runnable_list, scx.runnable_node) {
  3017. unsigned long last_runnable = p->scx.runnable_at;
  3018. if (unlikely(time_after(jiffies,
  3019. last_runnable + scx_watchdog_timeout))) {
  3020. u32 dur_ms = jiffies_to_msecs(jiffies - last_runnable);
  3021. scx_ops_error_kind(SCX_EXIT_ERROR_STALL,
  3022. "%s[%d] failed to run for %u.%03us",
  3023. p->comm, p->pid,
  3024. dur_ms / 1000, dur_ms % 1000);
  3025. timed_out = true;
  3026. break;
  3027. }
  3028. }
  3029. rq_unlock_irqrestore(rq, &rf);
  3030. return timed_out;
  3031. }
  3032. static void scx_watchdog_workfn(struct work_struct *work)
  3033. {
  3034. int cpu;
  3035. WRITE_ONCE(scx_watchdog_timestamp, jiffies);
  3036. for_each_online_cpu(cpu) {
  3037. if (unlikely(check_rq_for_timeouts(cpu_rq(cpu))))
  3038. break;
  3039. cond_resched();
  3040. }
  3041. queue_delayed_work(system_unbound_wq, to_delayed_work(work),
  3042. scx_watchdog_timeout / 2);
  3043. }
  3044. void scx_tick(struct rq *rq)
  3045. {
  3046. unsigned long last_check;
  3047. if (!scx_enabled())
  3048. return;
  3049. last_check = READ_ONCE(scx_watchdog_timestamp);
  3050. if (unlikely(time_after(jiffies,
  3051. last_check + READ_ONCE(scx_watchdog_timeout)))) {
  3052. u32 dur_ms = jiffies_to_msecs(jiffies - last_check);
  3053. scx_ops_error_kind(SCX_EXIT_ERROR_STALL,
  3054. "watchdog failed to check in for %u.%03us",
  3055. dur_ms / 1000, dur_ms % 1000);
  3056. }
  3057. update_other_load_avgs(rq);
  3058. }
  3059. static void task_tick_scx(struct rq *rq, struct task_struct *curr, int queued)
  3060. {
  3061. update_curr_scx(rq);
  3062. /*
  3063. * While disabling, always resched and refresh core-sched timestamp as
  3064. * we can't trust the slice management or ops.core_sched_before().
  3065. */
  3066. if (scx_rq_bypassing(rq)) {
  3067. curr->scx.slice = 0;
  3068. touch_core_sched(rq, curr);
  3069. } else if (SCX_HAS_OP(tick)) {
  3070. SCX_CALL_OP_TASK(SCX_KF_REST, tick, curr);
  3071. }
  3072. if (!curr->scx.slice)
  3073. resched_curr(rq);
  3074. }
  3075. #ifdef CONFIG_EXT_GROUP_SCHED
  3076. static struct cgroup *tg_cgrp(struct task_group *tg)
  3077. {
  3078. /*
  3079. * If CGROUP_SCHED is disabled, @tg is NULL. If @tg is an autogroup,
  3080. * @tg->css.cgroup is NULL. In both cases, @tg can be treated as the
  3081. * root cgroup.
  3082. */
  3083. if (tg && tg->css.cgroup)
  3084. return tg->css.cgroup;
  3085. else
  3086. return &cgrp_dfl_root.cgrp;
  3087. }
  3088. #define SCX_INIT_TASK_ARGS_CGROUP(tg) .cgroup = tg_cgrp(tg),
  3089. #else /* CONFIG_EXT_GROUP_SCHED */
  3090. #define SCX_INIT_TASK_ARGS_CGROUP(tg)
  3091. #endif /* CONFIG_EXT_GROUP_SCHED */
  3092. static enum scx_task_state scx_get_task_state(const struct task_struct *p)
  3093. {
  3094. return (p->scx.flags & SCX_TASK_STATE_MASK) >> SCX_TASK_STATE_SHIFT;
  3095. }
  3096. static void scx_set_task_state(struct task_struct *p, enum scx_task_state state)
  3097. {
  3098. enum scx_task_state prev_state = scx_get_task_state(p);
  3099. bool warn = false;
  3100. BUILD_BUG_ON(SCX_TASK_NR_STATES > (1 << SCX_TASK_STATE_BITS));
  3101. switch (state) {
  3102. case SCX_TASK_NONE:
  3103. break;
  3104. case SCX_TASK_INIT:
  3105. warn = prev_state != SCX_TASK_NONE;
  3106. break;
  3107. case SCX_TASK_READY:
  3108. warn = prev_state == SCX_TASK_NONE;
  3109. break;
  3110. case SCX_TASK_ENABLED:
  3111. warn = prev_state != SCX_TASK_READY;
  3112. break;
  3113. default:
  3114. warn = true;
  3115. return;
  3116. }
  3117. WARN_ONCE(warn, "sched_ext: Invalid task state transition %d -> %d for %s[%d]",
  3118. prev_state, state, p->comm, p->pid);
  3119. p->scx.flags &= ~SCX_TASK_STATE_MASK;
  3120. p->scx.flags |= state << SCX_TASK_STATE_SHIFT;
  3121. }
  3122. static int scx_ops_init_task(struct task_struct *p, struct task_group *tg, bool fork)
  3123. {
  3124. int ret;
  3125. p->scx.disallow = false;
  3126. if (SCX_HAS_OP(init_task)) {
  3127. struct scx_init_task_args args = {
  3128. SCX_INIT_TASK_ARGS_CGROUP(tg)
  3129. .fork = fork,
  3130. };
  3131. ret = SCX_CALL_OP_RET(SCX_KF_UNLOCKED, init_task, p, &args);
  3132. if (unlikely(ret)) {
  3133. ret = ops_sanitize_err("init_task", ret);
  3134. return ret;
  3135. }
  3136. }
  3137. scx_set_task_state(p, SCX_TASK_INIT);
  3138. if (p->scx.disallow) {
  3139. if (!fork) {
  3140. struct rq *rq;
  3141. struct rq_flags rf;
  3142. rq = task_rq_lock(p, &rf);
  3143. /*
  3144. * We're in the load path and @p->policy will be applied
  3145. * right after. Reverting @p->policy here and rejecting
  3146. * %SCHED_EXT transitions from scx_check_setscheduler()
  3147. * guarantees that if ops.init_task() sets @p->disallow,
  3148. * @p can never be in SCX.
  3149. */
  3150. if (p->policy == SCHED_EXT) {
  3151. p->policy = SCHED_NORMAL;
  3152. atomic_long_inc(&scx_nr_rejected);
  3153. }
  3154. task_rq_unlock(rq, p, &rf);
  3155. } else if (p->policy == SCHED_EXT) {
  3156. scx_ops_error("ops.init_task() set task->scx.disallow for %s[%d] during fork",
  3157. p->comm, p->pid);
  3158. }
  3159. }
  3160. p->scx.flags |= SCX_TASK_RESET_RUNNABLE_AT;
  3161. return 0;
  3162. }
  3163. static void scx_ops_enable_task(struct task_struct *p)
  3164. {
  3165. u32 weight;
  3166. lockdep_assert_rq_held(task_rq(p));
  3167. /*
  3168. * Set the weight before calling ops.enable() so that the scheduler
  3169. * doesn't see a stale value if they inspect the task struct.
  3170. */
  3171. if (task_has_idle_policy(p))
  3172. weight = WEIGHT_IDLEPRIO;
  3173. else
  3174. weight = sched_prio_to_weight[p->static_prio - MAX_RT_PRIO];
  3175. p->scx.weight = sched_weight_to_cgroup(weight);
  3176. if (SCX_HAS_OP(enable))
  3177. SCX_CALL_OP_TASK(SCX_KF_REST, enable, p);
  3178. scx_set_task_state(p, SCX_TASK_ENABLED);
  3179. if (SCX_HAS_OP(set_weight))
  3180. SCX_CALL_OP_TASK(SCX_KF_REST, set_weight, p, p->scx.weight);
  3181. }
  3182. static void scx_ops_disable_task(struct task_struct *p)
  3183. {
  3184. lockdep_assert_rq_held(task_rq(p));
  3185. WARN_ON_ONCE(scx_get_task_state(p) != SCX_TASK_ENABLED);
  3186. if (SCX_HAS_OP(disable))
  3187. SCX_CALL_OP_TASK(SCX_KF_REST, disable, p);
  3188. scx_set_task_state(p, SCX_TASK_READY);
  3189. }
  3190. static void scx_ops_exit_task(struct task_struct *p)
  3191. {
  3192. struct scx_exit_task_args args = {
  3193. .cancelled = false,
  3194. };
  3195. lockdep_assert_rq_held(task_rq(p));
  3196. switch (scx_get_task_state(p)) {
  3197. case SCX_TASK_NONE:
  3198. return;
  3199. case SCX_TASK_INIT:
  3200. args.cancelled = true;
  3201. break;
  3202. case SCX_TASK_READY:
  3203. break;
  3204. case SCX_TASK_ENABLED:
  3205. scx_ops_disable_task(p);
  3206. break;
  3207. default:
  3208. WARN_ON_ONCE(true);
  3209. return;
  3210. }
  3211. if (SCX_HAS_OP(exit_task))
  3212. SCX_CALL_OP_TASK(SCX_KF_REST, exit_task, p, &args);
  3213. scx_set_task_state(p, SCX_TASK_NONE);
  3214. }
  3215. void init_scx_entity(struct sched_ext_entity *scx)
  3216. {
  3217. memset(scx, 0, sizeof(*scx));
  3218. INIT_LIST_HEAD(&scx->dsq_list.node);
  3219. RB_CLEAR_NODE(&scx->dsq_priq);
  3220. scx->sticky_cpu = -1;
  3221. scx->holding_cpu = -1;
  3222. INIT_LIST_HEAD(&scx->runnable_node);
  3223. scx->runnable_at = jiffies;
  3224. scx->ddsp_dsq_id = SCX_DSQ_INVALID;
  3225. scx->slice = SCX_SLICE_DFL;
  3226. }
  3227. void scx_pre_fork(struct task_struct *p)
  3228. {
  3229. /*
  3230. * BPF scheduler enable/disable paths want to be able to iterate and
  3231. * update all tasks which can become complex when racing forks. As
  3232. * enable/disable are very cold paths, let's use a percpu_rwsem to
  3233. * exclude forks.
  3234. */
  3235. percpu_down_read(&scx_fork_rwsem);
  3236. }
  3237. int scx_fork(struct task_struct *p)
  3238. {
  3239. percpu_rwsem_assert_held(&scx_fork_rwsem);
  3240. if (scx_ops_init_task_enabled)
  3241. return scx_ops_init_task(p, task_group(p), true);
  3242. else
  3243. return 0;
  3244. }
  3245. void scx_post_fork(struct task_struct *p)
  3246. {
  3247. if (scx_ops_init_task_enabled) {
  3248. scx_set_task_state(p, SCX_TASK_READY);
  3249. /*
  3250. * Enable the task immediately if it's running on sched_ext.
  3251. * Otherwise, it'll be enabled in switching_to_scx() if and
  3252. * when it's ever configured to run with a SCHED_EXT policy.
  3253. */
  3254. if (p->sched_class == &ext_sched_class) {
  3255. struct rq_flags rf;
  3256. struct rq *rq;
  3257. rq = task_rq_lock(p, &rf);
  3258. scx_ops_enable_task(p);
  3259. task_rq_unlock(rq, p, &rf);
  3260. }
  3261. }
  3262. spin_lock_irq(&scx_tasks_lock);
  3263. list_add_tail(&p->scx.tasks_node, &scx_tasks);
  3264. spin_unlock_irq(&scx_tasks_lock);
  3265. percpu_up_read(&scx_fork_rwsem);
  3266. }
  3267. void scx_cancel_fork(struct task_struct *p)
  3268. {
  3269. if (scx_enabled()) {
  3270. struct rq *rq;
  3271. struct rq_flags rf;
  3272. rq = task_rq_lock(p, &rf);
  3273. WARN_ON_ONCE(scx_get_task_state(p) >= SCX_TASK_READY);
  3274. scx_ops_exit_task(p);
  3275. task_rq_unlock(rq, p, &rf);
  3276. }
  3277. percpu_up_read(&scx_fork_rwsem);
  3278. }
  3279. void sched_ext_free(struct task_struct *p)
  3280. {
  3281. unsigned long flags;
  3282. spin_lock_irqsave(&scx_tasks_lock, flags);
  3283. list_del_init(&p->scx.tasks_node);
  3284. spin_unlock_irqrestore(&scx_tasks_lock, flags);
  3285. /*
  3286. * @p is off scx_tasks and wholly ours. scx_ops_enable()'s READY ->
  3287. * ENABLED transitions can't race us. Disable ops for @p.
  3288. */
  3289. if (scx_get_task_state(p) != SCX_TASK_NONE) {
  3290. struct rq_flags rf;
  3291. struct rq *rq;
  3292. rq = task_rq_lock(p, &rf);
  3293. scx_ops_exit_task(p);
  3294. task_rq_unlock(rq, p, &rf);
  3295. }
  3296. }
  3297. static void reweight_task_scx(struct rq *rq, struct task_struct *p,
  3298. const struct load_weight *lw)
  3299. {
  3300. lockdep_assert_rq_held(task_rq(p));
  3301. p->scx.weight = sched_weight_to_cgroup(scale_load_down(lw->weight));
  3302. if (SCX_HAS_OP(set_weight))
  3303. SCX_CALL_OP_TASK(SCX_KF_REST, set_weight, p, p->scx.weight);
  3304. }
  3305. static void prio_changed_scx(struct rq *rq, struct task_struct *p, int oldprio)
  3306. {
  3307. }
  3308. static void switching_to_scx(struct rq *rq, struct task_struct *p)
  3309. {
  3310. scx_ops_enable_task(p);
  3311. /*
  3312. * set_cpus_allowed_scx() is not called while @p is associated with a
  3313. * different scheduler class. Keep the BPF scheduler up-to-date.
  3314. */
  3315. if (SCX_HAS_OP(set_cpumask))
  3316. SCX_CALL_OP_TASK(SCX_KF_REST, set_cpumask, p,
  3317. (struct cpumask *)p->cpus_ptr);
  3318. }
  3319. static void switched_from_scx(struct rq *rq, struct task_struct *p)
  3320. {
  3321. scx_ops_disable_task(p);
  3322. }
  3323. static void wakeup_preempt_scx(struct rq *rq, struct task_struct *p,int wake_flags) {}
  3324. static void switched_to_scx(struct rq *rq, struct task_struct *p) {}
  3325. int scx_check_setscheduler(struct task_struct *p, int policy)
  3326. {
  3327. lockdep_assert_rq_held(task_rq(p));
  3328. /* if disallow, reject transitioning into SCX */
  3329. if (scx_enabled() && READ_ONCE(p->scx.disallow) &&
  3330. p->policy != policy && policy == SCHED_EXT)
  3331. return -EACCES;
  3332. return 0;
  3333. }
  3334. #ifdef CONFIG_NO_HZ_FULL
  3335. bool scx_can_stop_tick(struct rq *rq)
  3336. {
  3337. struct task_struct *p = rq->curr;
  3338. if (scx_rq_bypassing(rq))
  3339. return false;
  3340. if (p->sched_class != &ext_sched_class)
  3341. return true;
  3342. /*
  3343. * @rq can dispatch from different DSQs, so we can't tell whether it
  3344. * needs the tick or not by looking at nr_running. Allow stopping ticks
  3345. * iff the BPF scheduler indicated so. See set_next_task_scx().
  3346. */
  3347. return rq->scx.flags & SCX_RQ_CAN_STOP_TICK;
  3348. }
  3349. #endif
  3350. #ifdef CONFIG_EXT_GROUP_SCHED
  3351. DEFINE_STATIC_PERCPU_RWSEM(scx_cgroup_rwsem);
  3352. static bool scx_cgroup_enabled;
  3353. static bool cgroup_warned_missing_weight;
  3354. static bool cgroup_warned_missing_idle;
  3355. static void scx_cgroup_warn_missing_weight(struct task_group *tg)
  3356. {
  3357. if (scx_ops_enable_state() == SCX_OPS_DISABLED ||
  3358. cgroup_warned_missing_weight)
  3359. return;
  3360. if ((scx_ops.flags & SCX_OPS_HAS_CGROUP_WEIGHT) || !tg->css.parent)
  3361. return;
  3362. pr_warn("sched_ext: \"%s\" does not implement cgroup cpu.weight\n",
  3363. scx_ops.name);
  3364. cgroup_warned_missing_weight = true;
  3365. }
  3366. static void scx_cgroup_warn_missing_idle(struct task_group *tg)
  3367. {
  3368. if (!scx_cgroup_enabled || cgroup_warned_missing_idle)
  3369. return;
  3370. if (!tg->idle)
  3371. return;
  3372. pr_warn("sched_ext: \"%s\" does not implement cgroup cpu.idle\n",
  3373. scx_ops.name);
  3374. cgroup_warned_missing_idle = true;
  3375. }
  3376. void scx_tg_init(struct task_group *tg)
  3377. {
  3378. tg->scx_weight = CGROUP_WEIGHT_DFL;
  3379. }
  3380. int scx_tg_online(struct task_group *tg)
  3381. {
  3382. int ret = 0;
  3383. WARN_ON_ONCE(tg->scx_flags & (SCX_TG_ONLINE | SCX_TG_INITED));
  3384. percpu_down_read(&scx_cgroup_rwsem);
  3385. scx_cgroup_warn_missing_weight(tg);
  3386. if (scx_cgroup_enabled) {
  3387. if (SCX_HAS_OP(cgroup_init)) {
  3388. struct scx_cgroup_init_args args =
  3389. { .weight = tg->scx_weight };
  3390. ret = SCX_CALL_OP_RET(SCX_KF_UNLOCKED, cgroup_init,
  3391. tg->css.cgroup, &args);
  3392. if (ret)
  3393. ret = ops_sanitize_err("cgroup_init", ret);
  3394. }
  3395. if (ret == 0)
  3396. tg->scx_flags |= SCX_TG_ONLINE | SCX_TG_INITED;
  3397. } else {
  3398. tg->scx_flags |= SCX_TG_ONLINE;
  3399. }
  3400. percpu_up_read(&scx_cgroup_rwsem);
  3401. return ret;
  3402. }
  3403. void scx_tg_offline(struct task_group *tg)
  3404. {
  3405. WARN_ON_ONCE(!(tg->scx_flags & SCX_TG_ONLINE));
  3406. percpu_down_read(&scx_cgroup_rwsem);
  3407. if (SCX_HAS_OP(cgroup_exit) && (tg->scx_flags & SCX_TG_INITED))
  3408. SCX_CALL_OP(SCX_KF_UNLOCKED, cgroup_exit, tg->css.cgroup);
  3409. tg->scx_flags &= ~(SCX_TG_ONLINE | SCX_TG_INITED);
  3410. percpu_up_read(&scx_cgroup_rwsem);
  3411. }
  3412. int scx_cgroup_can_attach(struct cgroup_taskset *tset)
  3413. {
  3414. struct cgroup_subsys_state *css;
  3415. struct task_struct *p;
  3416. int ret;
  3417. /* released in scx_finish/cancel_attach() */
  3418. percpu_down_read(&scx_cgroup_rwsem);
  3419. if (!scx_cgroup_enabled)
  3420. return 0;
  3421. cgroup_taskset_for_each(p, css, tset) {
  3422. struct cgroup *from = tg_cgrp(task_group(p));
  3423. struct cgroup *to = tg_cgrp(css_tg(css));
  3424. WARN_ON_ONCE(p->scx.cgrp_moving_from);
  3425. /*
  3426. * sched_move_task() omits identity migrations. Let's match the
  3427. * behavior so that ops.cgroup_prep_move() and ops.cgroup_move()
  3428. * always match one-to-one.
  3429. */
  3430. if (from == to)
  3431. continue;
  3432. if (SCX_HAS_OP(cgroup_prep_move)) {
  3433. ret = SCX_CALL_OP_RET(SCX_KF_UNLOCKED, cgroup_prep_move,
  3434. p, from, css->cgroup);
  3435. if (ret)
  3436. goto err;
  3437. }
  3438. p->scx.cgrp_moving_from = from;
  3439. }
  3440. return 0;
  3441. err:
  3442. cgroup_taskset_for_each(p, css, tset) {
  3443. if (SCX_HAS_OP(cgroup_cancel_move) && p->scx.cgrp_moving_from)
  3444. SCX_CALL_OP(SCX_KF_UNLOCKED, cgroup_cancel_move, p,
  3445. p->scx.cgrp_moving_from, css->cgroup);
  3446. p->scx.cgrp_moving_from = NULL;
  3447. }
  3448. percpu_up_read(&scx_cgroup_rwsem);
  3449. return ops_sanitize_err("cgroup_prep_move", ret);
  3450. }
  3451. void scx_cgroup_move_task(struct task_struct *p)
  3452. {
  3453. if (!scx_cgroup_enabled)
  3454. return;
  3455. /*
  3456. * @p must have ops.cgroup_prep_move() called on it and thus
  3457. * cgrp_moving_from set.
  3458. */
  3459. if (SCX_HAS_OP(cgroup_move) && !WARN_ON_ONCE(!p->scx.cgrp_moving_from))
  3460. SCX_CALL_OP_TASK(SCX_KF_UNLOCKED, cgroup_move, p,
  3461. p->scx.cgrp_moving_from, tg_cgrp(task_group(p)));
  3462. p->scx.cgrp_moving_from = NULL;
  3463. }
  3464. void scx_cgroup_finish_attach(void)
  3465. {
  3466. percpu_up_read(&scx_cgroup_rwsem);
  3467. }
  3468. void scx_cgroup_cancel_attach(struct cgroup_taskset *tset)
  3469. {
  3470. struct cgroup_subsys_state *css;
  3471. struct task_struct *p;
  3472. if (!scx_cgroup_enabled)
  3473. goto out_unlock;
  3474. cgroup_taskset_for_each(p, css, tset) {
  3475. if (SCX_HAS_OP(cgroup_cancel_move) && p->scx.cgrp_moving_from)
  3476. SCX_CALL_OP(SCX_KF_UNLOCKED, cgroup_cancel_move, p,
  3477. p->scx.cgrp_moving_from, css->cgroup);
  3478. p->scx.cgrp_moving_from = NULL;
  3479. }
  3480. out_unlock:
  3481. percpu_up_read(&scx_cgroup_rwsem);
  3482. }
  3483. void scx_group_set_weight(struct task_group *tg, unsigned long weight)
  3484. {
  3485. percpu_down_read(&scx_cgroup_rwsem);
  3486. if (scx_cgroup_enabled && SCX_HAS_OP(cgroup_set_weight) &&
  3487. tg->scx_weight != weight)
  3488. SCX_CALL_OP(SCX_KF_UNLOCKED, cgroup_set_weight,
  3489. tg_cgrp(tg), weight);
  3490. tg->scx_weight = weight;
  3491. percpu_up_read(&scx_cgroup_rwsem);
  3492. }
  3493. void scx_group_set_idle(struct task_group *tg, bool idle)
  3494. {
  3495. percpu_down_read(&scx_cgroup_rwsem);
  3496. scx_cgroup_warn_missing_idle(tg);
  3497. percpu_up_read(&scx_cgroup_rwsem);
  3498. }
  3499. static void scx_cgroup_lock(void)
  3500. {
  3501. percpu_down_write(&scx_cgroup_rwsem);
  3502. }
  3503. static void scx_cgroup_unlock(void)
  3504. {
  3505. percpu_up_write(&scx_cgroup_rwsem);
  3506. }
  3507. #else /* CONFIG_EXT_GROUP_SCHED */
  3508. static inline void scx_cgroup_lock(void) {}
  3509. static inline void scx_cgroup_unlock(void) {}
  3510. #endif /* CONFIG_EXT_GROUP_SCHED */
  3511. /*
  3512. * Omitted operations:
  3513. *
  3514. * - wakeup_preempt: NOOP as it isn't useful in the wakeup path because the task
  3515. * isn't tied to the CPU at that point. Preemption is implemented by resetting
  3516. * the victim task's slice to 0 and triggering reschedule on the target CPU.
  3517. *
  3518. * - migrate_task_rq: Unnecessary as task to cpu mapping is transient.
  3519. *
  3520. * - task_fork/dead: We need fork/dead notifications for all tasks regardless of
  3521. * their current sched_class. Call them directly from sched core instead.
  3522. */
  3523. DEFINE_SCHED_CLASS(ext) = {
  3524. .enqueue_task = enqueue_task_scx,
  3525. .dequeue_task = dequeue_task_scx,
  3526. .yield_task = yield_task_scx,
  3527. .yield_to_task = yield_to_task_scx,
  3528. .wakeup_preempt = wakeup_preempt_scx,
  3529. .balance = balance_scx,
  3530. .pick_task = pick_task_scx,
  3531. .put_prev_task = put_prev_task_scx,
  3532. .set_next_task = set_next_task_scx,
  3533. #ifdef CONFIG_SMP
  3534. .select_task_rq = select_task_rq_scx,
  3535. .task_woken = task_woken_scx,
  3536. .set_cpus_allowed = set_cpus_allowed_scx,
  3537. .rq_online = rq_online_scx,
  3538. .rq_offline = rq_offline_scx,
  3539. #endif
  3540. .task_tick = task_tick_scx,
  3541. .switching_to = switching_to_scx,
  3542. .switched_from = switched_from_scx,
  3543. .switched_to = switched_to_scx,
  3544. .reweight_task = reweight_task_scx,
  3545. .prio_changed = prio_changed_scx,
  3546. .update_curr = update_curr_scx,
  3547. #ifdef CONFIG_UCLAMP_TASK
  3548. .uclamp_enabled = 1,
  3549. #endif
  3550. };
  3551. static void init_dsq(struct scx_dispatch_q *dsq, u64 dsq_id)
  3552. {
  3553. memset(dsq, 0, sizeof(*dsq));
  3554. raw_spin_lock_init(&dsq->lock);
  3555. INIT_LIST_HEAD(&dsq->list);
  3556. dsq->id = dsq_id;
  3557. }
  3558. static struct scx_dispatch_q *create_dsq(u64 dsq_id, int node)
  3559. {
  3560. struct scx_dispatch_q *dsq;
  3561. int ret;
  3562. if (dsq_id & SCX_DSQ_FLAG_BUILTIN)
  3563. return ERR_PTR(-EINVAL);
  3564. dsq = kmalloc_node(sizeof(*dsq), GFP_KERNEL, node);
  3565. if (!dsq)
  3566. return ERR_PTR(-ENOMEM);
  3567. init_dsq(dsq, dsq_id);
  3568. ret = rhashtable_lookup_insert_fast(&dsq_hash, &dsq->hash_node,
  3569. dsq_hash_params);
  3570. if (ret) {
  3571. kfree(dsq);
  3572. return ERR_PTR(ret);
  3573. }
  3574. return dsq;
  3575. }
  3576. static void free_dsq_irq_workfn(struct irq_work *irq_work)
  3577. {
  3578. struct llist_node *to_free = llist_del_all(&dsqs_to_free);
  3579. struct scx_dispatch_q *dsq, *tmp_dsq;
  3580. llist_for_each_entry_safe(dsq, tmp_dsq, to_free, free_node)
  3581. kfree_rcu(dsq, rcu);
  3582. }
  3583. static DEFINE_IRQ_WORK(free_dsq_irq_work, free_dsq_irq_workfn);
  3584. static void destroy_dsq(u64 dsq_id)
  3585. {
  3586. struct scx_dispatch_q *dsq;
  3587. unsigned long flags;
  3588. rcu_read_lock();
  3589. dsq = find_user_dsq(dsq_id);
  3590. if (!dsq)
  3591. goto out_unlock_rcu;
  3592. raw_spin_lock_irqsave(&dsq->lock, flags);
  3593. if (dsq->nr) {
  3594. scx_ops_error("attempting to destroy in-use dsq 0x%016llx (nr=%u)",
  3595. dsq->id, dsq->nr);
  3596. goto out_unlock_dsq;
  3597. }
  3598. if (rhashtable_remove_fast(&dsq_hash, &dsq->hash_node, dsq_hash_params))
  3599. goto out_unlock_dsq;
  3600. /*
  3601. * Mark dead by invalidating ->id to prevent dispatch_enqueue() from
  3602. * queueing more tasks. As this function can be called from anywhere,
  3603. * freeing is bounced through an irq work to avoid nesting RCU
  3604. * operations inside scheduler locks.
  3605. */
  3606. dsq->id = SCX_DSQ_INVALID;
  3607. llist_add(&dsq->free_node, &dsqs_to_free);
  3608. irq_work_queue(&free_dsq_irq_work);
  3609. out_unlock_dsq:
  3610. raw_spin_unlock_irqrestore(&dsq->lock, flags);
  3611. out_unlock_rcu:
  3612. rcu_read_unlock();
  3613. }
  3614. #ifdef CONFIG_EXT_GROUP_SCHED
  3615. static void scx_cgroup_exit(void)
  3616. {
  3617. struct cgroup_subsys_state *css;
  3618. percpu_rwsem_assert_held(&scx_cgroup_rwsem);
  3619. scx_cgroup_enabled = false;
  3620. /*
  3621. * scx_tg_on/offline() are excluded through scx_cgroup_rwsem. If we walk
  3622. * cgroups and exit all the inited ones, all online cgroups are exited.
  3623. */
  3624. rcu_read_lock();
  3625. css_for_each_descendant_post(css, &root_task_group.css) {
  3626. struct task_group *tg = css_tg(css);
  3627. if (!(tg->scx_flags & SCX_TG_INITED))
  3628. continue;
  3629. tg->scx_flags &= ~SCX_TG_INITED;
  3630. if (!scx_ops.cgroup_exit)
  3631. continue;
  3632. if (WARN_ON_ONCE(!css_tryget(css)))
  3633. continue;
  3634. rcu_read_unlock();
  3635. SCX_CALL_OP(SCX_KF_UNLOCKED, cgroup_exit, css->cgroup);
  3636. rcu_read_lock();
  3637. css_put(css);
  3638. }
  3639. rcu_read_unlock();
  3640. }
  3641. static int scx_cgroup_init(void)
  3642. {
  3643. struct cgroup_subsys_state *css;
  3644. int ret;
  3645. percpu_rwsem_assert_held(&scx_cgroup_rwsem);
  3646. cgroup_warned_missing_weight = false;
  3647. cgroup_warned_missing_idle = false;
  3648. /*
  3649. * scx_tg_on/offline() are excluded thorugh scx_cgroup_rwsem. If we walk
  3650. * cgroups and init, all online cgroups are initialized.
  3651. */
  3652. rcu_read_lock();
  3653. css_for_each_descendant_pre(css, &root_task_group.css) {
  3654. struct task_group *tg = css_tg(css);
  3655. struct scx_cgroup_init_args args = { .weight = tg->scx_weight };
  3656. scx_cgroup_warn_missing_weight(tg);
  3657. scx_cgroup_warn_missing_idle(tg);
  3658. if ((tg->scx_flags &
  3659. (SCX_TG_ONLINE | SCX_TG_INITED)) != SCX_TG_ONLINE)
  3660. continue;
  3661. if (!scx_ops.cgroup_init) {
  3662. tg->scx_flags |= SCX_TG_INITED;
  3663. continue;
  3664. }
  3665. if (WARN_ON_ONCE(!css_tryget(css)))
  3666. continue;
  3667. rcu_read_unlock();
  3668. ret = SCX_CALL_OP_RET(SCX_KF_UNLOCKED, cgroup_init,
  3669. css->cgroup, &args);
  3670. if (ret) {
  3671. css_put(css);
  3672. scx_ops_error("ops.cgroup_init() failed (%d)", ret);
  3673. return ret;
  3674. }
  3675. tg->scx_flags |= SCX_TG_INITED;
  3676. rcu_read_lock();
  3677. css_put(css);
  3678. }
  3679. rcu_read_unlock();
  3680. WARN_ON_ONCE(scx_cgroup_enabled);
  3681. scx_cgroup_enabled = true;
  3682. return 0;
  3683. }
  3684. #else
  3685. static void scx_cgroup_exit(void) {}
  3686. static int scx_cgroup_init(void) { return 0; }
  3687. #endif
  3688. /********************************************************************************
  3689. * Sysfs interface and ops enable/disable.
  3690. */
  3691. #define SCX_ATTR(_name) \
  3692. static struct kobj_attribute scx_attr_##_name = { \
  3693. .attr = { .name = __stringify(_name), .mode = 0444 }, \
  3694. .show = scx_attr_##_name##_show, \
  3695. }
  3696. static ssize_t scx_attr_state_show(struct kobject *kobj,
  3697. struct kobj_attribute *ka, char *buf)
  3698. {
  3699. return sysfs_emit(buf, "%s\n",
  3700. scx_ops_enable_state_str[scx_ops_enable_state()]);
  3701. }
  3702. SCX_ATTR(state);
  3703. static ssize_t scx_attr_switch_all_show(struct kobject *kobj,
  3704. struct kobj_attribute *ka, char *buf)
  3705. {
  3706. return sysfs_emit(buf, "%d\n", READ_ONCE(scx_switching_all));
  3707. }
  3708. SCX_ATTR(switch_all);
  3709. static ssize_t scx_attr_nr_rejected_show(struct kobject *kobj,
  3710. struct kobj_attribute *ka, char *buf)
  3711. {
  3712. return sysfs_emit(buf, "%ld\n", atomic_long_read(&scx_nr_rejected));
  3713. }
  3714. SCX_ATTR(nr_rejected);
  3715. static ssize_t scx_attr_hotplug_seq_show(struct kobject *kobj,
  3716. struct kobj_attribute *ka, char *buf)
  3717. {
  3718. return sysfs_emit(buf, "%ld\n", atomic_long_read(&scx_hotplug_seq));
  3719. }
  3720. SCX_ATTR(hotplug_seq);
  3721. static ssize_t scx_attr_enable_seq_show(struct kobject *kobj,
  3722. struct kobj_attribute *ka, char *buf)
  3723. {
  3724. return sysfs_emit(buf, "%ld\n", atomic_long_read(&scx_enable_seq));
  3725. }
  3726. SCX_ATTR(enable_seq);
  3727. static struct attribute *scx_global_attrs[] = {
  3728. &scx_attr_state.attr,
  3729. &scx_attr_switch_all.attr,
  3730. &scx_attr_nr_rejected.attr,
  3731. &scx_attr_hotplug_seq.attr,
  3732. &scx_attr_enable_seq.attr,
  3733. NULL,
  3734. };
  3735. static const struct attribute_group scx_global_attr_group = {
  3736. .attrs = scx_global_attrs,
  3737. };
  3738. static void scx_kobj_release(struct kobject *kobj)
  3739. {
  3740. kfree(kobj);
  3741. }
  3742. static ssize_t scx_attr_ops_show(struct kobject *kobj,
  3743. struct kobj_attribute *ka, char *buf)
  3744. {
  3745. return sysfs_emit(buf, "%s\n", scx_ops.name);
  3746. }
  3747. SCX_ATTR(ops);
  3748. static struct attribute *scx_sched_attrs[] = {
  3749. &scx_attr_ops.attr,
  3750. NULL,
  3751. };
  3752. ATTRIBUTE_GROUPS(scx_sched);
  3753. static const struct kobj_type scx_ktype = {
  3754. .release = scx_kobj_release,
  3755. .sysfs_ops = &kobj_sysfs_ops,
  3756. .default_groups = scx_sched_groups,
  3757. };
  3758. static int scx_uevent(const struct kobject *kobj, struct kobj_uevent_env *env)
  3759. {
  3760. return add_uevent_var(env, "SCXOPS=%s", scx_ops.name);
  3761. }
  3762. static const struct kset_uevent_ops scx_uevent_ops = {
  3763. .uevent = scx_uevent,
  3764. };
  3765. /*
  3766. * Used by sched_fork() and __setscheduler_prio() to pick the matching
  3767. * sched_class. dl/rt are already handled.
  3768. */
  3769. bool task_should_scx(int policy)
  3770. {
  3771. if (!scx_enabled() ||
  3772. unlikely(scx_ops_enable_state() == SCX_OPS_DISABLING))
  3773. return false;
  3774. if (READ_ONCE(scx_switching_all))
  3775. return true;
  3776. return policy == SCHED_EXT;
  3777. }
  3778. /**
  3779. * scx_ops_bypass - [Un]bypass scx_ops and guarantee forward progress
  3780. *
  3781. * Bypassing guarantees that all runnable tasks make forward progress without
  3782. * trusting the BPF scheduler. We can't grab any mutexes or rwsems as they might
  3783. * be held by tasks that the BPF scheduler is forgetting to run, which
  3784. * unfortunately also excludes toggling the static branches.
  3785. *
  3786. * Let's work around by overriding a couple ops and modifying behaviors based on
  3787. * the DISABLING state and then cycling the queued tasks through dequeue/enqueue
  3788. * to force global FIFO scheduling.
  3789. *
  3790. * - ops.select_cpu() is ignored and the default select_cpu() is used.
  3791. *
  3792. * - ops.enqueue() is ignored and tasks are queued in simple global FIFO order.
  3793. * %SCX_OPS_ENQ_LAST is also ignored.
  3794. *
  3795. * - ops.dispatch() is ignored.
  3796. *
  3797. * - balance_scx() does not set %SCX_RQ_BAL_KEEP on non-zero slice as slice
  3798. * can't be trusted. Whenever a tick triggers, the running task is rotated to
  3799. * the tail of the queue with core_sched_at touched.
  3800. *
  3801. * - pick_next_task() suppresses zero slice warning.
  3802. *
  3803. * - scx_bpf_kick_cpu() is disabled to avoid irq_work malfunction during PM
  3804. * operations.
  3805. *
  3806. * - scx_prio_less() reverts to the default core_sched_at order.
  3807. */
  3808. static void scx_ops_bypass(bool bypass)
  3809. {
  3810. int cpu;
  3811. unsigned long flags;
  3812. raw_spin_lock_irqsave(&__scx_ops_bypass_lock, flags);
  3813. if (bypass) {
  3814. scx_ops_bypass_depth++;
  3815. WARN_ON_ONCE(scx_ops_bypass_depth <= 0);
  3816. if (scx_ops_bypass_depth != 1)
  3817. goto unlock;
  3818. } else {
  3819. scx_ops_bypass_depth--;
  3820. WARN_ON_ONCE(scx_ops_bypass_depth < 0);
  3821. if (scx_ops_bypass_depth != 0)
  3822. goto unlock;
  3823. }
  3824. /*
  3825. * No task property is changing. We just need to make sure all currently
  3826. * queued tasks are re-queued according to the new scx_rq_bypassing()
  3827. * state. As an optimization, walk each rq's runnable_list instead of
  3828. * the scx_tasks list.
  3829. *
  3830. * This function can't trust the scheduler and thus can't use
  3831. * cpus_read_lock(). Walk all possible CPUs instead of online.
  3832. */
  3833. for_each_possible_cpu(cpu) {
  3834. struct rq *rq = cpu_rq(cpu);
  3835. struct task_struct *p, *n;
  3836. raw_spin_rq_lock(rq);
  3837. if (bypass) {
  3838. WARN_ON_ONCE(rq->scx.flags & SCX_RQ_BYPASSING);
  3839. rq->scx.flags |= SCX_RQ_BYPASSING;
  3840. } else {
  3841. WARN_ON_ONCE(!(rq->scx.flags & SCX_RQ_BYPASSING));
  3842. rq->scx.flags &= ~SCX_RQ_BYPASSING;
  3843. }
  3844. /*
  3845. * We need to guarantee that no tasks are on the BPF scheduler
  3846. * while bypassing. Either we see enabled or the enable path
  3847. * sees scx_rq_bypassing() before moving tasks to SCX.
  3848. */
  3849. if (!scx_enabled()) {
  3850. raw_spin_rq_unlock(rq);
  3851. continue;
  3852. }
  3853. /*
  3854. * The use of list_for_each_entry_safe_reverse() is required
  3855. * because each task is going to be removed from and added back
  3856. * to the runnable_list during iteration. Because they're added
  3857. * to the tail of the list, safe reverse iteration can still
  3858. * visit all nodes.
  3859. */
  3860. list_for_each_entry_safe_reverse(p, n, &rq->scx.runnable_list,
  3861. scx.runnable_node) {
  3862. struct sched_enq_and_set_ctx ctx;
  3863. /* cycling deq/enq is enough, see the function comment */
  3864. sched_deq_and_put_task(p, DEQUEUE_SAVE | DEQUEUE_MOVE, &ctx);
  3865. sched_enq_and_set_task(&ctx);
  3866. }
  3867. /* resched to restore ticks and idle state */
  3868. if (cpu_online(cpu) || cpu == smp_processor_id())
  3869. resched_curr(rq);
  3870. raw_spin_rq_unlock(rq);
  3871. }
  3872. unlock:
  3873. raw_spin_unlock_irqrestore(&__scx_ops_bypass_lock, flags);
  3874. }
  3875. static void free_exit_info(struct scx_exit_info *ei)
  3876. {
  3877. kvfree(ei->dump);
  3878. kfree(ei->msg);
  3879. kfree(ei->bt);
  3880. kfree(ei);
  3881. }
  3882. static struct scx_exit_info *alloc_exit_info(size_t exit_dump_len)
  3883. {
  3884. struct scx_exit_info *ei;
  3885. ei = kzalloc(sizeof(*ei), GFP_KERNEL);
  3886. if (!ei)
  3887. return NULL;
  3888. ei->bt = kcalloc(SCX_EXIT_BT_LEN, sizeof(ei->bt[0]), GFP_KERNEL);
  3889. ei->msg = kzalloc(SCX_EXIT_MSG_LEN, GFP_KERNEL);
  3890. ei->dump = kvzalloc(exit_dump_len, GFP_KERNEL);
  3891. if (!ei->bt || !ei->msg || !ei->dump) {
  3892. free_exit_info(ei);
  3893. return NULL;
  3894. }
  3895. return ei;
  3896. }
  3897. static const char *scx_exit_reason(enum scx_exit_kind kind)
  3898. {
  3899. switch (kind) {
  3900. case SCX_EXIT_UNREG:
  3901. return "unregistered from user space";
  3902. case SCX_EXIT_UNREG_BPF:
  3903. return "unregistered from BPF";
  3904. case SCX_EXIT_UNREG_KERN:
  3905. return "unregistered from the main kernel";
  3906. case SCX_EXIT_SYSRQ:
  3907. return "disabled by sysrq-S";
  3908. case SCX_EXIT_ERROR:
  3909. return "runtime error";
  3910. case SCX_EXIT_ERROR_BPF:
  3911. return "scx_bpf_error";
  3912. case SCX_EXIT_ERROR_STALL:
  3913. return "runnable task stall";
  3914. default:
  3915. return "<UNKNOWN>";
  3916. }
  3917. }
  3918. static void scx_ops_disable_workfn(struct kthread_work *work)
  3919. {
  3920. struct scx_exit_info *ei = scx_exit_info;
  3921. struct scx_task_iter sti;
  3922. struct task_struct *p;
  3923. struct rhashtable_iter rht_iter;
  3924. struct scx_dispatch_q *dsq;
  3925. int i, kind;
  3926. kind = atomic_read(&scx_exit_kind);
  3927. while (true) {
  3928. /*
  3929. * NONE indicates that a new scx_ops has been registered since
  3930. * disable was scheduled - don't kill the new ops. DONE
  3931. * indicates that the ops has already been disabled.
  3932. */
  3933. if (kind == SCX_EXIT_NONE || kind == SCX_EXIT_DONE)
  3934. return;
  3935. if (atomic_try_cmpxchg(&scx_exit_kind, &kind, SCX_EXIT_DONE))
  3936. break;
  3937. }
  3938. ei->kind = kind;
  3939. ei->reason = scx_exit_reason(ei->kind);
  3940. /* guarantee forward progress by bypassing scx_ops */
  3941. scx_ops_bypass(true);
  3942. switch (scx_ops_set_enable_state(SCX_OPS_DISABLING)) {
  3943. case SCX_OPS_DISABLING:
  3944. WARN_ONCE(true, "sched_ext: duplicate disabling instance?");
  3945. break;
  3946. case SCX_OPS_DISABLED:
  3947. pr_warn("sched_ext: ops error detected without ops (%s)\n",
  3948. scx_exit_info->msg);
  3949. WARN_ON_ONCE(scx_ops_set_enable_state(SCX_OPS_DISABLED) !=
  3950. SCX_OPS_DISABLING);
  3951. goto done;
  3952. default:
  3953. break;
  3954. }
  3955. /*
  3956. * Here, every runnable task is guaranteed to make forward progress and
  3957. * we can safely use blocking synchronization constructs. Actually
  3958. * disable ops.
  3959. */
  3960. mutex_lock(&scx_ops_enable_mutex);
  3961. static_branch_disable(&__scx_switched_all);
  3962. WRITE_ONCE(scx_switching_all, false);
  3963. /*
  3964. * Shut down cgroup support before tasks so that the cgroup attach path
  3965. * doesn't race against scx_ops_exit_task().
  3966. */
  3967. scx_cgroup_lock();
  3968. scx_cgroup_exit();
  3969. scx_cgroup_unlock();
  3970. /*
  3971. * The BPF scheduler is going away. All tasks including %TASK_DEAD ones
  3972. * must be switched out and exited synchronously.
  3973. */
  3974. percpu_down_write(&scx_fork_rwsem);
  3975. scx_ops_init_task_enabled = false;
  3976. scx_task_iter_start(&sti);
  3977. while ((p = scx_task_iter_next_locked(&sti))) {
  3978. const struct sched_class *old_class = p->sched_class;
  3979. const struct sched_class *new_class =
  3980. __setscheduler_class(p->policy, p->prio);
  3981. struct sched_enq_and_set_ctx ctx;
  3982. if (old_class != new_class && p->se.sched_delayed)
  3983. dequeue_task(task_rq(p), p, DEQUEUE_SLEEP | DEQUEUE_DELAYED);
  3984. sched_deq_and_put_task(p, DEQUEUE_SAVE | DEQUEUE_MOVE, &ctx);
  3985. p->sched_class = new_class;
  3986. check_class_changing(task_rq(p), p, old_class);
  3987. sched_enq_and_set_task(&ctx);
  3988. check_class_changed(task_rq(p), p, old_class, p->prio);
  3989. scx_ops_exit_task(p);
  3990. }
  3991. scx_task_iter_stop(&sti);
  3992. percpu_up_write(&scx_fork_rwsem);
  3993. /* no task is on scx, turn off all the switches and flush in-progress calls */
  3994. static_branch_disable(&__scx_ops_enabled);
  3995. for (i = SCX_OPI_BEGIN; i < SCX_OPI_END; i++)
  3996. static_branch_disable(&scx_has_op[i]);
  3997. static_branch_disable(&scx_ops_enq_last);
  3998. static_branch_disable(&scx_ops_enq_exiting);
  3999. static_branch_disable(&scx_ops_cpu_preempt);
  4000. static_branch_disable(&scx_builtin_idle_enabled);
  4001. synchronize_rcu();
  4002. if (ei->kind >= SCX_EXIT_ERROR) {
  4003. pr_err("sched_ext: BPF scheduler \"%s\" disabled (%s)\n",
  4004. scx_ops.name, ei->reason);
  4005. if (ei->msg[0] != '\0')
  4006. pr_err("sched_ext: %s: %s\n", scx_ops.name, ei->msg);
  4007. #ifdef CONFIG_STACKTRACE
  4008. stack_trace_print(ei->bt, ei->bt_len, 2);
  4009. #endif
  4010. } else {
  4011. pr_info("sched_ext: BPF scheduler \"%s\" disabled (%s)\n",
  4012. scx_ops.name, ei->reason);
  4013. }
  4014. if (scx_ops.exit)
  4015. SCX_CALL_OP(SCX_KF_UNLOCKED, exit, ei);
  4016. cancel_delayed_work_sync(&scx_watchdog_work);
  4017. /*
  4018. * Delete the kobject from the hierarchy eagerly in addition to just
  4019. * dropping a reference. Otherwise, if the object is deleted
  4020. * asynchronously, sysfs could observe an object of the same name still
  4021. * in the hierarchy when another scheduler is loaded.
  4022. */
  4023. kobject_del(scx_root_kobj);
  4024. kobject_put(scx_root_kobj);
  4025. scx_root_kobj = NULL;
  4026. memset(&scx_ops, 0, sizeof(scx_ops));
  4027. rhashtable_walk_enter(&dsq_hash, &rht_iter);
  4028. do {
  4029. rhashtable_walk_start(&rht_iter);
  4030. while ((dsq = rhashtable_walk_next(&rht_iter)) && !IS_ERR(dsq))
  4031. destroy_dsq(dsq->id);
  4032. rhashtable_walk_stop(&rht_iter);
  4033. } while (dsq == ERR_PTR(-EAGAIN));
  4034. rhashtable_walk_exit(&rht_iter);
  4035. free_percpu(scx_dsp_ctx);
  4036. scx_dsp_ctx = NULL;
  4037. scx_dsp_max_batch = 0;
  4038. free_exit_info(scx_exit_info);
  4039. scx_exit_info = NULL;
  4040. mutex_unlock(&scx_ops_enable_mutex);
  4041. WARN_ON_ONCE(scx_ops_set_enable_state(SCX_OPS_DISABLED) !=
  4042. SCX_OPS_DISABLING);
  4043. done:
  4044. scx_ops_bypass(false);
  4045. }
  4046. static DEFINE_KTHREAD_WORK(scx_ops_disable_work, scx_ops_disable_workfn);
  4047. static void schedule_scx_ops_disable_work(void)
  4048. {
  4049. struct kthread_worker *helper = READ_ONCE(scx_ops_helper);
  4050. /*
  4051. * We may be called spuriously before the first bpf_sched_ext_reg(). If
  4052. * scx_ops_helper isn't set up yet, there's nothing to do.
  4053. */
  4054. if (helper)
  4055. kthread_queue_work(helper, &scx_ops_disable_work);
  4056. }
  4057. static void scx_ops_disable(enum scx_exit_kind kind)
  4058. {
  4059. int none = SCX_EXIT_NONE;
  4060. if (WARN_ON_ONCE(kind == SCX_EXIT_NONE || kind == SCX_EXIT_DONE))
  4061. kind = SCX_EXIT_ERROR;
  4062. atomic_try_cmpxchg(&scx_exit_kind, &none, kind);
  4063. schedule_scx_ops_disable_work();
  4064. }
  4065. static void dump_newline(struct seq_buf *s)
  4066. {
  4067. trace_sched_ext_dump("");
  4068. /* @s may be zero sized and seq_buf triggers WARN if so */
  4069. if (s->size)
  4070. seq_buf_putc(s, '\n');
  4071. }
  4072. static __printf(2, 3) void dump_line(struct seq_buf *s, const char *fmt, ...)
  4073. {
  4074. va_list args;
  4075. #ifdef CONFIG_TRACEPOINTS
  4076. if (trace_sched_ext_dump_enabled()) {
  4077. /* protected by scx_dump_state()::dump_lock */
  4078. static char line_buf[SCX_EXIT_MSG_LEN];
  4079. va_start(args, fmt);
  4080. vscnprintf(line_buf, sizeof(line_buf), fmt, args);
  4081. va_end(args);
  4082. trace_sched_ext_dump(line_buf);
  4083. }
  4084. #endif
  4085. /* @s may be zero sized and seq_buf triggers WARN if so */
  4086. if (s->size) {
  4087. va_start(args, fmt);
  4088. seq_buf_vprintf(s, fmt, args);
  4089. va_end(args);
  4090. seq_buf_putc(s, '\n');
  4091. }
  4092. }
  4093. static void dump_stack_trace(struct seq_buf *s, const char *prefix,
  4094. const unsigned long *bt, unsigned int len)
  4095. {
  4096. unsigned int i;
  4097. for (i = 0; i < len; i++)
  4098. dump_line(s, "%s%pS", prefix, (void *)bt[i]);
  4099. }
  4100. static void ops_dump_init(struct seq_buf *s, const char *prefix)
  4101. {
  4102. struct scx_dump_data *dd = &scx_dump_data;
  4103. lockdep_assert_irqs_disabled();
  4104. dd->cpu = smp_processor_id(); /* allow scx_bpf_dump() */
  4105. dd->first = true;
  4106. dd->cursor = 0;
  4107. dd->s = s;
  4108. dd->prefix = prefix;
  4109. }
  4110. static void ops_dump_flush(void)
  4111. {
  4112. struct scx_dump_data *dd = &scx_dump_data;
  4113. char *line = dd->buf.line;
  4114. if (!dd->cursor)
  4115. return;
  4116. /*
  4117. * There's something to flush and this is the first line. Insert a blank
  4118. * line to distinguish ops dump.
  4119. */
  4120. if (dd->first) {
  4121. dump_newline(dd->s);
  4122. dd->first = false;
  4123. }
  4124. /*
  4125. * There may be multiple lines in $line. Scan and emit each line
  4126. * separately.
  4127. */
  4128. while (true) {
  4129. char *end = line;
  4130. char c;
  4131. while (*end != '\n' && *end != '\0')
  4132. end++;
  4133. /*
  4134. * If $line overflowed, it may not have newline at the end.
  4135. * Always emit with a newline.
  4136. */
  4137. c = *end;
  4138. *end = '\0';
  4139. dump_line(dd->s, "%s%s", dd->prefix, line);
  4140. if (c == '\0')
  4141. break;
  4142. /* move to the next line */
  4143. end++;
  4144. if (*end == '\0')
  4145. break;
  4146. line = end;
  4147. }
  4148. dd->cursor = 0;
  4149. }
  4150. static void ops_dump_exit(void)
  4151. {
  4152. ops_dump_flush();
  4153. scx_dump_data.cpu = -1;
  4154. }
  4155. static void scx_dump_task(struct seq_buf *s, struct scx_dump_ctx *dctx,
  4156. struct task_struct *p, char marker)
  4157. {
  4158. static unsigned long bt[SCX_EXIT_BT_LEN];
  4159. char dsq_id_buf[19] = "(n/a)";
  4160. unsigned long ops_state = atomic_long_read(&p->scx.ops_state);
  4161. unsigned int bt_len = 0;
  4162. if (p->scx.dsq)
  4163. scnprintf(dsq_id_buf, sizeof(dsq_id_buf), "0x%llx",
  4164. (unsigned long long)p->scx.dsq->id);
  4165. dump_newline(s);
  4166. dump_line(s, " %c%c %s[%d] %+ldms",
  4167. marker, task_state_to_char(p), p->comm, p->pid,
  4168. jiffies_delta_msecs(p->scx.runnable_at, dctx->at_jiffies));
  4169. dump_line(s, " scx_state/flags=%u/0x%x dsq_flags=0x%x ops_state/qseq=%lu/%lu",
  4170. scx_get_task_state(p), p->scx.flags & ~SCX_TASK_STATE_MASK,
  4171. p->scx.dsq_flags, ops_state & SCX_OPSS_STATE_MASK,
  4172. ops_state >> SCX_OPSS_QSEQ_SHIFT);
  4173. dump_line(s, " sticky/holding_cpu=%d/%d dsq_id=%s dsq_vtime=%llu",
  4174. p->scx.sticky_cpu, p->scx.holding_cpu, dsq_id_buf,
  4175. p->scx.dsq_vtime);
  4176. dump_line(s, " cpus=%*pb", cpumask_pr_args(p->cpus_ptr));
  4177. if (SCX_HAS_OP(dump_task)) {
  4178. ops_dump_init(s, " ");
  4179. SCX_CALL_OP(SCX_KF_REST, dump_task, dctx, p);
  4180. ops_dump_exit();
  4181. }
  4182. #ifdef CONFIG_STACKTRACE
  4183. bt_len = stack_trace_save_tsk(p, bt, SCX_EXIT_BT_LEN, 1);
  4184. #endif
  4185. if (bt_len) {
  4186. dump_newline(s);
  4187. dump_stack_trace(s, " ", bt, bt_len);
  4188. }
  4189. }
  4190. static void scx_dump_state(struct scx_exit_info *ei, size_t dump_len)
  4191. {
  4192. static DEFINE_SPINLOCK(dump_lock);
  4193. static const char trunc_marker[] = "\n\n~~~~ TRUNCATED ~~~~\n";
  4194. struct scx_dump_ctx dctx = {
  4195. .kind = ei->kind,
  4196. .exit_code = ei->exit_code,
  4197. .reason = ei->reason,
  4198. .at_ns = ktime_get_ns(),
  4199. .at_jiffies = jiffies,
  4200. };
  4201. struct seq_buf s;
  4202. unsigned long flags;
  4203. char *buf;
  4204. int cpu;
  4205. spin_lock_irqsave(&dump_lock, flags);
  4206. seq_buf_init(&s, ei->dump, dump_len);
  4207. if (ei->kind == SCX_EXIT_NONE) {
  4208. dump_line(&s, "Debug dump triggered by %s", ei->reason);
  4209. } else {
  4210. dump_line(&s, "%s[%d] triggered exit kind %d:",
  4211. current->comm, current->pid, ei->kind);
  4212. dump_line(&s, " %s (%s)", ei->reason, ei->msg);
  4213. dump_newline(&s);
  4214. dump_line(&s, "Backtrace:");
  4215. dump_stack_trace(&s, " ", ei->bt, ei->bt_len);
  4216. }
  4217. if (SCX_HAS_OP(dump)) {
  4218. ops_dump_init(&s, "");
  4219. SCX_CALL_OP(SCX_KF_UNLOCKED, dump, &dctx);
  4220. ops_dump_exit();
  4221. }
  4222. dump_newline(&s);
  4223. dump_line(&s, "CPU states");
  4224. dump_line(&s, "----------");
  4225. for_each_possible_cpu(cpu) {
  4226. struct rq *rq = cpu_rq(cpu);
  4227. struct rq_flags rf;
  4228. struct task_struct *p;
  4229. struct seq_buf ns;
  4230. size_t avail, used;
  4231. bool idle;
  4232. rq_lock(rq, &rf);
  4233. idle = list_empty(&rq->scx.runnable_list) &&
  4234. rq->curr->sched_class == &idle_sched_class;
  4235. if (idle && !SCX_HAS_OP(dump_cpu))
  4236. goto next;
  4237. /*
  4238. * We don't yet know whether ops.dump_cpu() will produce output
  4239. * and we may want to skip the default CPU dump if it doesn't.
  4240. * Use a nested seq_buf to generate the standard dump so that we
  4241. * can decide whether to commit later.
  4242. */
  4243. avail = seq_buf_get_buf(&s, &buf);
  4244. seq_buf_init(&ns, buf, avail);
  4245. dump_newline(&ns);
  4246. dump_line(&ns, "CPU %-4d: nr_run=%u flags=0x%x cpu_rel=%d ops_qseq=%lu pnt_seq=%lu",
  4247. cpu, rq->scx.nr_running, rq->scx.flags,
  4248. rq->scx.cpu_released, rq->scx.ops_qseq,
  4249. rq->scx.pnt_seq);
  4250. dump_line(&ns, " curr=%s[%d] class=%ps",
  4251. rq->curr->comm, rq->curr->pid,
  4252. rq->curr->sched_class);
  4253. if (!cpumask_empty(rq->scx.cpus_to_kick))
  4254. dump_line(&ns, " cpus_to_kick : %*pb",
  4255. cpumask_pr_args(rq->scx.cpus_to_kick));
  4256. if (!cpumask_empty(rq->scx.cpus_to_kick_if_idle))
  4257. dump_line(&ns, " idle_to_kick : %*pb",
  4258. cpumask_pr_args(rq->scx.cpus_to_kick_if_idle));
  4259. if (!cpumask_empty(rq->scx.cpus_to_preempt))
  4260. dump_line(&ns, " cpus_to_preempt: %*pb",
  4261. cpumask_pr_args(rq->scx.cpus_to_preempt));
  4262. if (!cpumask_empty(rq->scx.cpus_to_wait))
  4263. dump_line(&ns, " cpus_to_wait : %*pb",
  4264. cpumask_pr_args(rq->scx.cpus_to_wait));
  4265. used = seq_buf_used(&ns);
  4266. if (SCX_HAS_OP(dump_cpu)) {
  4267. ops_dump_init(&ns, " ");
  4268. SCX_CALL_OP(SCX_KF_REST, dump_cpu, &dctx, cpu, idle);
  4269. ops_dump_exit();
  4270. }
  4271. /*
  4272. * If idle && nothing generated by ops.dump_cpu(), there's
  4273. * nothing interesting. Skip.
  4274. */
  4275. if (idle && used == seq_buf_used(&ns))
  4276. goto next;
  4277. /*
  4278. * $s may already have overflowed when $ns was created. If so,
  4279. * calling commit on it will trigger BUG.
  4280. */
  4281. if (avail) {
  4282. seq_buf_commit(&s, seq_buf_used(&ns));
  4283. if (seq_buf_has_overflowed(&ns))
  4284. seq_buf_set_overflow(&s);
  4285. }
  4286. if (rq->curr->sched_class == &ext_sched_class)
  4287. scx_dump_task(&s, &dctx, rq->curr, '*');
  4288. list_for_each_entry(p, &rq->scx.runnable_list, scx.runnable_node)
  4289. scx_dump_task(&s, &dctx, p, ' ');
  4290. next:
  4291. rq_unlock(rq, &rf);
  4292. }
  4293. if (seq_buf_has_overflowed(&s) && dump_len >= sizeof(trunc_marker))
  4294. memcpy(ei->dump + dump_len - sizeof(trunc_marker),
  4295. trunc_marker, sizeof(trunc_marker));
  4296. spin_unlock_irqrestore(&dump_lock, flags);
  4297. }
  4298. static void scx_ops_error_irq_workfn(struct irq_work *irq_work)
  4299. {
  4300. struct scx_exit_info *ei = scx_exit_info;
  4301. if (ei->kind >= SCX_EXIT_ERROR)
  4302. scx_dump_state(ei, scx_ops.exit_dump_len);
  4303. schedule_scx_ops_disable_work();
  4304. }
  4305. static DEFINE_IRQ_WORK(scx_ops_error_irq_work, scx_ops_error_irq_workfn);
  4306. static __printf(3, 4) void scx_ops_exit_kind(enum scx_exit_kind kind,
  4307. s64 exit_code,
  4308. const char *fmt, ...)
  4309. {
  4310. struct scx_exit_info *ei = scx_exit_info;
  4311. int none = SCX_EXIT_NONE;
  4312. va_list args;
  4313. if (!atomic_try_cmpxchg(&scx_exit_kind, &none, kind))
  4314. return;
  4315. ei->exit_code = exit_code;
  4316. #ifdef CONFIG_STACKTRACE
  4317. if (kind >= SCX_EXIT_ERROR)
  4318. ei->bt_len = stack_trace_save(ei->bt, SCX_EXIT_BT_LEN, 1);
  4319. #endif
  4320. va_start(args, fmt);
  4321. vscnprintf(ei->msg, SCX_EXIT_MSG_LEN, fmt, args);
  4322. va_end(args);
  4323. /*
  4324. * Set ei->kind and ->reason for scx_dump_state(). They'll be set again
  4325. * in scx_ops_disable_workfn().
  4326. */
  4327. ei->kind = kind;
  4328. ei->reason = scx_exit_reason(ei->kind);
  4329. irq_work_queue(&scx_ops_error_irq_work);
  4330. }
  4331. static struct kthread_worker *scx_create_rt_helper(const char *name)
  4332. {
  4333. struct kthread_worker *helper;
  4334. helper = kthread_create_worker(0, name);
  4335. if (helper)
  4336. sched_set_fifo(helper->task);
  4337. return helper;
  4338. }
  4339. static void check_hotplug_seq(const struct sched_ext_ops *ops)
  4340. {
  4341. unsigned long long global_hotplug_seq;
  4342. /*
  4343. * If a hotplug event has occurred between when a scheduler was
  4344. * initialized, and when we were able to attach, exit and notify user
  4345. * space about it.
  4346. */
  4347. if (ops->hotplug_seq) {
  4348. global_hotplug_seq = atomic_long_read(&scx_hotplug_seq);
  4349. if (ops->hotplug_seq != global_hotplug_seq) {
  4350. scx_ops_exit(SCX_ECODE_ACT_RESTART | SCX_ECODE_RSN_HOTPLUG,
  4351. "expected hotplug seq %llu did not match actual %llu",
  4352. ops->hotplug_seq, global_hotplug_seq);
  4353. }
  4354. }
  4355. }
  4356. static int validate_ops(const struct sched_ext_ops *ops)
  4357. {
  4358. /*
  4359. * It doesn't make sense to specify the SCX_OPS_ENQ_LAST flag if the
  4360. * ops.enqueue() callback isn't implemented.
  4361. */
  4362. if ((ops->flags & SCX_OPS_ENQ_LAST) && !ops->enqueue) {
  4363. scx_ops_error("SCX_OPS_ENQ_LAST requires ops.enqueue() to be implemented");
  4364. return -EINVAL;
  4365. }
  4366. return 0;
  4367. }
  4368. static int scx_ops_enable(struct sched_ext_ops *ops, struct bpf_link *link)
  4369. {
  4370. struct scx_task_iter sti;
  4371. struct task_struct *p;
  4372. unsigned long timeout;
  4373. int i, cpu, node, ret;
  4374. if (!cpumask_equal(housekeeping_cpumask(HK_TYPE_DOMAIN),
  4375. cpu_possible_mask)) {
  4376. pr_err("sched_ext: Not compatible with \"isolcpus=\" domain isolation\n");
  4377. return -EINVAL;
  4378. }
  4379. mutex_lock(&scx_ops_enable_mutex);
  4380. if (!scx_ops_helper) {
  4381. WRITE_ONCE(scx_ops_helper,
  4382. scx_create_rt_helper("sched_ext_ops_helper"));
  4383. if (!scx_ops_helper) {
  4384. ret = -ENOMEM;
  4385. goto err_unlock;
  4386. }
  4387. }
  4388. if (!global_dsqs) {
  4389. struct scx_dispatch_q **dsqs;
  4390. dsqs = kcalloc(nr_node_ids, sizeof(dsqs[0]), GFP_KERNEL);
  4391. if (!dsqs) {
  4392. ret = -ENOMEM;
  4393. goto err_unlock;
  4394. }
  4395. for_each_node_state(node, N_POSSIBLE) {
  4396. struct scx_dispatch_q *dsq;
  4397. dsq = kzalloc_node(sizeof(*dsq), GFP_KERNEL, node);
  4398. if (!dsq) {
  4399. for_each_node_state(node, N_POSSIBLE)
  4400. kfree(dsqs[node]);
  4401. kfree(dsqs);
  4402. ret = -ENOMEM;
  4403. goto err_unlock;
  4404. }
  4405. init_dsq(dsq, SCX_DSQ_GLOBAL);
  4406. dsqs[node] = dsq;
  4407. }
  4408. global_dsqs = dsqs;
  4409. }
  4410. if (scx_ops_enable_state() != SCX_OPS_DISABLED) {
  4411. ret = -EBUSY;
  4412. goto err_unlock;
  4413. }
  4414. scx_root_kobj = kzalloc(sizeof(*scx_root_kobj), GFP_KERNEL);
  4415. if (!scx_root_kobj) {
  4416. ret = -ENOMEM;
  4417. goto err_unlock;
  4418. }
  4419. scx_root_kobj->kset = scx_kset;
  4420. ret = kobject_init_and_add(scx_root_kobj, &scx_ktype, NULL, "root");
  4421. if (ret < 0)
  4422. goto err;
  4423. scx_exit_info = alloc_exit_info(ops->exit_dump_len);
  4424. if (!scx_exit_info) {
  4425. ret = -ENOMEM;
  4426. goto err_del;
  4427. }
  4428. /*
  4429. * Set scx_ops, transition to ENABLING and clear exit info to arm the
  4430. * disable path. Failure triggers full disabling from here on.
  4431. */
  4432. scx_ops = *ops;
  4433. WARN_ON_ONCE(scx_ops_set_enable_state(SCX_OPS_ENABLING) !=
  4434. SCX_OPS_DISABLED);
  4435. atomic_set(&scx_exit_kind, SCX_EXIT_NONE);
  4436. scx_warned_zero_slice = false;
  4437. atomic_long_set(&scx_nr_rejected, 0);
  4438. for_each_possible_cpu(cpu)
  4439. cpu_rq(cpu)->scx.cpuperf_target = SCX_CPUPERF_ONE;
  4440. if (!ops->update_idle || (ops->flags & SCX_OPS_KEEP_BUILTIN_IDLE)) {
  4441. reset_idle_masks();
  4442. static_branch_enable(&scx_builtin_idle_enabled);
  4443. } else {
  4444. static_branch_disable(&scx_builtin_idle_enabled);
  4445. }
  4446. /*
  4447. * Keep CPUs stable during enable so that the BPF scheduler can track
  4448. * online CPUs by watching ->on/offline_cpu() after ->init().
  4449. */
  4450. cpus_read_lock();
  4451. if (scx_ops.init) {
  4452. ret = SCX_CALL_OP_RET(SCX_KF_UNLOCKED, init);
  4453. if (ret) {
  4454. ret = ops_sanitize_err("init", ret);
  4455. cpus_read_unlock();
  4456. scx_ops_error("ops.init() failed (%d)", ret);
  4457. goto err_disable;
  4458. }
  4459. }
  4460. for (i = SCX_OPI_CPU_HOTPLUG_BEGIN; i < SCX_OPI_CPU_HOTPLUG_END; i++)
  4461. if (((void (**)(void))ops)[i])
  4462. static_branch_enable_cpuslocked(&scx_has_op[i]);
  4463. check_hotplug_seq(ops);
  4464. cpus_read_unlock();
  4465. ret = validate_ops(ops);
  4466. if (ret)
  4467. goto err_disable;
  4468. WARN_ON_ONCE(scx_dsp_ctx);
  4469. scx_dsp_max_batch = ops->dispatch_max_batch ?: SCX_DSP_DFL_MAX_BATCH;
  4470. scx_dsp_ctx = __alloc_percpu(struct_size_t(struct scx_dsp_ctx, buf,
  4471. scx_dsp_max_batch),
  4472. __alignof__(struct scx_dsp_ctx));
  4473. if (!scx_dsp_ctx) {
  4474. ret = -ENOMEM;
  4475. goto err_disable;
  4476. }
  4477. if (ops->timeout_ms)
  4478. timeout = msecs_to_jiffies(ops->timeout_ms);
  4479. else
  4480. timeout = SCX_WATCHDOG_MAX_TIMEOUT;
  4481. WRITE_ONCE(scx_watchdog_timeout, timeout);
  4482. WRITE_ONCE(scx_watchdog_timestamp, jiffies);
  4483. queue_delayed_work(system_unbound_wq, &scx_watchdog_work,
  4484. scx_watchdog_timeout / 2);
  4485. /*
  4486. * Once __scx_ops_enabled is set, %current can be switched to SCX
  4487. * anytime. This can lead to stalls as some BPF schedulers (e.g.
  4488. * userspace scheduling) may not function correctly before all tasks are
  4489. * switched. Init in bypass mode to guarantee forward progress.
  4490. */
  4491. scx_ops_bypass(true);
  4492. for (i = SCX_OPI_NORMAL_BEGIN; i < SCX_OPI_NORMAL_END; i++)
  4493. if (((void (**)(void))ops)[i])
  4494. static_branch_enable(&scx_has_op[i]);
  4495. if (ops->flags & SCX_OPS_ENQ_LAST)
  4496. static_branch_enable(&scx_ops_enq_last);
  4497. if (ops->flags & SCX_OPS_ENQ_EXITING)
  4498. static_branch_enable(&scx_ops_enq_exiting);
  4499. if (scx_ops.cpu_acquire || scx_ops.cpu_release)
  4500. static_branch_enable(&scx_ops_cpu_preempt);
  4501. /*
  4502. * Lock out forks, cgroup on/offlining and moves before opening the
  4503. * floodgate so that they don't wander into the operations prematurely.
  4504. */
  4505. percpu_down_write(&scx_fork_rwsem);
  4506. WARN_ON_ONCE(scx_ops_init_task_enabled);
  4507. scx_ops_init_task_enabled = true;
  4508. /*
  4509. * Enable ops for every task. Fork is excluded by scx_fork_rwsem
  4510. * preventing new tasks from being added. No need to exclude tasks
  4511. * leaving as sched_ext_free() can handle both prepped and enabled
  4512. * tasks. Prep all tasks first and then enable them with preemption
  4513. * disabled.
  4514. *
  4515. * All cgroups should be initialized before scx_ops_init_task() so that
  4516. * the BPF scheduler can reliably track each task's cgroup membership
  4517. * from scx_ops_init_task(). Lock out cgroup on/offlining and task
  4518. * migrations while tasks are being initialized so that
  4519. * scx_cgroup_can_attach() never sees uninitialized tasks.
  4520. */
  4521. scx_cgroup_lock();
  4522. ret = scx_cgroup_init();
  4523. if (ret)
  4524. goto err_disable_unlock_all;
  4525. scx_task_iter_start(&sti);
  4526. while ((p = scx_task_iter_next_locked(&sti))) {
  4527. /*
  4528. * @p may already be dead, have lost all its usages counts and
  4529. * be waiting for RCU grace period before being freed. @p can't
  4530. * be initialized for SCX in such cases and should be ignored.
  4531. */
  4532. if (!tryget_task_struct(p))
  4533. continue;
  4534. scx_task_iter_unlock(&sti);
  4535. ret = scx_ops_init_task(p, task_group(p), false);
  4536. if (ret) {
  4537. put_task_struct(p);
  4538. scx_task_iter_relock(&sti);
  4539. scx_task_iter_stop(&sti);
  4540. scx_ops_error("ops.init_task() failed (%d) for %s[%d]",
  4541. ret, p->comm, p->pid);
  4542. goto err_disable_unlock_all;
  4543. }
  4544. scx_set_task_state(p, SCX_TASK_READY);
  4545. put_task_struct(p);
  4546. scx_task_iter_relock(&sti);
  4547. }
  4548. scx_task_iter_stop(&sti);
  4549. scx_cgroup_unlock();
  4550. percpu_up_write(&scx_fork_rwsem);
  4551. /*
  4552. * All tasks are READY. It's safe to turn on scx_enabled() and switch
  4553. * all eligible tasks.
  4554. */
  4555. WRITE_ONCE(scx_switching_all, !(ops->flags & SCX_OPS_SWITCH_PARTIAL));
  4556. static_branch_enable(&__scx_ops_enabled);
  4557. /*
  4558. * We're fully committed and can't fail. The task READY -> ENABLED
  4559. * transitions here are synchronized against sched_ext_free() through
  4560. * scx_tasks_lock.
  4561. */
  4562. percpu_down_write(&scx_fork_rwsem);
  4563. scx_task_iter_start(&sti);
  4564. while ((p = scx_task_iter_next_locked(&sti))) {
  4565. const struct sched_class *old_class = p->sched_class;
  4566. const struct sched_class *new_class =
  4567. __setscheduler_class(p->policy, p->prio);
  4568. struct sched_enq_and_set_ctx ctx;
  4569. if (!tryget_task_struct(p))
  4570. continue;
  4571. if (old_class != new_class && p->se.sched_delayed)
  4572. dequeue_task(task_rq(p), p, DEQUEUE_SLEEP | DEQUEUE_DELAYED);
  4573. sched_deq_and_put_task(p, DEQUEUE_SAVE | DEQUEUE_MOVE, &ctx);
  4574. p->scx.slice = SCX_SLICE_DFL;
  4575. p->sched_class = new_class;
  4576. check_class_changing(task_rq(p), p, old_class);
  4577. sched_enq_and_set_task(&ctx);
  4578. check_class_changed(task_rq(p), p, old_class, p->prio);
  4579. put_task_struct(p);
  4580. }
  4581. scx_task_iter_stop(&sti);
  4582. percpu_up_write(&scx_fork_rwsem);
  4583. scx_ops_bypass(false);
  4584. if (!scx_ops_tryset_enable_state(SCX_OPS_ENABLED, SCX_OPS_ENABLING)) {
  4585. WARN_ON_ONCE(atomic_read(&scx_exit_kind) == SCX_EXIT_NONE);
  4586. goto err_disable;
  4587. }
  4588. if (!(ops->flags & SCX_OPS_SWITCH_PARTIAL))
  4589. static_branch_enable(&__scx_switched_all);
  4590. pr_info("sched_ext: BPF scheduler \"%s\" enabled%s\n",
  4591. scx_ops.name, scx_switched_all() ? "" : " (partial)");
  4592. kobject_uevent(scx_root_kobj, KOBJ_ADD);
  4593. mutex_unlock(&scx_ops_enable_mutex);
  4594. atomic_long_inc(&scx_enable_seq);
  4595. return 0;
  4596. err_del:
  4597. kobject_del(scx_root_kobj);
  4598. err:
  4599. kobject_put(scx_root_kobj);
  4600. scx_root_kobj = NULL;
  4601. if (scx_exit_info) {
  4602. free_exit_info(scx_exit_info);
  4603. scx_exit_info = NULL;
  4604. }
  4605. err_unlock:
  4606. mutex_unlock(&scx_ops_enable_mutex);
  4607. return ret;
  4608. err_disable_unlock_all:
  4609. scx_cgroup_unlock();
  4610. percpu_up_write(&scx_fork_rwsem);
  4611. scx_ops_bypass(false);
  4612. err_disable:
  4613. mutex_unlock(&scx_ops_enable_mutex);
  4614. /*
  4615. * Returning an error code here would not pass all the error information
  4616. * to userspace. Record errno using scx_ops_error() for cases
  4617. * scx_ops_error() wasn't already invoked and exit indicating success so
  4618. * that the error is notified through ops.exit() with all the details.
  4619. *
  4620. * Flush scx_ops_disable_work to ensure that error is reported before
  4621. * init completion.
  4622. */
  4623. scx_ops_error("scx_ops_enable() failed (%d)", ret);
  4624. kthread_flush_work(&scx_ops_disable_work);
  4625. return 0;
  4626. }
  4627. /********************************************************************************
  4628. * bpf_struct_ops plumbing.
  4629. */
  4630. #include <linux/bpf_verifier.h>
  4631. #include <linux/bpf.h>
  4632. #include <linux/btf.h>
  4633. extern struct btf *btf_vmlinux;
  4634. static const struct btf_type *task_struct_type;
  4635. static u32 task_struct_type_id;
  4636. static bool set_arg_maybe_null(const char *op, int arg_n, int off, int size,
  4637. enum bpf_access_type type,
  4638. const struct bpf_prog *prog,
  4639. struct bpf_insn_access_aux *info)
  4640. {
  4641. struct btf *btf = bpf_get_btf_vmlinux();
  4642. const struct bpf_struct_ops_desc *st_ops_desc;
  4643. const struct btf_member *member;
  4644. const struct btf_type *t;
  4645. u32 btf_id, member_idx;
  4646. const char *mname;
  4647. /* struct_ops op args are all sequential, 64-bit numbers */
  4648. if (off != arg_n * sizeof(__u64))
  4649. return false;
  4650. /* btf_id should be the type id of struct sched_ext_ops */
  4651. btf_id = prog->aux->attach_btf_id;
  4652. st_ops_desc = bpf_struct_ops_find(btf, btf_id);
  4653. if (!st_ops_desc)
  4654. return false;
  4655. /* BTF type of struct sched_ext_ops */
  4656. t = st_ops_desc->type;
  4657. member_idx = prog->expected_attach_type;
  4658. if (member_idx >= btf_type_vlen(t))
  4659. return false;
  4660. /*
  4661. * Get the member name of this struct_ops program, which corresponds to
  4662. * a field in struct sched_ext_ops. For example, the member name of the
  4663. * dispatch struct_ops program (callback) is "dispatch".
  4664. */
  4665. member = &btf_type_member(t)[member_idx];
  4666. mname = btf_name_by_offset(btf_vmlinux, member->name_off);
  4667. if (!strcmp(mname, op)) {
  4668. /*
  4669. * The value is a pointer to a type (struct task_struct) given
  4670. * by a BTF ID (PTR_TO_BTF_ID). It is trusted (PTR_TRUSTED),
  4671. * however, can be a NULL (PTR_MAYBE_NULL). The BPF program
  4672. * should check the pointer to make sure it is not NULL before
  4673. * using it, or the verifier will reject the program.
  4674. *
  4675. * Longer term, this is something that should be addressed by
  4676. * BTF, and be fully contained within the verifier.
  4677. */
  4678. info->reg_type = PTR_MAYBE_NULL | PTR_TO_BTF_ID | PTR_TRUSTED;
  4679. info->btf = btf_vmlinux;
  4680. info->btf_id = task_struct_type_id;
  4681. return true;
  4682. }
  4683. return false;
  4684. }
  4685. static bool bpf_scx_is_valid_access(int off, int size,
  4686. enum bpf_access_type type,
  4687. const struct bpf_prog *prog,
  4688. struct bpf_insn_access_aux *info)
  4689. {
  4690. if (type != BPF_READ)
  4691. return false;
  4692. if (set_arg_maybe_null("dispatch", 1, off, size, type, prog, info) ||
  4693. set_arg_maybe_null("yield", 1, off, size, type, prog, info))
  4694. return true;
  4695. if (off < 0 || off >= sizeof(__u64) * MAX_BPF_FUNC_ARGS)
  4696. return false;
  4697. if (off % size != 0)
  4698. return false;
  4699. return btf_ctx_access(off, size, type, prog, info);
  4700. }
  4701. static int bpf_scx_btf_struct_access(struct bpf_verifier_log *log,
  4702. const struct bpf_reg_state *reg, int off,
  4703. int size)
  4704. {
  4705. const struct btf_type *t;
  4706. t = btf_type_by_id(reg->btf, reg->btf_id);
  4707. if (t == task_struct_type) {
  4708. if (off >= offsetof(struct task_struct, scx.slice) &&
  4709. off + size <= offsetofend(struct task_struct, scx.slice))
  4710. return SCALAR_VALUE;
  4711. if (off >= offsetof(struct task_struct, scx.dsq_vtime) &&
  4712. off + size <= offsetofend(struct task_struct, scx.dsq_vtime))
  4713. return SCALAR_VALUE;
  4714. if (off >= offsetof(struct task_struct, scx.disallow) &&
  4715. off + size <= offsetofend(struct task_struct, scx.disallow))
  4716. return SCALAR_VALUE;
  4717. }
  4718. return -EACCES;
  4719. }
  4720. static const struct bpf_func_proto *
  4721. bpf_scx_get_func_proto(enum bpf_func_id func_id, const struct bpf_prog *prog)
  4722. {
  4723. switch (func_id) {
  4724. case BPF_FUNC_task_storage_get:
  4725. return &bpf_task_storage_get_proto;
  4726. case BPF_FUNC_task_storage_delete:
  4727. return &bpf_task_storage_delete_proto;
  4728. default:
  4729. return bpf_base_func_proto(func_id, prog);
  4730. }
  4731. }
  4732. static const struct bpf_verifier_ops bpf_scx_verifier_ops = {
  4733. .get_func_proto = bpf_scx_get_func_proto,
  4734. .is_valid_access = bpf_scx_is_valid_access,
  4735. .btf_struct_access = bpf_scx_btf_struct_access,
  4736. };
  4737. static int bpf_scx_init_member(const struct btf_type *t,
  4738. const struct btf_member *member,
  4739. void *kdata, const void *udata)
  4740. {
  4741. const struct sched_ext_ops *uops = udata;
  4742. struct sched_ext_ops *ops = kdata;
  4743. u32 moff = __btf_member_bit_offset(t, member) / 8;
  4744. int ret;
  4745. switch (moff) {
  4746. case offsetof(struct sched_ext_ops, dispatch_max_batch):
  4747. if (*(u32 *)(udata + moff) > INT_MAX)
  4748. return -E2BIG;
  4749. ops->dispatch_max_batch = *(u32 *)(udata + moff);
  4750. return 1;
  4751. case offsetof(struct sched_ext_ops, flags):
  4752. if (*(u64 *)(udata + moff) & ~SCX_OPS_ALL_FLAGS)
  4753. return -EINVAL;
  4754. ops->flags = *(u64 *)(udata + moff);
  4755. return 1;
  4756. case offsetof(struct sched_ext_ops, name):
  4757. ret = bpf_obj_name_cpy(ops->name, uops->name,
  4758. sizeof(ops->name));
  4759. if (ret < 0)
  4760. return ret;
  4761. if (ret == 0)
  4762. return -EINVAL;
  4763. return 1;
  4764. case offsetof(struct sched_ext_ops, timeout_ms):
  4765. if (msecs_to_jiffies(*(u32 *)(udata + moff)) >
  4766. SCX_WATCHDOG_MAX_TIMEOUT)
  4767. return -E2BIG;
  4768. ops->timeout_ms = *(u32 *)(udata + moff);
  4769. return 1;
  4770. case offsetof(struct sched_ext_ops, exit_dump_len):
  4771. ops->exit_dump_len =
  4772. *(u32 *)(udata + moff) ?: SCX_EXIT_DUMP_DFL_LEN;
  4773. return 1;
  4774. case offsetof(struct sched_ext_ops, hotplug_seq):
  4775. ops->hotplug_seq = *(u64 *)(udata + moff);
  4776. return 1;
  4777. }
  4778. return 0;
  4779. }
  4780. static int bpf_scx_check_member(const struct btf_type *t,
  4781. const struct btf_member *member,
  4782. const struct bpf_prog *prog)
  4783. {
  4784. u32 moff = __btf_member_bit_offset(t, member) / 8;
  4785. switch (moff) {
  4786. case offsetof(struct sched_ext_ops, init_task):
  4787. #ifdef CONFIG_EXT_GROUP_SCHED
  4788. case offsetof(struct sched_ext_ops, cgroup_init):
  4789. case offsetof(struct sched_ext_ops, cgroup_exit):
  4790. case offsetof(struct sched_ext_ops, cgroup_prep_move):
  4791. #endif
  4792. case offsetof(struct sched_ext_ops, cpu_online):
  4793. case offsetof(struct sched_ext_ops, cpu_offline):
  4794. case offsetof(struct sched_ext_ops, init):
  4795. case offsetof(struct sched_ext_ops, exit):
  4796. break;
  4797. default:
  4798. if (prog->sleepable)
  4799. return -EINVAL;
  4800. }
  4801. return 0;
  4802. }
  4803. static int bpf_scx_reg(void *kdata, struct bpf_link *link)
  4804. {
  4805. return scx_ops_enable(kdata, link);
  4806. }
  4807. static void bpf_scx_unreg(void *kdata, struct bpf_link *link)
  4808. {
  4809. scx_ops_disable(SCX_EXIT_UNREG);
  4810. kthread_flush_work(&scx_ops_disable_work);
  4811. }
  4812. static int bpf_scx_init(struct btf *btf)
  4813. {
  4814. s32 type_id;
  4815. type_id = btf_find_by_name_kind(btf, "task_struct", BTF_KIND_STRUCT);
  4816. if (type_id < 0)
  4817. return -EINVAL;
  4818. task_struct_type = btf_type_by_id(btf, type_id);
  4819. task_struct_type_id = type_id;
  4820. return 0;
  4821. }
  4822. static int bpf_scx_update(void *kdata, void *old_kdata, struct bpf_link *link)
  4823. {
  4824. /*
  4825. * sched_ext does not support updating the actively-loaded BPF
  4826. * scheduler, as registering a BPF scheduler can always fail if the
  4827. * scheduler returns an error code for e.g. ops.init(), ops.init_task(),
  4828. * etc. Similarly, we can always race with unregistration happening
  4829. * elsewhere, such as with sysrq.
  4830. */
  4831. return -EOPNOTSUPP;
  4832. }
  4833. static int bpf_scx_validate(void *kdata)
  4834. {
  4835. return 0;
  4836. }
  4837. static s32 select_cpu_stub(struct task_struct *p, s32 prev_cpu, u64 wake_flags) { return -EINVAL; }
  4838. static void enqueue_stub(struct task_struct *p, u64 enq_flags) {}
  4839. static void dequeue_stub(struct task_struct *p, u64 enq_flags) {}
  4840. static void dispatch_stub(s32 prev_cpu, struct task_struct *p) {}
  4841. static void tick_stub(struct task_struct *p) {}
  4842. static void runnable_stub(struct task_struct *p, u64 enq_flags) {}
  4843. static void running_stub(struct task_struct *p) {}
  4844. static void stopping_stub(struct task_struct *p, bool runnable) {}
  4845. static void quiescent_stub(struct task_struct *p, u64 deq_flags) {}
  4846. static bool yield_stub(struct task_struct *from, struct task_struct *to) { return false; }
  4847. static bool core_sched_before_stub(struct task_struct *a, struct task_struct *b) { return false; }
  4848. static void set_weight_stub(struct task_struct *p, u32 weight) {}
  4849. static void set_cpumask_stub(struct task_struct *p, const struct cpumask *mask) {}
  4850. static void update_idle_stub(s32 cpu, bool idle) {}
  4851. static void cpu_acquire_stub(s32 cpu, struct scx_cpu_acquire_args *args) {}
  4852. static void cpu_release_stub(s32 cpu, struct scx_cpu_release_args *args) {}
  4853. static s32 init_task_stub(struct task_struct *p, struct scx_init_task_args *args) { return -EINVAL; }
  4854. static void exit_task_stub(struct task_struct *p, struct scx_exit_task_args *args) {}
  4855. static void enable_stub(struct task_struct *p) {}
  4856. static void disable_stub(struct task_struct *p) {}
  4857. #ifdef CONFIG_EXT_GROUP_SCHED
  4858. static s32 cgroup_init_stub(struct cgroup *cgrp, struct scx_cgroup_init_args *args) { return -EINVAL; }
  4859. static void cgroup_exit_stub(struct cgroup *cgrp) {}
  4860. static s32 cgroup_prep_move_stub(struct task_struct *p, struct cgroup *from, struct cgroup *to) { return -EINVAL; }
  4861. static void cgroup_move_stub(struct task_struct *p, struct cgroup *from, struct cgroup *to) {}
  4862. static void cgroup_cancel_move_stub(struct task_struct *p, struct cgroup *from, struct cgroup *to) {}
  4863. static void cgroup_set_weight_stub(struct cgroup *cgrp, u32 weight) {}
  4864. #endif
  4865. static void cpu_online_stub(s32 cpu) {}
  4866. static void cpu_offline_stub(s32 cpu) {}
  4867. static s32 init_stub(void) { return -EINVAL; }
  4868. static void exit_stub(struct scx_exit_info *info) {}
  4869. static void dump_stub(struct scx_dump_ctx *ctx) {}
  4870. static void dump_cpu_stub(struct scx_dump_ctx *ctx, s32 cpu, bool idle) {}
  4871. static void dump_task_stub(struct scx_dump_ctx *ctx, struct task_struct *p) {}
  4872. static struct sched_ext_ops __bpf_ops_sched_ext_ops = {
  4873. .select_cpu = select_cpu_stub,
  4874. .enqueue = enqueue_stub,
  4875. .dequeue = dequeue_stub,
  4876. .dispatch = dispatch_stub,
  4877. .tick = tick_stub,
  4878. .runnable = runnable_stub,
  4879. .running = running_stub,
  4880. .stopping = stopping_stub,
  4881. .quiescent = quiescent_stub,
  4882. .yield = yield_stub,
  4883. .core_sched_before = core_sched_before_stub,
  4884. .set_weight = set_weight_stub,
  4885. .set_cpumask = set_cpumask_stub,
  4886. .update_idle = update_idle_stub,
  4887. .cpu_acquire = cpu_acquire_stub,
  4888. .cpu_release = cpu_release_stub,
  4889. .init_task = init_task_stub,
  4890. .exit_task = exit_task_stub,
  4891. .enable = enable_stub,
  4892. .disable = disable_stub,
  4893. #ifdef CONFIG_EXT_GROUP_SCHED
  4894. .cgroup_init = cgroup_init_stub,
  4895. .cgroup_exit = cgroup_exit_stub,
  4896. .cgroup_prep_move = cgroup_prep_move_stub,
  4897. .cgroup_move = cgroup_move_stub,
  4898. .cgroup_cancel_move = cgroup_cancel_move_stub,
  4899. .cgroup_set_weight = cgroup_set_weight_stub,
  4900. #endif
  4901. .cpu_online = cpu_online_stub,
  4902. .cpu_offline = cpu_offline_stub,
  4903. .init = init_stub,
  4904. .exit = exit_stub,
  4905. .dump = dump_stub,
  4906. .dump_cpu = dump_cpu_stub,
  4907. .dump_task = dump_task_stub,
  4908. };
  4909. static struct bpf_struct_ops bpf_sched_ext_ops = {
  4910. .verifier_ops = &bpf_scx_verifier_ops,
  4911. .reg = bpf_scx_reg,
  4912. .unreg = bpf_scx_unreg,
  4913. .check_member = bpf_scx_check_member,
  4914. .init_member = bpf_scx_init_member,
  4915. .init = bpf_scx_init,
  4916. .update = bpf_scx_update,
  4917. .validate = bpf_scx_validate,
  4918. .name = "sched_ext_ops",
  4919. .owner = THIS_MODULE,
  4920. .cfi_stubs = &__bpf_ops_sched_ext_ops
  4921. };
  4922. /********************************************************************************
  4923. * System integration and init.
  4924. */
  4925. static void sysrq_handle_sched_ext_reset(u8 key)
  4926. {
  4927. if (scx_ops_helper)
  4928. scx_ops_disable(SCX_EXIT_SYSRQ);
  4929. else
  4930. pr_info("sched_ext: BPF scheduler not yet used\n");
  4931. }
  4932. static const struct sysrq_key_op sysrq_sched_ext_reset_op = {
  4933. .handler = sysrq_handle_sched_ext_reset,
  4934. .help_msg = "reset-sched-ext(S)",
  4935. .action_msg = "Disable sched_ext and revert all tasks to CFS",
  4936. .enable_mask = SYSRQ_ENABLE_RTNICE,
  4937. };
  4938. static void sysrq_handle_sched_ext_dump(u8 key)
  4939. {
  4940. struct scx_exit_info ei = { .kind = SCX_EXIT_NONE, .reason = "SysRq-D" };
  4941. if (scx_enabled())
  4942. scx_dump_state(&ei, 0);
  4943. }
  4944. static const struct sysrq_key_op sysrq_sched_ext_dump_op = {
  4945. .handler = sysrq_handle_sched_ext_dump,
  4946. .help_msg = "dump-sched-ext(D)",
  4947. .action_msg = "Trigger sched_ext debug dump",
  4948. .enable_mask = SYSRQ_ENABLE_RTNICE,
  4949. };
  4950. static bool can_skip_idle_kick(struct rq *rq)
  4951. {
  4952. lockdep_assert_rq_held(rq);
  4953. /*
  4954. * We can skip idle kicking if @rq is going to go through at least one
  4955. * full SCX scheduling cycle before going idle. Just checking whether
  4956. * curr is not idle is insufficient because we could be racing
  4957. * balance_one() trying to pull the next task from a remote rq, which
  4958. * may fail, and @rq may become idle afterwards.
  4959. *
  4960. * The race window is small and we don't and can't guarantee that @rq is
  4961. * only kicked while idle anyway. Skip only when sure.
  4962. */
  4963. return !is_idle_task(rq->curr) && !(rq->scx.flags & SCX_RQ_IN_BALANCE);
  4964. }
  4965. static bool kick_one_cpu(s32 cpu, struct rq *this_rq, unsigned long *pseqs)
  4966. {
  4967. struct rq *rq = cpu_rq(cpu);
  4968. struct scx_rq *this_scx = &this_rq->scx;
  4969. bool should_wait = false;
  4970. unsigned long flags;
  4971. raw_spin_rq_lock_irqsave(rq, flags);
  4972. /*
  4973. * During CPU hotplug, a CPU may depend on kicking itself to make
  4974. * forward progress. Allow kicking self regardless of online state.
  4975. */
  4976. if (cpu_online(cpu) || cpu == cpu_of(this_rq)) {
  4977. if (cpumask_test_cpu(cpu, this_scx->cpus_to_preempt)) {
  4978. if (rq->curr->sched_class == &ext_sched_class)
  4979. rq->curr->scx.slice = 0;
  4980. cpumask_clear_cpu(cpu, this_scx->cpus_to_preempt);
  4981. }
  4982. if (cpumask_test_cpu(cpu, this_scx->cpus_to_wait)) {
  4983. pseqs[cpu] = rq->scx.pnt_seq;
  4984. should_wait = true;
  4985. }
  4986. resched_curr(rq);
  4987. } else {
  4988. cpumask_clear_cpu(cpu, this_scx->cpus_to_preempt);
  4989. cpumask_clear_cpu(cpu, this_scx->cpus_to_wait);
  4990. }
  4991. raw_spin_rq_unlock_irqrestore(rq, flags);
  4992. return should_wait;
  4993. }
  4994. static void kick_one_cpu_if_idle(s32 cpu, struct rq *this_rq)
  4995. {
  4996. struct rq *rq = cpu_rq(cpu);
  4997. unsigned long flags;
  4998. raw_spin_rq_lock_irqsave(rq, flags);
  4999. if (!can_skip_idle_kick(rq) &&
  5000. (cpu_online(cpu) || cpu == cpu_of(this_rq)))
  5001. resched_curr(rq);
  5002. raw_spin_rq_unlock_irqrestore(rq, flags);
  5003. }
  5004. static void kick_cpus_irq_workfn(struct irq_work *irq_work)
  5005. {
  5006. struct rq *this_rq = this_rq();
  5007. struct scx_rq *this_scx = &this_rq->scx;
  5008. unsigned long *pseqs = this_cpu_ptr(scx_kick_cpus_pnt_seqs);
  5009. bool should_wait = false;
  5010. s32 cpu;
  5011. for_each_cpu(cpu, this_scx->cpus_to_kick) {
  5012. should_wait |= kick_one_cpu(cpu, this_rq, pseqs);
  5013. cpumask_clear_cpu(cpu, this_scx->cpus_to_kick);
  5014. cpumask_clear_cpu(cpu, this_scx->cpus_to_kick_if_idle);
  5015. }
  5016. for_each_cpu(cpu, this_scx->cpus_to_kick_if_idle) {
  5017. kick_one_cpu_if_idle(cpu, this_rq);
  5018. cpumask_clear_cpu(cpu, this_scx->cpus_to_kick_if_idle);
  5019. }
  5020. if (!should_wait)
  5021. return;
  5022. for_each_cpu(cpu, this_scx->cpus_to_wait) {
  5023. unsigned long *wait_pnt_seq = &cpu_rq(cpu)->scx.pnt_seq;
  5024. if (cpu != cpu_of(this_rq)) {
  5025. /*
  5026. * Pairs with smp_store_release() issued by this CPU in
  5027. * scx_next_task_picked() on the resched path.
  5028. *
  5029. * We busy-wait here to guarantee that no other task can
  5030. * be scheduled on our core before the target CPU has
  5031. * entered the resched path.
  5032. */
  5033. while (smp_load_acquire(wait_pnt_seq) == pseqs[cpu])
  5034. cpu_relax();
  5035. }
  5036. cpumask_clear_cpu(cpu, this_scx->cpus_to_wait);
  5037. }
  5038. }
  5039. /**
  5040. * print_scx_info - print out sched_ext scheduler state
  5041. * @log_lvl: the log level to use when printing
  5042. * @p: target task
  5043. *
  5044. * If a sched_ext scheduler is enabled, print the name and state of the
  5045. * scheduler. If @p is on sched_ext, print further information about the task.
  5046. *
  5047. * This function can be safely called on any task as long as the task_struct
  5048. * itself is accessible. While safe, this function isn't synchronized and may
  5049. * print out mixups or garbages of limited length.
  5050. */
  5051. void print_scx_info(const char *log_lvl, struct task_struct *p)
  5052. {
  5053. enum scx_ops_enable_state state = scx_ops_enable_state();
  5054. const char *all = READ_ONCE(scx_switching_all) ? "+all" : "";
  5055. char runnable_at_buf[22] = "?";
  5056. struct sched_class *class;
  5057. unsigned long runnable_at;
  5058. if (state == SCX_OPS_DISABLED)
  5059. return;
  5060. /*
  5061. * Carefully check if the task was running on sched_ext, and then
  5062. * carefully copy the time it's been runnable, and its state.
  5063. */
  5064. if (copy_from_kernel_nofault(&class, &p->sched_class, sizeof(class)) ||
  5065. class != &ext_sched_class) {
  5066. printk("%sSched_ext: %s (%s%s)", log_lvl, scx_ops.name,
  5067. scx_ops_enable_state_str[state], all);
  5068. return;
  5069. }
  5070. if (!copy_from_kernel_nofault(&runnable_at, &p->scx.runnable_at,
  5071. sizeof(runnable_at)))
  5072. scnprintf(runnable_at_buf, sizeof(runnable_at_buf), "%+ldms",
  5073. jiffies_delta_msecs(runnable_at, jiffies));
  5074. /* print everything onto one line to conserve console space */
  5075. printk("%sSched_ext: %s (%s%s), task: runnable_at=%s",
  5076. log_lvl, scx_ops.name, scx_ops_enable_state_str[state], all,
  5077. runnable_at_buf);
  5078. }
  5079. static int scx_pm_handler(struct notifier_block *nb, unsigned long event, void *ptr)
  5080. {
  5081. /*
  5082. * SCX schedulers often have userspace components which are sometimes
  5083. * involved in critial scheduling paths. PM operations involve freezing
  5084. * userspace which can lead to scheduling misbehaviors including stalls.
  5085. * Let's bypass while PM operations are in progress.
  5086. */
  5087. switch (event) {
  5088. case PM_HIBERNATION_PREPARE:
  5089. case PM_SUSPEND_PREPARE:
  5090. case PM_RESTORE_PREPARE:
  5091. scx_ops_bypass(true);
  5092. break;
  5093. case PM_POST_HIBERNATION:
  5094. case PM_POST_SUSPEND:
  5095. case PM_POST_RESTORE:
  5096. scx_ops_bypass(false);
  5097. break;
  5098. }
  5099. return NOTIFY_OK;
  5100. }
  5101. static struct notifier_block scx_pm_notifier = {
  5102. .notifier_call = scx_pm_handler,
  5103. };
  5104. void __init init_sched_ext_class(void)
  5105. {
  5106. s32 cpu, v;
  5107. /*
  5108. * The following is to prevent the compiler from optimizing out the enum
  5109. * definitions so that BPF scheduler implementations can use them
  5110. * through the generated vmlinux.h.
  5111. */
  5112. WRITE_ONCE(v, SCX_ENQ_WAKEUP | SCX_DEQ_SLEEP | SCX_KICK_PREEMPT |
  5113. SCX_TG_ONLINE);
  5114. BUG_ON(rhashtable_init(&dsq_hash, &dsq_hash_params));
  5115. #ifdef CONFIG_SMP
  5116. BUG_ON(!alloc_cpumask_var(&idle_masks.cpu, GFP_KERNEL));
  5117. BUG_ON(!alloc_cpumask_var(&idle_masks.smt, GFP_KERNEL));
  5118. #endif
  5119. scx_kick_cpus_pnt_seqs =
  5120. __alloc_percpu(sizeof(scx_kick_cpus_pnt_seqs[0]) * nr_cpu_ids,
  5121. __alignof__(scx_kick_cpus_pnt_seqs[0]));
  5122. BUG_ON(!scx_kick_cpus_pnt_seqs);
  5123. for_each_possible_cpu(cpu) {
  5124. struct rq *rq = cpu_rq(cpu);
  5125. init_dsq(&rq->scx.local_dsq, SCX_DSQ_LOCAL);
  5126. INIT_LIST_HEAD(&rq->scx.runnable_list);
  5127. INIT_LIST_HEAD(&rq->scx.ddsp_deferred_locals);
  5128. BUG_ON(!zalloc_cpumask_var(&rq->scx.cpus_to_kick, GFP_KERNEL));
  5129. BUG_ON(!zalloc_cpumask_var(&rq->scx.cpus_to_kick_if_idle, GFP_KERNEL));
  5130. BUG_ON(!zalloc_cpumask_var(&rq->scx.cpus_to_preempt, GFP_KERNEL));
  5131. BUG_ON(!zalloc_cpumask_var(&rq->scx.cpus_to_wait, GFP_KERNEL));
  5132. init_irq_work(&rq->scx.deferred_irq_work, deferred_irq_workfn);
  5133. init_irq_work(&rq->scx.kick_cpus_irq_work, kick_cpus_irq_workfn);
  5134. if (cpu_online(cpu))
  5135. cpu_rq(cpu)->scx.flags |= SCX_RQ_ONLINE;
  5136. }
  5137. register_sysrq_key('S', &sysrq_sched_ext_reset_op);
  5138. register_sysrq_key('D', &sysrq_sched_ext_dump_op);
  5139. INIT_DELAYED_WORK(&scx_watchdog_work, scx_watchdog_workfn);
  5140. }
  5141. /********************************************************************************
  5142. * Helpers that can be called from the BPF scheduler.
  5143. */
  5144. #include <linux/btf_ids.h>
  5145. __bpf_kfunc_start_defs();
  5146. /**
  5147. * scx_bpf_select_cpu_dfl - The default implementation of ops.select_cpu()
  5148. * @p: task_struct to select a CPU for
  5149. * @prev_cpu: CPU @p was on previously
  5150. * @wake_flags: %SCX_WAKE_* flags
  5151. * @is_idle: out parameter indicating whether the returned CPU is idle
  5152. *
  5153. * Can only be called from ops.select_cpu() if the built-in CPU selection is
  5154. * enabled - ops.update_idle() is missing or %SCX_OPS_KEEP_BUILTIN_IDLE is set.
  5155. * @p, @prev_cpu and @wake_flags match ops.select_cpu().
  5156. *
  5157. * Returns the picked CPU with *@is_idle indicating whether the picked CPU is
  5158. * currently idle and thus a good candidate for direct dispatching.
  5159. */
  5160. __bpf_kfunc s32 scx_bpf_select_cpu_dfl(struct task_struct *p, s32 prev_cpu,
  5161. u64 wake_flags, bool *is_idle)
  5162. {
  5163. if (!ops_cpu_valid(prev_cpu, NULL))
  5164. goto prev_cpu;
  5165. if (!static_branch_likely(&scx_builtin_idle_enabled)) {
  5166. scx_ops_error("built-in idle tracking is disabled");
  5167. goto prev_cpu;
  5168. }
  5169. if (!scx_kf_allowed(SCX_KF_SELECT_CPU))
  5170. goto prev_cpu;
  5171. #ifdef CONFIG_SMP
  5172. return scx_select_cpu_dfl(p, prev_cpu, wake_flags, is_idle);
  5173. #endif
  5174. prev_cpu:
  5175. *is_idle = false;
  5176. return prev_cpu;
  5177. }
  5178. __bpf_kfunc_end_defs();
  5179. BTF_KFUNCS_START(scx_kfunc_ids_select_cpu)
  5180. BTF_ID_FLAGS(func, scx_bpf_select_cpu_dfl, KF_RCU)
  5181. BTF_KFUNCS_END(scx_kfunc_ids_select_cpu)
  5182. static const struct btf_kfunc_id_set scx_kfunc_set_select_cpu = {
  5183. .owner = THIS_MODULE,
  5184. .set = &scx_kfunc_ids_select_cpu,
  5185. };
  5186. static bool scx_dispatch_preamble(struct task_struct *p, u64 enq_flags)
  5187. {
  5188. if (!scx_kf_allowed(SCX_KF_ENQUEUE | SCX_KF_DISPATCH))
  5189. return false;
  5190. lockdep_assert_irqs_disabled();
  5191. if (unlikely(!p)) {
  5192. scx_ops_error("called with NULL task");
  5193. return false;
  5194. }
  5195. if (unlikely(enq_flags & __SCX_ENQ_INTERNAL_MASK)) {
  5196. scx_ops_error("invalid enq_flags 0x%llx", enq_flags);
  5197. return false;
  5198. }
  5199. return true;
  5200. }
  5201. static void scx_dispatch_commit(struct task_struct *p, u64 dsq_id, u64 enq_flags)
  5202. {
  5203. struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
  5204. struct task_struct *ddsp_task;
  5205. ddsp_task = __this_cpu_read(direct_dispatch_task);
  5206. if (ddsp_task) {
  5207. mark_direct_dispatch(ddsp_task, p, dsq_id, enq_flags);
  5208. return;
  5209. }
  5210. if (unlikely(dspc->cursor >= scx_dsp_max_batch)) {
  5211. scx_ops_error("dispatch buffer overflow");
  5212. return;
  5213. }
  5214. dspc->buf[dspc->cursor++] = (struct scx_dsp_buf_ent){
  5215. .task = p,
  5216. .qseq = atomic_long_read(&p->scx.ops_state) & SCX_OPSS_QSEQ_MASK,
  5217. .dsq_id = dsq_id,
  5218. .enq_flags = enq_flags,
  5219. };
  5220. }
  5221. __bpf_kfunc_start_defs();
  5222. /**
  5223. * scx_bpf_dispatch - Dispatch a task into the FIFO queue of a DSQ
  5224. * @p: task_struct to dispatch
  5225. * @dsq_id: DSQ to dispatch to
  5226. * @slice: duration @p can run for in nsecs, 0 to keep the current value
  5227. * @enq_flags: SCX_ENQ_*
  5228. *
  5229. * Dispatch @p into the FIFO queue of the DSQ identified by @dsq_id. It is safe
  5230. * to call this function spuriously. Can be called from ops.enqueue(),
  5231. * ops.select_cpu(), and ops.dispatch().
  5232. *
  5233. * When called from ops.select_cpu() or ops.enqueue(), it's for direct dispatch
  5234. * and @p must match the task being enqueued. Also, %SCX_DSQ_LOCAL_ON can't be
  5235. * used to target the local DSQ of a CPU other than the enqueueing one. Use
  5236. * ops.select_cpu() to be on the target CPU in the first place.
  5237. *
  5238. * When called from ops.select_cpu(), @enq_flags and @dsp_id are stored, and @p
  5239. * will be directly dispatched to the corresponding dispatch queue after
  5240. * ops.select_cpu() returns. If @p is dispatched to SCX_DSQ_LOCAL, it will be
  5241. * dispatched to the local DSQ of the CPU returned by ops.select_cpu().
  5242. * @enq_flags are OR'd with the enqueue flags on the enqueue path before the
  5243. * task is dispatched.
  5244. *
  5245. * When called from ops.dispatch(), there are no restrictions on @p or @dsq_id
  5246. * and this function can be called upto ops.dispatch_max_batch times to dispatch
  5247. * multiple tasks. scx_bpf_dispatch_nr_slots() returns the number of the
  5248. * remaining slots. scx_bpf_consume() flushes the batch and resets the counter.
  5249. *
  5250. * This function doesn't have any locking restrictions and may be called under
  5251. * BPF locks (in the future when BPF introduces more flexible locking).
  5252. *
  5253. * @p is allowed to run for @slice. The scheduling path is triggered on slice
  5254. * exhaustion. If zero, the current residual slice is maintained. If
  5255. * %SCX_SLICE_INF, @p never expires and the BPF scheduler must kick the CPU with
  5256. * scx_bpf_kick_cpu() to trigger scheduling.
  5257. */
  5258. __bpf_kfunc void scx_bpf_dispatch(struct task_struct *p, u64 dsq_id, u64 slice,
  5259. u64 enq_flags)
  5260. {
  5261. if (!scx_dispatch_preamble(p, enq_flags))
  5262. return;
  5263. if (slice)
  5264. p->scx.slice = slice;
  5265. else
  5266. p->scx.slice = p->scx.slice ?: 1;
  5267. scx_dispatch_commit(p, dsq_id, enq_flags);
  5268. }
  5269. /**
  5270. * scx_bpf_dispatch_vtime - Dispatch a task into the vtime priority queue of a DSQ
  5271. * @p: task_struct to dispatch
  5272. * @dsq_id: DSQ to dispatch to
  5273. * @slice: duration @p can run for in nsecs, 0 to keep the current value
  5274. * @vtime: @p's ordering inside the vtime-sorted queue of the target DSQ
  5275. * @enq_flags: SCX_ENQ_*
  5276. *
  5277. * Dispatch @p into the vtime priority queue of the DSQ identified by @dsq_id.
  5278. * Tasks queued into the priority queue are ordered by @vtime and always
  5279. * consumed after the tasks in the FIFO queue. All other aspects are identical
  5280. * to scx_bpf_dispatch().
  5281. *
  5282. * @vtime ordering is according to time_before64() which considers wrapping. A
  5283. * numerically larger vtime may indicate an earlier position in the ordering and
  5284. * vice-versa.
  5285. */
  5286. __bpf_kfunc void scx_bpf_dispatch_vtime(struct task_struct *p, u64 dsq_id,
  5287. u64 slice, u64 vtime, u64 enq_flags)
  5288. {
  5289. if (!scx_dispatch_preamble(p, enq_flags))
  5290. return;
  5291. if (slice)
  5292. p->scx.slice = slice;
  5293. else
  5294. p->scx.slice = p->scx.slice ?: 1;
  5295. p->scx.dsq_vtime = vtime;
  5296. scx_dispatch_commit(p, dsq_id, enq_flags | SCX_ENQ_DSQ_PRIQ);
  5297. }
  5298. __bpf_kfunc_end_defs();
  5299. BTF_KFUNCS_START(scx_kfunc_ids_enqueue_dispatch)
  5300. BTF_ID_FLAGS(func, scx_bpf_dispatch, KF_RCU)
  5301. BTF_ID_FLAGS(func, scx_bpf_dispatch_vtime, KF_RCU)
  5302. BTF_KFUNCS_END(scx_kfunc_ids_enqueue_dispatch)
  5303. static const struct btf_kfunc_id_set scx_kfunc_set_enqueue_dispatch = {
  5304. .owner = THIS_MODULE,
  5305. .set = &scx_kfunc_ids_enqueue_dispatch,
  5306. };
  5307. static bool scx_dispatch_from_dsq(struct bpf_iter_scx_dsq_kern *kit,
  5308. struct task_struct *p, u64 dsq_id,
  5309. u64 enq_flags)
  5310. {
  5311. struct scx_dispatch_q *src_dsq = kit->dsq, *dst_dsq;
  5312. struct rq *this_rq, *src_rq, *locked_rq;
  5313. bool dispatched = false;
  5314. bool in_balance;
  5315. unsigned long flags;
  5316. if (!scx_kf_allowed_if_unlocked() && !scx_kf_allowed(SCX_KF_DISPATCH))
  5317. return false;
  5318. /*
  5319. * Can be called from either ops.dispatch() locking this_rq() or any
  5320. * context where no rq lock is held. If latter, lock @p's task_rq which
  5321. * we'll likely need anyway.
  5322. */
  5323. src_rq = task_rq(p);
  5324. local_irq_save(flags);
  5325. this_rq = this_rq();
  5326. in_balance = this_rq->scx.flags & SCX_RQ_IN_BALANCE;
  5327. if (in_balance) {
  5328. if (this_rq != src_rq) {
  5329. raw_spin_rq_unlock(this_rq);
  5330. raw_spin_rq_lock(src_rq);
  5331. }
  5332. } else {
  5333. raw_spin_rq_lock(src_rq);
  5334. }
  5335. locked_rq = src_rq;
  5336. raw_spin_lock(&src_dsq->lock);
  5337. /*
  5338. * Did someone else get to it? @p could have already left $src_dsq, got
  5339. * re-enqueud, or be in the process of being consumed by someone else.
  5340. */
  5341. if (unlikely(p->scx.dsq != src_dsq ||
  5342. u32_before(kit->cursor.priv, p->scx.dsq_seq) ||
  5343. p->scx.holding_cpu >= 0) ||
  5344. WARN_ON_ONCE(src_rq != task_rq(p))) {
  5345. raw_spin_unlock(&src_dsq->lock);
  5346. goto out;
  5347. }
  5348. /* @p is still on $src_dsq and stable, determine the destination */
  5349. dst_dsq = find_dsq_for_dispatch(this_rq, dsq_id, p);
  5350. /*
  5351. * Apply vtime and slice updates before moving so that the new time is
  5352. * visible before inserting into $dst_dsq. @p is still on $src_dsq but
  5353. * this is safe as we're locking it.
  5354. */
  5355. if (kit->cursor.flags & __SCX_DSQ_ITER_HAS_VTIME)
  5356. p->scx.dsq_vtime = kit->vtime;
  5357. if (kit->cursor.flags & __SCX_DSQ_ITER_HAS_SLICE)
  5358. p->scx.slice = kit->slice;
  5359. /* execute move */
  5360. locked_rq = move_task_between_dsqs(p, enq_flags, src_dsq, dst_dsq);
  5361. dispatched = true;
  5362. out:
  5363. if (in_balance) {
  5364. if (this_rq != locked_rq) {
  5365. raw_spin_rq_unlock(locked_rq);
  5366. raw_spin_rq_lock(this_rq);
  5367. }
  5368. } else {
  5369. raw_spin_rq_unlock_irqrestore(locked_rq, flags);
  5370. }
  5371. kit->cursor.flags &= ~(__SCX_DSQ_ITER_HAS_SLICE |
  5372. __SCX_DSQ_ITER_HAS_VTIME);
  5373. return dispatched;
  5374. }
  5375. __bpf_kfunc_start_defs();
  5376. /**
  5377. * scx_bpf_dispatch_nr_slots - Return the number of remaining dispatch slots
  5378. *
  5379. * Can only be called from ops.dispatch().
  5380. */
  5381. __bpf_kfunc u32 scx_bpf_dispatch_nr_slots(void)
  5382. {
  5383. if (!scx_kf_allowed(SCX_KF_DISPATCH))
  5384. return 0;
  5385. return scx_dsp_max_batch - __this_cpu_read(scx_dsp_ctx->cursor);
  5386. }
  5387. /**
  5388. * scx_bpf_dispatch_cancel - Cancel the latest dispatch
  5389. *
  5390. * Cancel the latest dispatch. Can be called multiple times to cancel further
  5391. * dispatches. Can only be called from ops.dispatch().
  5392. */
  5393. __bpf_kfunc void scx_bpf_dispatch_cancel(void)
  5394. {
  5395. struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
  5396. if (!scx_kf_allowed(SCX_KF_DISPATCH))
  5397. return;
  5398. if (dspc->cursor > 0)
  5399. dspc->cursor--;
  5400. else
  5401. scx_ops_error("dispatch buffer underflow");
  5402. }
  5403. /**
  5404. * scx_bpf_consume - Transfer a task from a DSQ to the current CPU's local DSQ
  5405. * @dsq_id: DSQ to consume
  5406. *
  5407. * Consume a task from the non-local DSQ identified by @dsq_id and transfer it
  5408. * to the current CPU's local DSQ for execution. Can only be called from
  5409. * ops.dispatch().
  5410. *
  5411. * This function flushes the in-flight dispatches from scx_bpf_dispatch() before
  5412. * trying to consume the specified DSQ. It may also grab rq locks and thus can't
  5413. * be called under any BPF locks.
  5414. *
  5415. * Returns %true if a task has been consumed, %false if there isn't any task to
  5416. * consume.
  5417. */
  5418. __bpf_kfunc bool scx_bpf_consume(u64 dsq_id)
  5419. {
  5420. struct scx_dsp_ctx *dspc = this_cpu_ptr(scx_dsp_ctx);
  5421. struct scx_dispatch_q *dsq;
  5422. if (!scx_kf_allowed(SCX_KF_DISPATCH))
  5423. return false;
  5424. flush_dispatch_buf(dspc->rq);
  5425. dsq = find_user_dsq(dsq_id);
  5426. if (unlikely(!dsq)) {
  5427. scx_ops_error("invalid DSQ ID 0x%016llx", dsq_id);
  5428. return false;
  5429. }
  5430. if (consume_dispatch_q(dspc->rq, dsq)) {
  5431. /*
  5432. * A successfully consumed task can be dequeued before it starts
  5433. * running while the CPU is trying to migrate other dispatched
  5434. * tasks. Bump nr_tasks to tell balance_scx() to retry on empty
  5435. * local DSQ.
  5436. */
  5437. dspc->nr_tasks++;
  5438. return true;
  5439. } else {
  5440. return false;
  5441. }
  5442. }
  5443. /**
  5444. * scx_bpf_dispatch_from_dsq_set_slice - Override slice when dispatching from DSQ
  5445. * @it__iter: DSQ iterator in progress
  5446. * @slice: duration the dispatched task can run for in nsecs
  5447. *
  5448. * Override the slice of the next task that will be dispatched from @it__iter
  5449. * using scx_bpf_dispatch_from_dsq[_vtime](). If this function is not called,
  5450. * the previous slice duration is kept.
  5451. */
  5452. __bpf_kfunc void scx_bpf_dispatch_from_dsq_set_slice(
  5453. struct bpf_iter_scx_dsq *it__iter, u64 slice)
  5454. {
  5455. struct bpf_iter_scx_dsq_kern *kit = (void *)it__iter;
  5456. kit->slice = slice;
  5457. kit->cursor.flags |= __SCX_DSQ_ITER_HAS_SLICE;
  5458. }
  5459. /**
  5460. * scx_bpf_dispatch_from_dsq_set_vtime - Override vtime when dispatching from DSQ
  5461. * @it__iter: DSQ iterator in progress
  5462. * @vtime: task's ordering inside the vtime-sorted queue of the target DSQ
  5463. *
  5464. * Override the vtime of the next task that will be dispatched from @it__iter
  5465. * using scx_bpf_dispatch_from_dsq_vtime(). If this function is not called, the
  5466. * previous slice vtime is kept. If scx_bpf_dispatch_from_dsq() is used to
  5467. * dispatch the next task, the override is ignored and cleared.
  5468. */
  5469. __bpf_kfunc void scx_bpf_dispatch_from_dsq_set_vtime(
  5470. struct bpf_iter_scx_dsq *it__iter, u64 vtime)
  5471. {
  5472. struct bpf_iter_scx_dsq_kern *kit = (void *)it__iter;
  5473. kit->vtime = vtime;
  5474. kit->cursor.flags |= __SCX_DSQ_ITER_HAS_VTIME;
  5475. }
  5476. /**
  5477. * scx_bpf_dispatch_from_dsq - Move a task from DSQ iteration to a DSQ
  5478. * @it__iter: DSQ iterator in progress
  5479. * @p: task to transfer
  5480. * @dsq_id: DSQ to move @p to
  5481. * @enq_flags: SCX_ENQ_*
  5482. *
  5483. * Transfer @p which is on the DSQ currently iterated by @it__iter to the DSQ
  5484. * specified by @dsq_id. All DSQs - local DSQs, global DSQ and user DSQs - can
  5485. * be the destination.
  5486. *
  5487. * For the transfer to be successful, @p must still be on the DSQ and have been
  5488. * queued before the DSQ iteration started. This function doesn't care whether
  5489. * @p was obtained from the DSQ iteration. @p just has to be on the DSQ and have
  5490. * been queued before the iteration started.
  5491. *
  5492. * @p's slice is kept by default. Use scx_bpf_dispatch_from_dsq_set_slice() to
  5493. * update.
  5494. *
  5495. * Can be called from ops.dispatch() or any BPF context which doesn't hold a rq
  5496. * lock (e.g. BPF timers or SYSCALL programs).
  5497. *
  5498. * Returns %true if @p has been consumed, %false if @p had already been consumed
  5499. * or dequeued.
  5500. */
  5501. __bpf_kfunc bool scx_bpf_dispatch_from_dsq(struct bpf_iter_scx_dsq *it__iter,
  5502. struct task_struct *p, u64 dsq_id,
  5503. u64 enq_flags)
  5504. {
  5505. return scx_dispatch_from_dsq((struct bpf_iter_scx_dsq_kern *)it__iter,
  5506. p, dsq_id, enq_flags);
  5507. }
  5508. /**
  5509. * scx_bpf_dispatch_vtime_from_dsq - Move a task from DSQ iteration to a PRIQ DSQ
  5510. * @it__iter: DSQ iterator in progress
  5511. * @p: task to transfer
  5512. * @dsq_id: DSQ to move @p to
  5513. * @enq_flags: SCX_ENQ_*
  5514. *
  5515. * Transfer @p which is on the DSQ currently iterated by @it__iter to the
  5516. * priority queue of the DSQ specified by @dsq_id. The destination must be a
  5517. * user DSQ as only user DSQs support priority queue.
  5518. *
  5519. * @p's slice and vtime are kept by default. Use
  5520. * scx_bpf_dispatch_from_dsq_set_slice() and
  5521. * scx_bpf_dispatch_from_dsq_set_vtime() to update.
  5522. *
  5523. * All other aspects are identical to scx_bpf_dispatch_from_dsq(). See
  5524. * scx_bpf_dispatch_vtime() for more information on @vtime.
  5525. */
  5526. __bpf_kfunc bool scx_bpf_dispatch_vtime_from_dsq(struct bpf_iter_scx_dsq *it__iter,
  5527. struct task_struct *p, u64 dsq_id,
  5528. u64 enq_flags)
  5529. {
  5530. return scx_dispatch_from_dsq((struct bpf_iter_scx_dsq_kern *)it__iter,
  5531. p, dsq_id, enq_flags | SCX_ENQ_DSQ_PRIQ);
  5532. }
  5533. __bpf_kfunc_end_defs();
  5534. BTF_KFUNCS_START(scx_kfunc_ids_dispatch)
  5535. BTF_ID_FLAGS(func, scx_bpf_dispatch_nr_slots)
  5536. BTF_ID_FLAGS(func, scx_bpf_dispatch_cancel)
  5537. BTF_ID_FLAGS(func, scx_bpf_consume)
  5538. BTF_ID_FLAGS(func, scx_bpf_dispatch_from_dsq_set_slice)
  5539. BTF_ID_FLAGS(func, scx_bpf_dispatch_from_dsq_set_vtime)
  5540. BTF_ID_FLAGS(func, scx_bpf_dispatch_from_dsq, KF_RCU)
  5541. BTF_ID_FLAGS(func, scx_bpf_dispatch_vtime_from_dsq, KF_RCU)
  5542. BTF_KFUNCS_END(scx_kfunc_ids_dispatch)
  5543. static const struct btf_kfunc_id_set scx_kfunc_set_dispatch = {
  5544. .owner = THIS_MODULE,
  5545. .set = &scx_kfunc_ids_dispatch,
  5546. };
  5547. __bpf_kfunc_start_defs();
  5548. /**
  5549. * scx_bpf_reenqueue_local - Re-enqueue tasks on a local DSQ
  5550. *
  5551. * Iterate over all of the tasks currently enqueued on the local DSQ of the
  5552. * caller's CPU, and re-enqueue them in the BPF scheduler. Returns the number of
  5553. * processed tasks. Can only be called from ops.cpu_release().
  5554. */
  5555. __bpf_kfunc u32 scx_bpf_reenqueue_local(void)
  5556. {
  5557. LIST_HEAD(tasks);
  5558. u32 nr_enqueued = 0;
  5559. struct rq *rq;
  5560. struct task_struct *p, *n;
  5561. if (!scx_kf_allowed(SCX_KF_CPU_RELEASE))
  5562. return 0;
  5563. rq = cpu_rq(smp_processor_id());
  5564. lockdep_assert_rq_held(rq);
  5565. /*
  5566. * The BPF scheduler may choose to dispatch tasks back to
  5567. * @rq->scx.local_dsq. Move all candidate tasks off to a private list
  5568. * first to avoid processing the same tasks repeatedly.
  5569. */
  5570. list_for_each_entry_safe(p, n, &rq->scx.local_dsq.list,
  5571. scx.dsq_list.node) {
  5572. /*
  5573. * If @p is being migrated, @p's current CPU may not agree with
  5574. * its allowed CPUs and the migration_cpu_stop is about to
  5575. * deactivate and re-activate @p anyway. Skip re-enqueueing.
  5576. *
  5577. * While racing sched property changes may also dequeue and
  5578. * re-enqueue a migrating task while its current CPU and allowed
  5579. * CPUs disagree, they use %ENQUEUE_RESTORE which is bypassed to
  5580. * the current local DSQ for running tasks and thus are not
  5581. * visible to the BPF scheduler.
  5582. */
  5583. if (p->migration_pending)
  5584. continue;
  5585. dispatch_dequeue(rq, p);
  5586. list_add_tail(&p->scx.dsq_list.node, &tasks);
  5587. }
  5588. list_for_each_entry_safe(p, n, &tasks, scx.dsq_list.node) {
  5589. list_del_init(&p->scx.dsq_list.node);
  5590. do_enqueue_task(rq, p, SCX_ENQ_REENQ, -1);
  5591. nr_enqueued++;
  5592. }
  5593. return nr_enqueued;
  5594. }
  5595. __bpf_kfunc_end_defs();
  5596. BTF_KFUNCS_START(scx_kfunc_ids_cpu_release)
  5597. BTF_ID_FLAGS(func, scx_bpf_reenqueue_local)
  5598. BTF_KFUNCS_END(scx_kfunc_ids_cpu_release)
  5599. static const struct btf_kfunc_id_set scx_kfunc_set_cpu_release = {
  5600. .owner = THIS_MODULE,
  5601. .set = &scx_kfunc_ids_cpu_release,
  5602. };
  5603. __bpf_kfunc_start_defs();
  5604. /**
  5605. * scx_bpf_create_dsq - Create a custom DSQ
  5606. * @dsq_id: DSQ to create
  5607. * @node: NUMA node to allocate from
  5608. *
  5609. * Create a custom DSQ identified by @dsq_id. Can be called from any sleepable
  5610. * scx callback, and any BPF_PROG_TYPE_SYSCALL prog.
  5611. */
  5612. __bpf_kfunc s32 scx_bpf_create_dsq(u64 dsq_id, s32 node)
  5613. {
  5614. if (unlikely(node >= (int)nr_node_ids ||
  5615. (node < 0 && node != NUMA_NO_NODE)))
  5616. return -EINVAL;
  5617. return PTR_ERR_OR_ZERO(create_dsq(dsq_id, node));
  5618. }
  5619. __bpf_kfunc_end_defs();
  5620. BTF_KFUNCS_START(scx_kfunc_ids_unlocked)
  5621. BTF_ID_FLAGS(func, scx_bpf_create_dsq, KF_SLEEPABLE)
  5622. BTF_ID_FLAGS(func, scx_bpf_dispatch_from_dsq_set_slice)
  5623. BTF_ID_FLAGS(func, scx_bpf_dispatch_from_dsq_set_vtime)
  5624. BTF_ID_FLAGS(func, scx_bpf_dispatch_from_dsq, KF_RCU)
  5625. BTF_ID_FLAGS(func, scx_bpf_dispatch_vtime_from_dsq, KF_RCU)
  5626. BTF_KFUNCS_END(scx_kfunc_ids_unlocked)
  5627. static const struct btf_kfunc_id_set scx_kfunc_set_unlocked = {
  5628. .owner = THIS_MODULE,
  5629. .set = &scx_kfunc_ids_unlocked,
  5630. };
  5631. __bpf_kfunc_start_defs();
  5632. /**
  5633. * scx_bpf_kick_cpu - Trigger reschedule on a CPU
  5634. * @cpu: cpu to kick
  5635. * @flags: %SCX_KICK_* flags
  5636. *
  5637. * Kick @cpu into rescheduling. This can be used to wake up an idle CPU or
  5638. * trigger rescheduling on a busy CPU. This can be called from any online
  5639. * scx_ops operation and the actual kicking is performed asynchronously through
  5640. * an irq work.
  5641. */
  5642. __bpf_kfunc void scx_bpf_kick_cpu(s32 cpu, u64 flags)
  5643. {
  5644. struct rq *this_rq;
  5645. unsigned long irq_flags;
  5646. if (!ops_cpu_valid(cpu, NULL))
  5647. return;
  5648. local_irq_save(irq_flags);
  5649. this_rq = this_rq();
  5650. /*
  5651. * While bypassing for PM ops, IRQ handling may not be online which can
  5652. * lead to irq_work_queue() malfunction such as infinite busy wait for
  5653. * IRQ status update. Suppress kicking.
  5654. */
  5655. if (scx_rq_bypassing(this_rq))
  5656. goto out;
  5657. /*
  5658. * Actual kicking is bounced to kick_cpus_irq_workfn() to avoid nesting
  5659. * rq locks. We can probably be smarter and avoid bouncing if called
  5660. * from ops which don't hold a rq lock.
  5661. */
  5662. if (flags & SCX_KICK_IDLE) {
  5663. struct rq *target_rq = cpu_rq(cpu);
  5664. if (unlikely(flags & (SCX_KICK_PREEMPT | SCX_KICK_WAIT)))
  5665. scx_ops_error("PREEMPT/WAIT cannot be used with SCX_KICK_IDLE");
  5666. if (raw_spin_rq_trylock(target_rq)) {
  5667. if (can_skip_idle_kick(target_rq)) {
  5668. raw_spin_rq_unlock(target_rq);
  5669. goto out;
  5670. }
  5671. raw_spin_rq_unlock(target_rq);
  5672. }
  5673. cpumask_set_cpu(cpu, this_rq->scx.cpus_to_kick_if_idle);
  5674. } else {
  5675. cpumask_set_cpu(cpu, this_rq->scx.cpus_to_kick);
  5676. if (flags & SCX_KICK_PREEMPT)
  5677. cpumask_set_cpu(cpu, this_rq->scx.cpus_to_preempt);
  5678. if (flags & SCX_KICK_WAIT)
  5679. cpumask_set_cpu(cpu, this_rq->scx.cpus_to_wait);
  5680. }
  5681. irq_work_queue(&this_rq->scx.kick_cpus_irq_work);
  5682. out:
  5683. local_irq_restore(irq_flags);
  5684. }
  5685. /**
  5686. * scx_bpf_dsq_nr_queued - Return the number of queued tasks
  5687. * @dsq_id: id of the DSQ
  5688. *
  5689. * Return the number of tasks in the DSQ matching @dsq_id. If not found,
  5690. * -%ENOENT is returned.
  5691. */
  5692. __bpf_kfunc s32 scx_bpf_dsq_nr_queued(u64 dsq_id)
  5693. {
  5694. struct scx_dispatch_q *dsq;
  5695. s32 ret;
  5696. preempt_disable();
  5697. if (dsq_id == SCX_DSQ_LOCAL) {
  5698. ret = READ_ONCE(this_rq()->scx.local_dsq.nr);
  5699. goto out;
  5700. } else if ((dsq_id & SCX_DSQ_LOCAL_ON) == SCX_DSQ_LOCAL_ON) {
  5701. s32 cpu = dsq_id & SCX_DSQ_LOCAL_CPU_MASK;
  5702. if (ops_cpu_valid(cpu, NULL)) {
  5703. ret = READ_ONCE(cpu_rq(cpu)->scx.local_dsq.nr);
  5704. goto out;
  5705. }
  5706. } else {
  5707. dsq = find_user_dsq(dsq_id);
  5708. if (dsq) {
  5709. ret = READ_ONCE(dsq->nr);
  5710. goto out;
  5711. }
  5712. }
  5713. ret = -ENOENT;
  5714. out:
  5715. preempt_enable();
  5716. return ret;
  5717. }
  5718. /**
  5719. * scx_bpf_destroy_dsq - Destroy a custom DSQ
  5720. * @dsq_id: DSQ to destroy
  5721. *
  5722. * Destroy the custom DSQ identified by @dsq_id. Only DSQs created with
  5723. * scx_bpf_create_dsq() can be destroyed. The caller must ensure that the DSQ is
  5724. * empty and no further tasks are dispatched to it. Ignored if called on a DSQ
  5725. * which doesn't exist. Can be called from any online scx_ops operations.
  5726. */
  5727. __bpf_kfunc void scx_bpf_destroy_dsq(u64 dsq_id)
  5728. {
  5729. destroy_dsq(dsq_id);
  5730. }
  5731. /**
  5732. * bpf_iter_scx_dsq_new - Create a DSQ iterator
  5733. * @it: iterator to initialize
  5734. * @dsq_id: DSQ to iterate
  5735. * @flags: %SCX_DSQ_ITER_*
  5736. *
  5737. * Initialize BPF iterator @it which can be used with bpf_for_each() to walk
  5738. * tasks in the DSQ specified by @dsq_id. Iteration using @it only includes
  5739. * tasks which are already queued when this function is invoked.
  5740. */
  5741. __bpf_kfunc int bpf_iter_scx_dsq_new(struct bpf_iter_scx_dsq *it, u64 dsq_id,
  5742. u64 flags)
  5743. {
  5744. struct bpf_iter_scx_dsq_kern *kit = (void *)it;
  5745. BUILD_BUG_ON(sizeof(struct bpf_iter_scx_dsq_kern) >
  5746. sizeof(struct bpf_iter_scx_dsq));
  5747. BUILD_BUG_ON(__alignof__(struct bpf_iter_scx_dsq_kern) !=
  5748. __alignof__(struct bpf_iter_scx_dsq));
  5749. /*
  5750. * next() and destroy() will be called regardless of the return value.
  5751. * Always clear $kit->dsq.
  5752. */
  5753. kit->dsq = NULL;
  5754. if (flags & ~__SCX_DSQ_ITER_USER_FLAGS)
  5755. return -EINVAL;
  5756. kit->dsq = find_user_dsq(dsq_id);
  5757. if (!kit->dsq)
  5758. return -ENOENT;
  5759. INIT_LIST_HEAD(&kit->cursor.node);
  5760. kit->cursor.flags = SCX_DSQ_LNODE_ITER_CURSOR | flags;
  5761. kit->cursor.priv = READ_ONCE(kit->dsq->seq);
  5762. return 0;
  5763. }
  5764. /**
  5765. * bpf_iter_scx_dsq_next - Progress a DSQ iterator
  5766. * @it: iterator to progress
  5767. *
  5768. * Return the next task. See bpf_iter_scx_dsq_new().
  5769. */
  5770. __bpf_kfunc struct task_struct *bpf_iter_scx_dsq_next(struct bpf_iter_scx_dsq *it)
  5771. {
  5772. struct bpf_iter_scx_dsq_kern *kit = (void *)it;
  5773. bool rev = kit->cursor.flags & SCX_DSQ_ITER_REV;
  5774. struct task_struct *p;
  5775. unsigned long flags;
  5776. if (!kit->dsq)
  5777. return NULL;
  5778. raw_spin_lock_irqsave(&kit->dsq->lock, flags);
  5779. if (list_empty(&kit->cursor.node))
  5780. p = NULL;
  5781. else
  5782. p = container_of(&kit->cursor, struct task_struct, scx.dsq_list);
  5783. /*
  5784. * Only tasks which were queued before the iteration started are
  5785. * visible. This bounds BPF iterations and guarantees that vtime never
  5786. * jumps in the other direction while iterating.
  5787. */
  5788. do {
  5789. p = nldsq_next_task(kit->dsq, p, rev);
  5790. } while (p && unlikely(u32_before(kit->cursor.priv, p->scx.dsq_seq)));
  5791. if (p) {
  5792. if (rev)
  5793. list_move_tail(&kit->cursor.node, &p->scx.dsq_list.node);
  5794. else
  5795. list_move(&kit->cursor.node, &p->scx.dsq_list.node);
  5796. } else {
  5797. list_del_init(&kit->cursor.node);
  5798. }
  5799. raw_spin_unlock_irqrestore(&kit->dsq->lock, flags);
  5800. return p;
  5801. }
  5802. /**
  5803. * bpf_iter_scx_dsq_destroy - Destroy a DSQ iterator
  5804. * @it: iterator to destroy
  5805. *
  5806. * Undo scx_iter_scx_dsq_new().
  5807. */
  5808. __bpf_kfunc void bpf_iter_scx_dsq_destroy(struct bpf_iter_scx_dsq *it)
  5809. {
  5810. struct bpf_iter_scx_dsq_kern *kit = (void *)it;
  5811. if (!kit->dsq)
  5812. return;
  5813. if (!list_empty(&kit->cursor.node)) {
  5814. unsigned long flags;
  5815. raw_spin_lock_irqsave(&kit->dsq->lock, flags);
  5816. list_del_init(&kit->cursor.node);
  5817. raw_spin_unlock_irqrestore(&kit->dsq->lock, flags);
  5818. }
  5819. kit->dsq = NULL;
  5820. }
  5821. __bpf_kfunc_end_defs();
  5822. static s32 __bstr_format(u64 *data_buf, char *line_buf, size_t line_size,
  5823. char *fmt, unsigned long long *data, u32 data__sz)
  5824. {
  5825. struct bpf_bprintf_data bprintf_data = { .get_bin_args = true };
  5826. s32 ret;
  5827. if (data__sz % 8 || data__sz > MAX_BPRINTF_VARARGS * 8 ||
  5828. (data__sz && !data)) {
  5829. scx_ops_error("invalid data=%p and data__sz=%u",
  5830. (void *)data, data__sz);
  5831. return -EINVAL;
  5832. }
  5833. ret = copy_from_kernel_nofault(data_buf, data, data__sz);
  5834. if (ret < 0) {
  5835. scx_ops_error("failed to read data fields (%d)", ret);
  5836. return ret;
  5837. }
  5838. ret = bpf_bprintf_prepare(fmt, UINT_MAX, data_buf, data__sz / 8,
  5839. &bprintf_data);
  5840. if (ret < 0) {
  5841. scx_ops_error("format preparation failed (%d)", ret);
  5842. return ret;
  5843. }
  5844. ret = bstr_printf(line_buf, line_size, fmt,
  5845. bprintf_data.bin_args);
  5846. bpf_bprintf_cleanup(&bprintf_data);
  5847. if (ret < 0) {
  5848. scx_ops_error("(\"%s\", %p, %u) failed to format",
  5849. fmt, data, data__sz);
  5850. return ret;
  5851. }
  5852. return ret;
  5853. }
  5854. static s32 bstr_format(struct scx_bstr_buf *buf,
  5855. char *fmt, unsigned long long *data, u32 data__sz)
  5856. {
  5857. return __bstr_format(buf->data, buf->line, sizeof(buf->line),
  5858. fmt, data, data__sz);
  5859. }
  5860. __bpf_kfunc_start_defs();
  5861. /**
  5862. * scx_bpf_exit_bstr - Gracefully exit the BPF scheduler.
  5863. * @exit_code: Exit value to pass to user space via struct scx_exit_info.
  5864. * @fmt: error message format string
  5865. * @data: format string parameters packaged using ___bpf_fill() macro
  5866. * @data__sz: @data len, must end in '__sz' for the verifier
  5867. *
  5868. * Indicate that the BPF scheduler wants to exit gracefully, and initiate ops
  5869. * disabling.
  5870. */
  5871. __bpf_kfunc void scx_bpf_exit_bstr(s64 exit_code, char *fmt,
  5872. unsigned long long *data, u32 data__sz)
  5873. {
  5874. unsigned long flags;
  5875. raw_spin_lock_irqsave(&scx_exit_bstr_buf_lock, flags);
  5876. if (bstr_format(&scx_exit_bstr_buf, fmt, data, data__sz) >= 0)
  5877. scx_ops_exit_kind(SCX_EXIT_UNREG_BPF, exit_code, "%s",
  5878. scx_exit_bstr_buf.line);
  5879. raw_spin_unlock_irqrestore(&scx_exit_bstr_buf_lock, flags);
  5880. }
  5881. /**
  5882. * scx_bpf_error_bstr - Indicate fatal error
  5883. * @fmt: error message format string
  5884. * @data: format string parameters packaged using ___bpf_fill() macro
  5885. * @data__sz: @data len, must end in '__sz' for the verifier
  5886. *
  5887. * Indicate that the BPF scheduler encountered a fatal error and initiate ops
  5888. * disabling.
  5889. */
  5890. __bpf_kfunc void scx_bpf_error_bstr(char *fmt, unsigned long long *data,
  5891. u32 data__sz)
  5892. {
  5893. unsigned long flags;
  5894. raw_spin_lock_irqsave(&scx_exit_bstr_buf_lock, flags);
  5895. if (bstr_format(&scx_exit_bstr_buf, fmt, data, data__sz) >= 0)
  5896. scx_ops_exit_kind(SCX_EXIT_ERROR_BPF, 0, "%s",
  5897. scx_exit_bstr_buf.line);
  5898. raw_spin_unlock_irqrestore(&scx_exit_bstr_buf_lock, flags);
  5899. }
  5900. /**
  5901. * scx_bpf_dump - Generate extra debug dump specific to the BPF scheduler
  5902. * @fmt: format string
  5903. * @data: format string parameters packaged using ___bpf_fill() macro
  5904. * @data__sz: @data len, must end in '__sz' for the verifier
  5905. *
  5906. * To be called through scx_bpf_dump() helper from ops.dump(), dump_cpu() and
  5907. * dump_task() to generate extra debug dump specific to the BPF scheduler.
  5908. *
  5909. * The extra dump may be multiple lines. A single line may be split over
  5910. * multiple calls. The last line is automatically terminated.
  5911. */
  5912. __bpf_kfunc void scx_bpf_dump_bstr(char *fmt, unsigned long long *data,
  5913. u32 data__sz)
  5914. {
  5915. struct scx_dump_data *dd = &scx_dump_data;
  5916. struct scx_bstr_buf *buf = &dd->buf;
  5917. s32 ret;
  5918. if (raw_smp_processor_id() != dd->cpu) {
  5919. scx_ops_error("scx_bpf_dump() must only be called from ops.dump() and friends");
  5920. return;
  5921. }
  5922. /* append the formatted string to the line buf */
  5923. ret = __bstr_format(buf->data, buf->line + dd->cursor,
  5924. sizeof(buf->line) - dd->cursor, fmt, data, data__sz);
  5925. if (ret < 0) {
  5926. dump_line(dd->s, "%s[!] (\"%s\", %p, %u) failed to format (%d)",
  5927. dd->prefix, fmt, data, data__sz, ret);
  5928. return;
  5929. }
  5930. dd->cursor += ret;
  5931. dd->cursor = min_t(s32, dd->cursor, sizeof(buf->line));
  5932. if (!dd->cursor)
  5933. return;
  5934. /*
  5935. * If the line buf overflowed or ends in a newline, flush it into the
  5936. * dump. This is to allow the caller to generate a single line over
  5937. * multiple calls. As ops_dump_flush() can also handle multiple lines in
  5938. * the line buf, the only case which can lead to an unexpected
  5939. * truncation is when the caller keeps generating newlines in the middle
  5940. * instead of the end consecutively. Don't do that.
  5941. */
  5942. if (dd->cursor >= sizeof(buf->line) || buf->line[dd->cursor - 1] == '\n')
  5943. ops_dump_flush();
  5944. }
  5945. /**
  5946. * scx_bpf_cpuperf_cap - Query the maximum relative capacity of a CPU
  5947. * @cpu: CPU of interest
  5948. *
  5949. * Return the maximum relative capacity of @cpu in relation to the most
  5950. * performant CPU in the system. The return value is in the range [1,
  5951. * %SCX_CPUPERF_ONE]. See scx_bpf_cpuperf_cur().
  5952. */
  5953. __bpf_kfunc u32 scx_bpf_cpuperf_cap(s32 cpu)
  5954. {
  5955. if (ops_cpu_valid(cpu, NULL))
  5956. return arch_scale_cpu_capacity(cpu);
  5957. else
  5958. return SCX_CPUPERF_ONE;
  5959. }
  5960. /**
  5961. * scx_bpf_cpuperf_cur - Query the current relative performance of a CPU
  5962. * @cpu: CPU of interest
  5963. *
  5964. * Return the current relative performance of @cpu in relation to its maximum.
  5965. * The return value is in the range [1, %SCX_CPUPERF_ONE].
  5966. *
  5967. * The current performance level of a CPU in relation to the maximum performance
  5968. * available in the system can be calculated as follows:
  5969. *
  5970. * scx_bpf_cpuperf_cap() * scx_bpf_cpuperf_cur() / %SCX_CPUPERF_ONE
  5971. *
  5972. * The result is in the range [1, %SCX_CPUPERF_ONE].
  5973. */
  5974. __bpf_kfunc u32 scx_bpf_cpuperf_cur(s32 cpu)
  5975. {
  5976. if (ops_cpu_valid(cpu, NULL))
  5977. return arch_scale_freq_capacity(cpu);
  5978. else
  5979. return SCX_CPUPERF_ONE;
  5980. }
  5981. /**
  5982. * scx_bpf_cpuperf_set - Set the relative performance target of a CPU
  5983. * @cpu: CPU of interest
  5984. * @perf: target performance level [0, %SCX_CPUPERF_ONE]
  5985. * @flags: %SCX_CPUPERF_* flags
  5986. *
  5987. * Set the target performance level of @cpu to @perf. @perf is in linear
  5988. * relative scale between 0 and %SCX_CPUPERF_ONE. This determines how the
  5989. * schedutil cpufreq governor chooses the target frequency.
  5990. *
  5991. * The actual performance level chosen, CPU grouping, and the overhead and
  5992. * latency of the operations are dependent on the hardware and cpufreq driver in
  5993. * use. Consult hardware and cpufreq documentation for more information. The
  5994. * current performance level can be monitored using scx_bpf_cpuperf_cur().
  5995. */
  5996. __bpf_kfunc void scx_bpf_cpuperf_set(s32 cpu, u32 perf)
  5997. {
  5998. if (unlikely(perf > SCX_CPUPERF_ONE)) {
  5999. scx_ops_error("Invalid cpuperf target %u for CPU %d", perf, cpu);
  6000. return;
  6001. }
  6002. if (ops_cpu_valid(cpu, NULL)) {
  6003. struct rq *rq = cpu_rq(cpu);
  6004. rq->scx.cpuperf_target = perf;
  6005. rcu_read_lock_sched_notrace();
  6006. cpufreq_update_util(cpu_rq(cpu), 0);
  6007. rcu_read_unlock_sched_notrace();
  6008. }
  6009. }
  6010. /**
  6011. * scx_bpf_nr_cpu_ids - Return the number of possible CPU IDs
  6012. *
  6013. * All valid CPU IDs in the system are smaller than the returned value.
  6014. */
  6015. __bpf_kfunc u32 scx_bpf_nr_cpu_ids(void)
  6016. {
  6017. return nr_cpu_ids;
  6018. }
  6019. /**
  6020. * scx_bpf_get_possible_cpumask - Get a referenced kptr to cpu_possible_mask
  6021. */
  6022. __bpf_kfunc const struct cpumask *scx_bpf_get_possible_cpumask(void)
  6023. {
  6024. return cpu_possible_mask;
  6025. }
  6026. /**
  6027. * scx_bpf_get_online_cpumask - Get a referenced kptr to cpu_online_mask
  6028. */
  6029. __bpf_kfunc const struct cpumask *scx_bpf_get_online_cpumask(void)
  6030. {
  6031. return cpu_online_mask;
  6032. }
  6033. /**
  6034. * scx_bpf_put_cpumask - Release a possible/online cpumask
  6035. * @cpumask: cpumask to release
  6036. */
  6037. __bpf_kfunc void scx_bpf_put_cpumask(const struct cpumask *cpumask)
  6038. {
  6039. /*
  6040. * Empty function body because we aren't actually acquiring or releasing
  6041. * a reference to a global cpumask, which is read-only in the caller and
  6042. * is never released. The acquire / release semantics here are just used
  6043. * to make the cpumask is a trusted pointer in the caller.
  6044. */
  6045. }
  6046. /**
  6047. * scx_bpf_get_idle_cpumask - Get a referenced kptr to the idle-tracking
  6048. * per-CPU cpumask.
  6049. *
  6050. * Returns NULL if idle tracking is not enabled, or running on a UP kernel.
  6051. */
  6052. __bpf_kfunc const struct cpumask *scx_bpf_get_idle_cpumask(void)
  6053. {
  6054. if (!static_branch_likely(&scx_builtin_idle_enabled)) {
  6055. scx_ops_error("built-in idle tracking is disabled");
  6056. return cpu_none_mask;
  6057. }
  6058. #ifdef CONFIG_SMP
  6059. return idle_masks.cpu;
  6060. #else
  6061. return cpu_none_mask;
  6062. #endif
  6063. }
  6064. /**
  6065. * scx_bpf_get_idle_smtmask - Get a referenced kptr to the idle-tracking,
  6066. * per-physical-core cpumask. Can be used to determine if an entire physical
  6067. * core is free.
  6068. *
  6069. * Returns NULL if idle tracking is not enabled, or running on a UP kernel.
  6070. */
  6071. __bpf_kfunc const struct cpumask *scx_bpf_get_idle_smtmask(void)
  6072. {
  6073. if (!static_branch_likely(&scx_builtin_idle_enabled)) {
  6074. scx_ops_error("built-in idle tracking is disabled");
  6075. return cpu_none_mask;
  6076. }
  6077. #ifdef CONFIG_SMP
  6078. if (sched_smt_active())
  6079. return idle_masks.smt;
  6080. else
  6081. return idle_masks.cpu;
  6082. #else
  6083. return cpu_none_mask;
  6084. #endif
  6085. }
  6086. /**
  6087. * scx_bpf_put_idle_cpumask - Release a previously acquired referenced kptr to
  6088. * either the percpu, or SMT idle-tracking cpumask.
  6089. */
  6090. __bpf_kfunc void scx_bpf_put_idle_cpumask(const struct cpumask *idle_mask)
  6091. {
  6092. /*
  6093. * Empty function body because we aren't actually acquiring or releasing
  6094. * a reference to a global idle cpumask, which is read-only in the
  6095. * caller and is never released. The acquire / release semantics here
  6096. * are just used to make the cpumask a trusted pointer in the caller.
  6097. */
  6098. }
  6099. /**
  6100. * scx_bpf_test_and_clear_cpu_idle - Test and clear @cpu's idle state
  6101. * @cpu: cpu to test and clear idle for
  6102. *
  6103. * Returns %true if @cpu was idle and its idle state was successfully cleared.
  6104. * %false otherwise.
  6105. *
  6106. * Unavailable if ops.update_idle() is implemented and
  6107. * %SCX_OPS_KEEP_BUILTIN_IDLE is not set.
  6108. */
  6109. __bpf_kfunc bool scx_bpf_test_and_clear_cpu_idle(s32 cpu)
  6110. {
  6111. if (!static_branch_likely(&scx_builtin_idle_enabled)) {
  6112. scx_ops_error("built-in idle tracking is disabled");
  6113. return false;
  6114. }
  6115. if (ops_cpu_valid(cpu, NULL))
  6116. return test_and_clear_cpu_idle(cpu);
  6117. else
  6118. return false;
  6119. }
  6120. /**
  6121. * scx_bpf_pick_idle_cpu - Pick and claim an idle cpu
  6122. * @cpus_allowed: Allowed cpumask
  6123. * @flags: %SCX_PICK_IDLE_CPU_* flags
  6124. *
  6125. * Pick and claim an idle cpu in @cpus_allowed. Returns the picked idle cpu
  6126. * number on success. -%EBUSY if no matching cpu was found.
  6127. *
  6128. * Idle CPU tracking may race against CPU scheduling state transitions. For
  6129. * example, this function may return -%EBUSY as CPUs are transitioning into the
  6130. * idle state. If the caller then assumes that there will be dispatch events on
  6131. * the CPUs as they were all busy, the scheduler may end up stalling with CPUs
  6132. * idling while there are pending tasks. Use scx_bpf_pick_any_cpu() and
  6133. * scx_bpf_kick_cpu() to guarantee that there will be at least one dispatch
  6134. * event in the near future.
  6135. *
  6136. * Unavailable if ops.update_idle() is implemented and
  6137. * %SCX_OPS_KEEP_BUILTIN_IDLE is not set.
  6138. */
  6139. __bpf_kfunc s32 scx_bpf_pick_idle_cpu(const struct cpumask *cpus_allowed,
  6140. u64 flags)
  6141. {
  6142. if (!static_branch_likely(&scx_builtin_idle_enabled)) {
  6143. scx_ops_error("built-in idle tracking is disabled");
  6144. return -EBUSY;
  6145. }
  6146. return scx_pick_idle_cpu(cpus_allowed, flags);
  6147. }
  6148. /**
  6149. * scx_bpf_pick_any_cpu - Pick and claim an idle cpu if available or pick any CPU
  6150. * @cpus_allowed: Allowed cpumask
  6151. * @flags: %SCX_PICK_IDLE_CPU_* flags
  6152. *
  6153. * Pick and claim an idle cpu in @cpus_allowed. If none is available, pick any
  6154. * CPU in @cpus_allowed. Guaranteed to succeed and returns the picked idle cpu
  6155. * number if @cpus_allowed is not empty. -%EBUSY is returned if @cpus_allowed is
  6156. * empty.
  6157. *
  6158. * If ops.update_idle() is implemented and %SCX_OPS_KEEP_BUILTIN_IDLE is not
  6159. * set, this function can't tell which CPUs are idle and will always pick any
  6160. * CPU.
  6161. */
  6162. __bpf_kfunc s32 scx_bpf_pick_any_cpu(const struct cpumask *cpus_allowed,
  6163. u64 flags)
  6164. {
  6165. s32 cpu;
  6166. if (static_branch_likely(&scx_builtin_idle_enabled)) {
  6167. cpu = scx_pick_idle_cpu(cpus_allowed, flags);
  6168. if (cpu >= 0)
  6169. return cpu;
  6170. }
  6171. cpu = cpumask_any_distribute(cpus_allowed);
  6172. if (cpu < nr_cpu_ids)
  6173. return cpu;
  6174. else
  6175. return -EBUSY;
  6176. }
  6177. /**
  6178. * scx_bpf_task_running - Is task currently running?
  6179. * @p: task of interest
  6180. */
  6181. __bpf_kfunc bool scx_bpf_task_running(const struct task_struct *p)
  6182. {
  6183. return task_rq(p)->curr == p;
  6184. }
  6185. /**
  6186. * scx_bpf_task_cpu - CPU a task is currently associated with
  6187. * @p: task of interest
  6188. */
  6189. __bpf_kfunc s32 scx_bpf_task_cpu(const struct task_struct *p)
  6190. {
  6191. return task_cpu(p);
  6192. }
  6193. /**
  6194. * scx_bpf_cpu_rq - Fetch the rq of a CPU
  6195. * @cpu: CPU of the rq
  6196. */
  6197. __bpf_kfunc struct rq *scx_bpf_cpu_rq(s32 cpu)
  6198. {
  6199. if (!ops_cpu_valid(cpu, NULL))
  6200. return NULL;
  6201. return cpu_rq(cpu);
  6202. }
  6203. /**
  6204. * scx_bpf_task_cgroup - Return the sched cgroup of a task
  6205. * @p: task of interest
  6206. *
  6207. * @p->sched_task_group->css.cgroup represents the cgroup @p is associated with
  6208. * from the scheduler's POV. SCX operations should use this function to
  6209. * determine @p's current cgroup as, unlike following @p->cgroups,
  6210. * @p->sched_task_group is protected by @p's rq lock and thus atomic w.r.t. all
  6211. * rq-locked operations. Can be called on the parameter tasks of rq-locked
  6212. * operations. The restriction guarantees that @p's rq is locked by the caller.
  6213. */
  6214. #ifdef CONFIG_CGROUP_SCHED
  6215. __bpf_kfunc struct cgroup *scx_bpf_task_cgroup(struct task_struct *p)
  6216. {
  6217. struct task_group *tg = p->sched_task_group;
  6218. struct cgroup *cgrp = &cgrp_dfl_root.cgrp;
  6219. if (!scx_kf_allowed_on_arg_tasks(__SCX_KF_RQ_LOCKED, p))
  6220. goto out;
  6221. /*
  6222. * A task_group may either be a cgroup or an autogroup. In the latter
  6223. * case, @tg->css.cgroup is %NULL. A task_group can't become the other
  6224. * kind once created.
  6225. */
  6226. if (tg && tg->css.cgroup)
  6227. cgrp = tg->css.cgroup;
  6228. else
  6229. cgrp = &cgrp_dfl_root.cgrp;
  6230. out:
  6231. cgroup_get(cgrp);
  6232. return cgrp;
  6233. }
  6234. #endif
  6235. __bpf_kfunc_end_defs();
  6236. BTF_KFUNCS_START(scx_kfunc_ids_any)
  6237. BTF_ID_FLAGS(func, scx_bpf_kick_cpu)
  6238. BTF_ID_FLAGS(func, scx_bpf_dsq_nr_queued)
  6239. BTF_ID_FLAGS(func, scx_bpf_destroy_dsq)
  6240. BTF_ID_FLAGS(func, bpf_iter_scx_dsq_new, KF_ITER_NEW | KF_RCU_PROTECTED)
  6241. BTF_ID_FLAGS(func, bpf_iter_scx_dsq_next, KF_ITER_NEXT | KF_RET_NULL)
  6242. BTF_ID_FLAGS(func, bpf_iter_scx_dsq_destroy, KF_ITER_DESTROY)
  6243. BTF_ID_FLAGS(func, scx_bpf_exit_bstr, KF_TRUSTED_ARGS)
  6244. BTF_ID_FLAGS(func, scx_bpf_error_bstr, KF_TRUSTED_ARGS)
  6245. BTF_ID_FLAGS(func, scx_bpf_dump_bstr, KF_TRUSTED_ARGS)
  6246. BTF_ID_FLAGS(func, scx_bpf_cpuperf_cap)
  6247. BTF_ID_FLAGS(func, scx_bpf_cpuperf_cur)
  6248. BTF_ID_FLAGS(func, scx_bpf_cpuperf_set)
  6249. BTF_ID_FLAGS(func, scx_bpf_nr_cpu_ids)
  6250. BTF_ID_FLAGS(func, scx_bpf_get_possible_cpumask, KF_ACQUIRE)
  6251. BTF_ID_FLAGS(func, scx_bpf_get_online_cpumask, KF_ACQUIRE)
  6252. BTF_ID_FLAGS(func, scx_bpf_put_cpumask, KF_RELEASE)
  6253. BTF_ID_FLAGS(func, scx_bpf_get_idle_cpumask, KF_ACQUIRE)
  6254. BTF_ID_FLAGS(func, scx_bpf_get_idle_smtmask, KF_ACQUIRE)
  6255. BTF_ID_FLAGS(func, scx_bpf_put_idle_cpumask, KF_RELEASE)
  6256. BTF_ID_FLAGS(func, scx_bpf_test_and_clear_cpu_idle)
  6257. BTF_ID_FLAGS(func, scx_bpf_pick_idle_cpu, KF_RCU)
  6258. BTF_ID_FLAGS(func, scx_bpf_pick_any_cpu, KF_RCU)
  6259. BTF_ID_FLAGS(func, scx_bpf_task_running, KF_RCU)
  6260. BTF_ID_FLAGS(func, scx_bpf_task_cpu, KF_RCU)
  6261. BTF_ID_FLAGS(func, scx_bpf_cpu_rq)
  6262. #ifdef CONFIG_CGROUP_SCHED
  6263. BTF_ID_FLAGS(func, scx_bpf_task_cgroup, KF_RCU | KF_ACQUIRE)
  6264. #endif
  6265. BTF_KFUNCS_END(scx_kfunc_ids_any)
  6266. static const struct btf_kfunc_id_set scx_kfunc_set_any = {
  6267. .owner = THIS_MODULE,
  6268. .set = &scx_kfunc_ids_any,
  6269. };
  6270. static int __init scx_init(void)
  6271. {
  6272. int ret;
  6273. /*
  6274. * kfunc registration can't be done from init_sched_ext_class() as
  6275. * register_btf_kfunc_id_set() needs most of the system to be up.
  6276. *
  6277. * Some kfuncs are context-sensitive and can only be called from
  6278. * specific SCX ops. They are grouped into BTF sets accordingly.
  6279. * Unfortunately, BPF currently doesn't have a way of enforcing such
  6280. * restrictions. Eventually, the verifier should be able to enforce
  6281. * them. For now, register them the same and make each kfunc explicitly
  6282. * check using scx_kf_allowed().
  6283. */
  6284. if ((ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
  6285. &scx_kfunc_set_select_cpu)) ||
  6286. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
  6287. &scx_kfunc_set_enqueue_dispatch)) ||
  6288. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
  6289. &scx_kfunc_set_dispatch)) ||
  6290. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
  6291. &scx_kfunc_set_cpu_release)) ||
  6292. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
  6293. &scx_kfunc_set_unlocked)) ||
  6294. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL,
  6295. &scx_kfunc_set_unlocked)) ||
  6296. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_STRUCT_OPS,
  6297. &scx_kfunc_set_any)) ||
  6298. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_TRACING,
  6299. &scx_kfunc_set_any)) ||
  6300. (ret = register_btf_kfunc_id_set(BPF_PROG_TYPE_SYSCALL,
  6301. &scx_kfunc_set_any))) {
  6302. pr_err("sched_ext: Failed to register kfunc sets (%d)\n", ret);
  6303. return ret;
  6304. }
  6305. ret = register_bpf_struct_ops(&bpf_sched_ext_ops, sched_ext_ops);
  6306. if (ret) {
  6307. pr_err("sched_ext: Failed to register struct_ops (%d)\n", ret);
  6308. return ret;
  6309. }
  6310. ret = register_pm_notifier(&scx_pm_notifier);
  6311. if (ret) {
  6312. pr_err("sched_ext: Failed to register PM notifier (%d)\n", ret);
  6313. return ret;
  6314. }
  6315. scx_kset = kset_create_and_add("sched_ext", &scx_uevent_ops, kernel_kobj);
  6316. if (!scx_kset) {
  6317. pr_err("sched_ext: Failed to create /sys/kernel/sched_ext\n");
  6318. return -ENOMEM;
  6319. }
  6320. ret = sysfs_create_group(&scx_kset->kobj, &scx_global_attr_group);
  6321. if (ret < 0) {
  6322. pr_err("sched_ext: Failed to add global attributes\n");
  6323. return ret;
  6324. }
  6325. return 0;
  6326. }
  6327. __initcall(scx_init);