print_scheduler.py 472 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401240224032404240524062407240824092410241124122413241424152416241724182419242024212422242324242425242624272428242924302431243224332434243524362437243824392440244124422443244424452446244724482449245024512452245324542455245624572458245924602461246224632464246524662467246824692470247124722473247424752476247724782479248024812482248324842485248624872488248924902491249224932494249524962497249824992500250125022503250425052506250725082509251025112512251325142515251625172518251925202521252225232524252525262527252825292530253125322533253425352536253725382539254025412542254325442545254625472548254925502551255225532554255525562557255825592560256125622563256425652566256725682569257025712572257325742575257625772578257925802581258225832584258525862587258825892590259125922593259425952596259725982599260026012602260326042605260626072608260926102611261226132614261526162617261826192620262126222623262426252626262726282629263026312632263326342635263626372638263926402641264226432644264526462647264826492650265126522653265426552656265726582659266026612662266326642665266626672668266926702671267226732674267526762677267826792680268126822683268426852686268726882689269026912692269326942695269626972698269927002701270227032704270527062707270827092710271127122713271427152716271727182719272027212722272327242725272627272728272927302731273227332734273527362737273827392740274127422743274427452746274727482749275027512752275327542755275627572758275927602761276227632764276527662767276827692770277127722773277427752776277727782779278027812782278327842785278627872788278927902791279227932794279527962797279827992800280128022803280428052806280728082809281028112812281328142815281628172818281928202821282228232824282528262827282828292830283128322833283428352836283728382839284028412842284328442845284628472848284928502851285228532854285528562857285828592860286128622863286428652866286728682869287028712872287328742875287628772878287928802881288228832884288528862887288828892890289128922893289428952896289728982899290029012902290329042905290629072908290929102911291229132914291529162917291829192920292129222923292429252926292729282929293029312932293329342935293629372938293929402941294229432944294529462947294829492950295129522953295429552956295729582959296029612962296329642965296629672968296929702971297229732974297529762977297829792980298129822983298429852986298729882989299029912992299329942995299629972998299930003001300230033004300530063007300830093010301130123013301430153016301730183019302030213022302330243025302630273028302930303031303230333034303530363037303830393040304130423043304430453046304730483049305030513052305330543055305630573058305930603061306230633064306530663067306830693070307130723073307430753076307730783079308030813082308330843085308630873088308930903091309230933094309530963097309830993100310131023103310431053106310731083109311031113112311331143115311631173118311931203121312231233124312531263127312831293130313131323133313431353136313731383139314031413142314331443145314631473148314931503151315231533154315531563157315831593160316131623163316431653166316731683169317031713172317331743175317631773178317931803181318231833184318531863187318831893190319131923193319431953196319731983199320032013202320332043205320632073208320932103211321232133214321532163217321832193220322132223223322432253226322732283229323032313232323332343235323632373238323932403241324232433244324532463247324832493250325132523253325432553256325732583259326032613262326332643265326632673268326932703271327232733274327532763277327832793280328132823283328432853286328732883289329032913292329332943295329632973298329933003301330233033304330533063307330833093310331133123313331433153316331733183319332033213322332333243325332633273328332933303331333233333334333533363337333833393340334133423343334433453346334733483349335033513352335333543355335633573358335933603361336233633364336533663367336833693370337133723373337433753376337733783379338033813382338333843385338633873388338933903391339233933394339533963397339833993400340134023403340434053406340734083409341034113412341334143415341634173418341934203421342234233424342534263427342834293430343134323433343434353436343734383439344034413442344334443445344634473448344934503451345234533454345534563457345834593460346134623463346434653466346734683469347034713472347334743475347634773478347934803481348234833484348534863487348834893490349134923493349434953496349734983499350035013502350335043505350635073508350935103511351235133514351535163517351835193520352135223523352435253526352735283529353035313532353335343535353635373538353935403541354235433544354535463547354835493550355135523553355435553556355735583559356035613562356335643565356635673568356935703571357235733574357535763577357835793580358135823583358435853586358735883589359035913592359335943595359635973598359936003601360236033604360536063607360836093610361136123613361436153616361736183619362036213622362336243625362636273628362936303631363236333634363536363637363836393640364136423643364436453646364736483649365036513652365336543655365636573658365936603661366236633664366536663667366836693670367136723673367436753676367736783679368036813682368336843685368636873688368936903691369236933694369536963697369836993700370137023703370437053706370737083709371037113712371337143715371637173718371937203721372237233724372537263727372837293730373137323733373437353736373737383739374037413742374337443745374637473748374937503751375237533754375537563757375837593760376137623763376437653766376737683769377037713772377337743775377637773778377937803781378237833784378537863787378837893790379137923793379437953796379737983799380038013802380338043805380638073808380938103811381238133814381538163817381838193820382138223823382438253826382738283829383038313832383338343835383638373838383938403841384238433844384538463847384838493850385138523853385438553856385738583859386038613862386338643865386638673868386938703871387238733874387538763877387838793880388138823883388438853886388738883889389038913892389338943895389638973898389939003901390239033904390539063907390839093910391139123913391439153916391739183919392039213922392339243925392639273928392939303931393239333934393539363937393839393940394139423943394439453946394739483949395039513952395339543955395639573958395939603961396239633964396539663967396839693970397139723973397439753976397739783979398039813982398339843985398639873988398939903991399239933994399539963997399839994000400140024003400440054006400740084009401040114012401340144015401640174018401940204021402240234024402540264027402840294030403140324033403440354036403740384039404040414042404340444045404640474048404940504051405240534054405540564057405840594060406140624063406440654066406740684069407040714072407340744075407640774078407940804081408240834084408540864087408840894090409140924093409440954096409740984099410041014102410341044105410641074108410941104111411241134114411541164117411841194120412141224123412441254126412741284129413041314132413341344135413641374138413941404141414241434144414541464147414841494150415141524153415441554156415741584159416041614162416341644165416641674168416941704171417241734174417541764177417841794180418141824183418441854186418741884189419041914192419341944195419641974198419942004201420242034204420542064207420842094210421142124213421442154216421742184219422042214222422342244225422642274228422942304231423242334234423542364237423842394240424142424243424442454246424742484249425042514252425342544255425642574258425942604261426242634264426542664267426842694270427142724273427442754276427742784279428042814282428342844285428642874288428942904291429242934294429542964297429842994300430143024303430443054306430743084309431043114312431343144315431643174318431943204321432243234324432543264327432843294330433143324333433443354336433743384339434043414342434343444345434643474348434943504351435243534354435543564357435843594360436143624363436443654366436743684369437043714372437343744375437643774378437943804381438243834384438543864387438843894390439143924393439443954396439743984399440044014402440344044405440644074408440944104411441244134414441544164417441844194420442144224423442444254426442744284429443044314432443344344435443644374438443944404441444244434444444544464447444844494450445144524453445444554456445744584459446044614462446344644465446644674468446944704471447244734474447544764477447844794480448144824483448444854486448744884489449044914492449344944495449644974498449945004501450245034504450545064507450845094510451145124513451445154516451745184519452045214522452345244525452645274528452945304531453245334534453545364537453845394540454145424543454445454546454745484549455045514552455345544555455645574558455945604561456245634564456545664567456845694570457145724573457445754576457745784579458045814582458345844585458645874588458945904591459245934594459545964597459845994600460146024603460446054606460746084609461046114612461346144615461646174618461946204621462246234624462546264627462846294630463146324633463446354636463746384639464046414642464346444645464646474648464946504651465246534654465546564657465846594660466146624663466446654666466746684669467046714672467346744675467646774678467946804681468246834684468546864687468846894690469146924693469446954696469746984699470047014702470347044705470647074708470947104711471247134714471547164717471847194720472147224723472447254726472747284729473047314732473347344735473647374738473947404741474247434744474547464747474847494750475147524753475447554756475747584759476047614762476347644765476647674768476947704771477247734774477547764777477847794780478147824783478447854786478747884789479047914792479347944795479647974798479948004801480248034804480548064807480848094810481148124813481448154816481748184819482048214822482348244825482648274828482948304831483248334834483548364837483848394840484148424843484448454846484748484849485048514852485348544855485648574858485948604861486248634864486548664867486848694870487148724873487448754876487748784879488048814882488348844885488648874888488948904891489248934894489548964897489848994900490149024903490449054906490749084909491049114912491349144915491649174918491949204921492249234924492549264927492849294930493149324933493449354936493749384939494049414942494349444945494649474948494949504951495249534954495549564957495849594960496149624963496449654966496749684969497049714972497349744975497649774978497949804981498249834984498549864987498849894990499149924993499449954996499749984999500050015002500350045005500650075008500950105011501250135014501550165017501850195020502150225023502450255026502750285029503050315032503350345035503650375038503950405041504250435044504550465047504850495050505150525053505450555056505750585059506050615062506350645065506650675068506950705071507250735074507550765077507850795080508150825083508450855086508750885089509050915092509350945095509650975098509951005101510251035104510551065107510851095110511151125113511451155116511751185119512051215122512351245125512651275128512951305131513251335134513551365137513851395140514151425143514451455146514751485149515051515152515351545155515651575158515951605161516251635164516551665167516851695170517151725173517451755176517751785179518051815182518351845185518651875188518951905191519251935194519551965197519851995200520152025203520452055206520752085209521052115212521352145215521652175218521952205221522252235224522552265227522852295230523152325233523452355236523752385239524052415242524352445245524652475248524952505251525252535254525552565257525852595260526152625263526452655266526752685269527052715272527352745275527652775278527952805281528252835284528552865287528852895290529152925293529452955296529752985299530053015302530353045305530653075308530953105311531253135314531553165317531853195320532153225323532453255326532753285329533053315332533353345335533653375338533953405341534253435344534553465347534853495350535153525353535453555356535753585359536053615362536353645365536653675368536953705371537253735374537553765377537853795380538153825383538453855386538753885389539053915392539353945395539653975398539954005401540254035404540554065407540854095410541154125413541454155416541754185419542054215422542354245425542654275428542954305431543254335434543554365437543854395440544154425443544454455446544754485449545054515452545354545455545654575458545954605461546254635464546554665467546854695470547154725473547454755476547754785479548054815482548354845485548654875488548954905491549254935494549554965497549854995500550155025503550455055506550755085509551055115512551355145515551655175518551955205521552255235524552555265527552855295530553155325533553455355536553755385539554055415542554355445545554655475548554955505551555255535554555555565557555855595560556155625563556455655566556755685569557055715572557355745575557655775578557955805581558255835584558555865587558855895590559155925593559455955596559755985599560056015602560356045605560656075608560956105611561256135614561556165617561856195620562156225623562456255626562756285629563056315632563356345635563656375638563956405641564256435644564556465647564856495650565156525653565456555656565756585659566056615662566356645665566656675668566956705671567256735674567556765677567856795680568156825683568456855686568756885689569056915692569356945695569656975698569957005701570257035704570557065707570857095710571157125713571457155716571757185719572057215722572357245725572657275728572957305731573257335734573557365737573857395740574157425743574457455746574757485749575057515752575357545755575657575758575957605761576257635764576557665767576857695770577157725773577457755776577757785779578057815782578357845785578657875788578957905791579257935794579557965797579857995800580158025803580458055806580758085809581058115812581358145815581658175818581958205821582258235824582558265827582858295830583158325833583458355836583758385839584058415842584358445845584658475848584958505851585258535854585558565857585858595860586158625863586458655866586758685869587058715872587358745875587658775878587958805881588258835884588558865887588858895890589158925893589458955896589758985899590059015902590359045905590659075908590959105911591259135914591559165917591859195920592159225923592459255926592759285929593059315932593359345935593659375938593959405941594259435944594559465947594859495950595159525953595459555956595759585959596059615962596359645965596659675968596959705971597259735974597559765977597859795980598159825983598459855986598759885989599059915992599359945995599659975998599960006001600260036004600560066007600860096010601160126013601460156016601760186019602060216022602360246025602660276028602960306031603260336034603560366037603860396040604160426043604460456046604760486049605060516052605360546055605660576058605960606061606260636064606560666067606860696070607160726073607460756076607760786079608060816082608360846085608660876088608960906091609260936094609560966097609860996100610161026103610461056106610761086109611061116112611361146115611661176118611961206121612261236124612561266127612861296130613161326133613461356136613761386139614061416142614361446145614661476148614961506151615261536154615561566157615861596160616161626163616461656166616761686169617061716172617361746175617661776178617961806181618261836184618561866187618861896190619161926193619461956196619761986199620062016202620362046205620662076208620962106211621262136214621562166217621862196220622162226223622462256226622762286229623062316232623362346235623662376238623962406241624262436244624562466247624862496250625162526253625462556256625762586259626062616262626362646265626662676268626962706271627262736274627562766277627862796280628162826283628462856286628762886289629062916292629362946295629662976298629963006301630263036304630563066307630863096310631163126313631463156316631763186319632063216322632363246325632663276328632963306331633263336334633563366337633863396340634163426343634463456346634763486349635063516352635363546355635663576358635963606361636263636364636563666367636863696370637163726373637463756376637763786379638063816382638363846385638663876388638963906391639263936394639563966397639863996400640164026403640464056406640764086409641064116412641364146415641664176418641964206421642264236424642564266427642864296430643164326433643464356436643764386439644064416442644364446445644664476448644964506451645264536454645564566457645864596460646164626463646464656466646764686469647064716472647364746475647664776478647964806481648264836484648564866487648864896490649164926493649464956496649764986499650065016502650365046505650665076508650965106511651265136514651565166517651865196520652165226523652465256526652765286529653065316532653365346535653665376538653965406541654265436544654565466547654865496550655165526553655465556556655765586559656065616562656365646565656665676568656965706571657265736574657565766577657865796580658165826583658465856586658765886589659065916592659365946595659665976598659966006601660266036604660566066607660866096610661166126613661466156616661766186619662066216622662366246625662666276628662966306631663266336634663566366637663866396640664166426643664466456646664766486649665066516652665366546655665666576658665966606661666266636664666566666667666866696670667166726673667466756676667766786679668066816682668366846685668666876688668966906691669266936694669566966697669866996700670167026703670467056706670767086709671067116712671367146715671667176718671967206721672267236724672567266727672867296730673167326733673467356736673767386739674067416742674367446745674667476748674967506751675267536754675567566757675867596760676167626763676467656766676767686769677067716772677367746775677667776778677967806781678267836784678567866787678867896790679167926793679467956796679767986799680068016802680368046805680668076808680968106811681268136814681568166817681868196820682168226823682468256826682768286829683068316832683368346835683668376838683968406841684268436844684568466847684868496850685168526853685468556856685768586859686068616862686368646865686668676868686968706871687268736874687568766877687868796880688168826883688468856886688768886889689068916892689368946895689668976898689969006901690269036904690569066907690869096910691169126913691469156916691769186919692069216922692369246925692669276928692969306931693269336934693569366937693869396940694169426943694469456946694769486949695069516952695369546955695669576958695969606961696269636964696569666967696869696970697169726973697469756976697769786979698069816982698369846985698669876988698969906991699269936994699569966997699869997000700170027003700470057006700770087009701070117012701370147015701670177018701970207021702270237024702570267027702870297030703170327033703470357036703770387039704070417042704370447045704670477048704970507051705270537054705570567057705870597060706170627063706470657066706770687069707070717072707370747075707670777078707970807081708270837084708570867087708870897090709170927093709470957096709770987099710071017102710371047105710671077108710971107111711271137114711571167117711871197120712171227123712471257126712771287129713071317132713371347135713671377138713971407141714271437144714571467147714871497150715171527153715471557156715771587159716071617162716371647165716671677168716971707171717271737174717571767177717871797180718171827183718471857186718771887189719071917192719371947195719671977198719972007201720272037204720572067207720872097210721172127213721472157216721772187219722072217222722372247225722672277228722972307231723272337234723572367237723872397240724172427243724472457246724772487249725072517252725372547255725672577258725972607261726272637264726572667267726872697270727172727273727472757276727772787279728072817282728372847285728672877288728972907291729272937294729572967297729872997300730173027303730473057306730773087309731073117312731373147315731673177318731973207321732273237324732573267327732873297330733173327333733473357336733773387339734073417342734373447345734673477348734973507351735273537354735573567357735873597360736173627363736473657366736773687369737073717372737373747375737673777378737973807381738273837384738573867387738873897390739173927393739473957396739773987399740074017402740374047405740674077408740974107411741274137414741574167417741874197420742174227423742474257426742774287429743074317432743374347435743674377438743974407441744274437444744574467447744874497450745174527453745474557456745774587459746074617462746374647465746674677468746974707471747274737474747574767477747874797480748174827483748474857486748774887489749074917492749374947495749674977498749975007501750275037504750575067507750875097510751175127513751475157516751775187519752075217522752375247525752675277528752975307531753275337534753575367537753875397540754175427543754475457546754775487549755075517552755375547555755675577558755975607561756275637564756575667567756875697570757175727573757475757576757775787579758075817582758375847585758675877588758975907591759275937594759575967597759875997600760176027603760476057606760776087609761076117612761376147615761676177618761976207621762276237624762576267627762876297630763176327633763476357636763776387639764076417642764376447645764676477648764976507651765276537654765576567657765876597660766176627663766476657666766776687669767076717672767376747675767676777678767976807681768276837684768576867687768876897690769176927693769476957696769776987699770077017702770377047705770677077708770977107711771277137714771577167717771877197720772177227723772477257726772777287729773077317732773377347735773677377738773977407741774277437744774577467747774877497750775177527753775477557756775777587759776077617762776377647765776677677768776977707771777277737774777577767777777877797780778177827783778477857786778777887789779077917792779377947795779677977798779978007801780278037804780578067807780878097810781178127813781478157816781778187819782078217822782378247825782678277828782978307831783278337834783578367837783878397840784178427843784478457846784778487849785078517852785378547855785678577858785978607861786278637864786578667867786878697870787178727873787478757876787778787879788078817882788378847885788678877888788978907891789278937894789578967897789878997900790179027903790479057906790779087909791079117912791379147915791679177918791979207921792279237924792579267927792879297930793179327933793479357936793779387939794079417942794379447945794679477948794979507951795279537954795579567957795879597960796179627963796479657966796779687969797079717972797379747975797679777978797979807981798279837984798579867987798879897990799179927993799479957996799779987999800080018002800380048005800680078008800980108011801280138014801580168017801880198020802180228023802480258026802780288029803080318032803380348035803680378038803980408041804280438044804580468047804880498050805180528053805480558056805780588059806080618062806380648065806680678068806980708071807280738074807580768077807880798080808180828083808480858086808780888089809080918092809380948095809680978098809981008101810281038104810581068107810881098110811181128113811481158116811781188119812081218122812381248125812681278128812981308131813281338134813581368137813881398140814181428143814481458146814781488149815081518152815381548155815681578158815981608161816281638164816581668167816881698170817181728173817481758176817781788179818081818182818381848185818681878188818981908191819281938194819581968197819881998200820182028203820482058206820782088209821082118212821382148215821682178218821982208221822282238224822582268227822882298230823182328233823482358236823782388239824082418242824382448245824682478248824982508251825282538254825582568257825882598260826182628263826482658266826782688269827082718272827382748275827682778278827982808281828282838284828582868287828882898290829182928293829482958296829782988299830083018302830383048305830683078308830983108311831283138314831583168317831883198320832183228323832483258326832783288329833083318332833383348335833683378338833983408341834283438344834583468347834883498350835183528353835483558356835783588359836083618362836383648365836683678368836983708371837283738374837583768377837883798380838183828383838483858386838783888389839083918392839383948395839683978398839984008401840284038404840584068407840884098410841184128413841484158416841784188419842084218422842384248425842684278428842984308431843284338434843584368437843884398440844184428443844484458446844784488449845084518452845384548455845684578458845984608461846284638464846584668467846884698470847184728473847484758476847784788479848084818482848384848485848684878488848984908491849284938494849584968497849884998500850185028503850485058506850785088509851085118512851385148515851685178518851985208521852285238524852585268527852885298530853185328533853485358536853785388539854085418542854385448545854685478548854985508551855285538554855585568557855885598560856185628563856485658566856785688569857085718572857385748575857685778578857985808581858285838584858585868587858885898590859185928593859485958596859785988599860086018602860386048605860686078608860986108611861286138614861586168617861886198620862186228623862486258626862786288629863086318632863386348635863686378638863986408641864286438644864586468647864886498650865186528653865486558656865786588659866086618662866386648665866686678668866986708671867286738674867586768677867886798680868186828683868486858686868786888689869086918692869386948695869686978698869987008701870287038704870587068707870887098710871187128713871487158716871787188719872087218722872387248725872687278728872987308731873287338734873587368737873887398740874187428743874487458746874787488749875087518752875387548755875687578758875987608761876287638764876587668767876887698770877187728773877487758776877787788779878087818782878387848785878687878788878987908791879287938794879587968797879887998800880188028803880488058806880788088809881088118812881388148815881688178818881988208821882288238824882588268827882888298830883188328833883488358836883788388839884088418842884388448845884688478848884988508851885288538854885588568857885888598860886188628863886488658866886788688869887088718872887388748875887688778878887988808881888288838884888588868887888888898890889188928893889488958896889788988899890089018902890389048905890689078908890989108911891289138914891589168917891889198920892189228923892489258926892789288929893089318932893389348935893689378938893989408941894289438944894589468947894889498950895189528953895489558956895789588959896089618962896389648965896689678968896989708971897289738974897589768977897889798980898189828983898489858986898789888989899089918992899389948995899689978998899990009001900290039004900590069007900890099010901190129013901490159016901790189019902090219022902390249025902690279028902990309031903290339034903590369037903890399040904190429043904490459046904790489049905090519052905390549055905690579058905990609061906290639064906590669067906890699070907190729073907490759076907790789079908090819082908390849085908690879088908990909091909290939094909590969097909890999100910191029103910491059106910791089109911091119112911391149115911691179118911991209121912291239124912591269127912891299130913191329133913491359136913791389139914091419142914391449145914691479148914991509151915291539154915591569157915891599160916191629163916491659166916791689169917091719172917391749175917691779178917991809181918291839184918591869187918891899190919191929193919491959196919791989199920092019202920392049205920692079208920992109211921292139214921592169217921892199220922192229223922492259226922792289229923092319232923392349235923692379238923992409241924292439244924592469247924892499250925192529253925492559256925792589259926092619262926392649265926692679268926992709271927292739274927592769277927892799280928192829283928492859286928792889289929092919292929392949295929692979298929993009301930293039304930593069307930893099310931193129313931493159316931793189319932093219322932393249325
  1. """Print scheduler service - processes the print queue."""
  2. import asyncio
  3. import json
  4. import logging
  5. import time
  6. import uuid
  7. import zipfile
  8. from collections import deque
  9. from collections.abc import Mapping
  10. from dataclasses import dataclass
  11. from datetime import datetime, timedelta, timezone
  12. from pathlib import Path
  13. from fastapi import HTTPException
  14. from sqlalchemy import delete, false, func, or_, select, true, update
  15. from sqlalchemy.ext.asyncio import AsyncSession
  16. from sqlalchemy.orm import selectinload
  17. from backend.app.core.config import settings
  18. from backend.app.core.database import async_session, run_with_retry
  19. from backend.app.core.printer_scope import ALL_PRINTERS, PrinterScope, resolve_user_id_printer_scope
  20. from backend.app.core.tasks import spawn_background_task
  21. from backend.app.core.websocket import ws_manager
  22. from backend.app.models.archive import PrintArchive
  23. from backend.app.models.library import LibraryFile
  24. from backend.app.models.print_queue import PrintQueueItem, PrintQueueVariant
  25. from backend.app.models.printer import Printer
  26. from backend.app.models.scheduled_drying import ScheduledDrying
  27. from backend.app.models.settings import Settings
  28. from backend.app.models.smart_plug import SmartPlug
  29. from backend.app.models.spool_assignment import SpoolAssignment
  30. from backend.app.models.spoolman_slot_assignment import SpoolmanSlotAssignment
  31. from backend.app.services import drying_preflight, kprofile_drift, print_dispatch_context, stock_forecast
  32. from backend.app.services.bambu_ftp import (
  33. FtpFailureKind,
  34. FtpFailureReport,
  35. UploadCancelled,
  36. cache_3mf_download,
  37. delete_file_async,
  38. describe_upload_failure,
  39. get_ftp_retry_settings,
  40. upload_file_async,
  41. with_ftp_retry,
  42. )
  43. from backend.app.services.bambu_mqtt import _RACK_NOZZLE_IDS, HMS_MQTT_VERIFY_FAILED, resolve_rack_plan_mapping
  44. from backend.app.services.filament_deficit import compute_deficit_for_queue_item
  45. from backend.app.services.finance_budget import (
  46. create_budget_reservation,
  47. release_budget_reservation,
  48. validate_print_budget,
  49. )
  50. from backend.app.services.ha_sensor_manager import ha_sensor_manager
  51. from backend.app.services.notification_service import notification_service
  52. from backend.app.services.print_cost_estimate import estimate_queue_source_cost
  53. from backend.app.services.printer_manager import (
  54. printer_manager,
  55. supports_airduct,
  56. supports_chamber_heater,
  57. supports_chamber_temp,
  58. supports_drying,
  59. supports_drying_while_printing,
  60. )
  61. from backend.app.services.smart_plug_manager import smart_plug_manager
  62. from backend.app.services.spool_assignment_notifications import (
  63. _global_tray_from_assignment,
  64. _slot_label_from_global_tray,
  65. )
  66. from backend.app.utils.ams_drying import is_countdown_parked
  67. from backend.app.utils.ams_humidity import ams_humidity_percent
  68. from backend.app.utils.archive_paths import archive_photos_dir
  69. from backend.app.utils.color_utils import perceptual_color_distance
  70. from backend.app.utils.filament_types import canonical_filament_type
  71. from backend.app.utils.filename import derive_remote_filename
  72. from backend.app.utils.library_paths import move_library_photos
  73. from backend.app.utils.local_time import utcnow_naive
  74. from backend.app.utils.printer_models import (
  75. is_dual_nozzle_model,
  76. is_gcode_compatible,
  77. is_nozzle_rack_model,
  78. normalize_printer_model,
  79. )
  80. from backend.app.utils.threemf_tools import (
  81. default_plate_number,
  82. extract_rack_plan_from_3mf,
  83. extract_slot_extruders_from_3mf,
  84. select_plate_gcode_name,
  85. )
  86. logger = logging.getLogger(__name__)
  87. def _ams_slot_label(ams_id: int, tray_id: int) -> str:
  88. """Human slot name for an assignment tuple, in the spelling already in use.
  89. Composed from the notification side's own pair rather than spelled out again
  90. here. The hand-rolled version this replaces got three real slot kinds wrong:
  91. ams_id 254 fell into its ``>= 128`` branch and came out as ``HT-`` plus
  92. ``chr(191)``; both externals collapsed to one label, because an external
  93. assignment stores ams_id 255 with tray_id picking left or right
  94. (inventory.py:1771, ``ext_id = data.tray_id + 254``); and an A2L AMS-Lite,
  95. normalised to unit 6, read as ``AMS-G`` instead of ``Lite-``.
  96. Naming the same slot two ways is its own defect once an alert and the
  97. Inventory page are meant to be talking about the same tray, so this defers
  98. rather than adding a fifth spelling.
  99. """
  100. return _slot_label_from_global_tray(_global_tray_from_assignment(ams_id, tray_id))
  101. # Minimum seconds between low-filament checks. Matches the scheduler's idle
  102. # interval: the fast path runs every 3 s while an upload is in flight, and a
  103. # low spool does not need answering at that resolution (#2913).
  104. _FILAMENT_LOW_MIN_INTERVAL = 30.0
  105. # Minimum seconds between stock forecast checks (#2955). The forecast is in whole
  106. # days, so a finer check would only repeat the same answer.
  107. _STOCK_FORECAST_MIN_INTERVAL = 3600.0
  108. # The settings row that holds which stock alerts have already been sent (#2955).
  109. _STOCK_ALERTS_SETTING_KEY = "stock_alerts_notified"
  110. # The Inventory page asks for this many usage records and the forecast panel
  111. # builds its rates from them; the alert reads the same window so the two agree.
  112. _STOCK_FORECAST_HISTORY_LIMIT = 5000
  113. def _remaining_percent(label_weight: int | float | None, weight_used: float | None) -> float | None:
  114. """Remaining filament as a percentage of the label weight.
  115. The same arithmetic the Inventory page's Low Stock count uses, and it works
  116. unchanged in both inventory modes: internal spools store ``weight_used``
  117. directly, and ``_map_spoolman_spool`` derives it from Spoolman's
  118. ``remaining_weight`` so the shape matches. Returns None when there is no
  119. label weight to be a percentage of -- a spool that cannot say how full it
  120. started cannot say how empty it is.
  121. """
  122. if not label_weight or label_weight <= 0:
  123. return None
  124. remaining = max(0.0, float(label_weight) - float(weight_used or 0.0))
  125. return remaining / float(label_weight) * 100.0
  126. # Dispatch-toast progress throttling (#1625 follow-up). Mirrors the legacy
  127. # background_dispatch.py upload_progress_callback (200 ms time gate + 256 KB
  128. # byte gate) from before the scheduler unification. Time gate keeps small
  129. # files from going silent (a single 8 KB chunk fires once and that's it);
  130. # byte gate caps the broadcast rate on slow LAN where 200 ms covers many
  131. # chunks. uploaded >= total always emits so the bar closes cleanly even on
  132. # sub-200 ms files.
  133. _DISPATCH_PROGRESS_BYTE_STEP = 256 * 1024
  134. _DISPATCH_PROGRESS_MIN_INTERVAL_SECS = 0.2
  135. # How far back chamber temperature samples are retained. 2h comfortably spans
  136. # any soak a user can configure (capped at 30 min) plus the plate-clearing gap
  137. # before the next print.
  138. _CHAMBER_HISTORY_TTL_SECONDS = 7200
  139. # Fallback for `queue_keep_warm_max_minutes` — how long the bed may be held
  140. # warm on a printer sitting in FINISH before the heaters are shut off. Users
  141. # who clear plates promptly will want far less than this; it is deliberately
  142. # the cautious end, since the cost of it being too short is only a re-soak.
  143. _KEEP_WARM_MAX_MINUTES_DEFAULT = 120
  144. # Max acceptable gap between two consecutive chamber samples before we treat
  145. # the older one as belonging to a separate observation run (printer went
  146. # offline and came back). Above ~30s cadence with a safety margin.
  147. _CHAMBER_SAMPLE_MAX_GAP_SECONDS = 60.0
  148. # How long the chamber must read below target before we accept that it really
  149. # cooled. An enclosed chamber's thermal mass cannot lose and regain several
  150. # degrees quickly: measured on an X1C, falling from ~55°C to below 48°C took
  151. # 23-73 minutes (~0.2 C/min), while the fastest drop ever recorded was
  152. # 27 C/min — impossible for that mass, i.e. a sensor artifact. Brief
  153. # sub-target readings are therefore a door opening or noise, not lost soak,
  154. # and a plate swap (exactly when keep-warm is running) produces one. Six
  155. # minutes clears the longest such artifact observed (~5 min once bracketed by
  156. # its neighbouring samples) and still sits far below the 23-minute floor for
  157. # real cooling.
  158. _CHAMBER_DIP_GRACE_SECONDS = 360.0
  159. # How often the preheat stage re-checks that the item it is heating for still
  160. # wants to be printed. Cancelling only writes `status` to the database — it
  161. # cannot interrupt a coroutine parked in `asyncio.sleep` — so without this the
  162. # heaters run for the rest of max_wait + soak (45 min at the default settings)
  163. # and the printer stays in `busy_printers`, blocking every other queued item
  164. # behind a print that is not happening.
  165. _PREHEAT_CANCEL_CHECK_SECONDS = 10.0
  166. # set_airduct_mode modeId values (bambu_mqtt.py:5937 — 0 cooling, 1 heating).
  167. _AIRDUCT_MODE_COOLING = 0
  168. _AIRDUCT_MODE_HEATING = 1
  169. # How long a queue row may stay 'printing' while its printer sits in a terminal
  170. # state before the scheduler closes it itself (#2829).
  171. #
  172. # A real completion arrives within seconds of the printer going terminal, so
  173. # five minutes is far outside the normal path — this only ever sees a row whose
  174. # completion was refused or never delivered. It is the whole cost of the
  175. # failure to the user, though: a stranded row blocks every later job for that
  176. # printer, so it should not be raised without reason.
  177. _STRANDED_PRINTING_GRACE_SECONDS = 300.0
  178. # gcode_state values that mean the print is over, mapped to the queue status
  179. # they imply. Mirrors the mapping in bambu_mqtt's completion detection
  180. # (FINISH -> completed, FAILED -> failed, anything else terminal -> aborted,
  181. # which the queue calls cancelled) so a recovered row cannot disagree with one
  182. # closed by the normal path.
  183. _TERMINAL_STATE_QUEUE_STATUS = {
  184. "FINISH": "completed",
  185. "FAILED": "failed",
  186. "IDLE": "cancelled",
  187. }
  188. def _terminal_queue_status(state) -> str | None:
  189. """Queue status implied by *state*, or None if the print is not over.
  190. None for a disconnected printer as well as a busy one: a printer we are not
  191. talking to has a stale ``state`` field that proves nothing about what it is
  192. doing now.
  193. """
  194. if state is None or not getattr(state, "connected", False):
  195. return None
  196. return _TERMINAL_STATE_QUEUE_STATUS.get(getattr(state, "state", None))
  197. @dataclass
  198. class _KeepWarmEntry:
  199. """Per-printer keep-warm state.
  200. - ``started``: monotonic time when keep-warm first fired for this printer.
  201. Used by the max-duration timeout.
  202. - ``held_target``: last bed target we successfully published. On release
  203. we only send bed-off when firmware still reports this value, so a user
  204. or subsequent print that changed the target isn't clobbered.
  205. - ``expired``: latched True when the max-duration timeout fires. Prevents
  206. re-engagement (and re-seeding of ``started``) on subsequent ticks. The
  207. release sweep drops the entry entirely once the printer leaves the
  208. candidate set.
  209. """
  210. started: float
  211. held_target: int
  212. expired: bool = False
  213. # Auto-drying re-arm guards (#2770).
  214. #
  215. # An AMS reports HIGHER relative humidity while it is warm than once it has
  216. # cooled: measured on an H2D/AMS 2 Pro at 10-13% cold against 15-20% throughout
  217. # every drying cycle, and the same unit read 16-20% across a full 12 h dry. A
  218. # threshold set inside that band therefore cannot be satisfied while the box is
  219. # hot, and the firmware is free to end a cycle whenever it decides the filament
  220. # is dry — so the next 30 s pass sees dry_time 0 with the reading still above
  221. # the threshold and arms another cycle. The reporter's log has five 12-hour
  222. # cycles armed inside four hours, one of them six seconds after the previous
  223. # ended.
  224. #
  225. # The cooldown stops the six-second re-arm; the unproductive-cycle cap stops the
  226. # loop. Neither ever stops a running cycle — both only gate STARTING one, so a
  227. # manual or firmware-run dry is untouched.
  228. AUTO_DRY_REARM_COOLDOWN_SECONDS = 30 * 60
  229. AUTO_DRY_MAX_UNPRODUCTIVE_CYCLES = 2
  230. # Sustained-humidity wait (#2518): an above-threshold streak is only
  231. # "continuous" if some pass observed it within the gap ceiling. A unit that
  232. # goes unobserved longer (print running, printer disconnected, sensor silent)
  233. # restarts its streak rather than inheriting a stale one. The ceiling is
  234. # derived from the scheduler cadence — four missed passes — with this floor so
  235. # a fast-polling configuration does not void streaks on a single hiccup.
  236. AUTO_DRY_SUSTAINED_GAP_FLOOR_SECONDS = 120
  237. # How long a finished scheduled drying row is kept before it is pruned.
  238. SCHEDULED_DRYING_RETENTION_DAYS = 7
  239. # How often that prune actually runs. The check itself is called on every queue
  240. # pass — every 3s while dispatching — and issuing the DELETE is what begins a
  241. # write transaction, which SQLite serialises against every other writer. Rows
  242. # only become prunable a week after they finish, so anything short of hourly is
  243. # paying that cost for nothing.
  244. SCHEDULED_DRYING_PRUNE_INTERVAL_SECONDS = 60 * 60
  245. class _UploadProgressBridge:
  246. """Thread-safe bridge from ``upload_file_async`` to the WS broadcaster.
  247. ``upload_file_async`` runs the FTP transfer in an executor thread and
  248. invokes its ``progress_callback`` from that thread, so the callback
  249. body cannot ``await`` directly. This bridge captures the asyncio loop
  250. at construction (on the scheduler thread) and uses
  251. ``run_coroutine_threadsafe`` to hop back. The byte/time throttle
  252. matches the legacy background_dispatch.py path 1:1 so the toast feels
  253. identical to the pre-#1625 experience.
  254. Failures inside the emit are swallowed — progress is a UX nicety, the
  255. upload itself must not fail because of a WS hiccup.
  256. """
  257. def __init__(self, user_id: int | None, queue_item_id: int):
  258. self._user_id = user_id
  259. self._queue_item_id = queue_item_id
  260. try:
  261. self._loop = asyncio.get_running_loop()
  262. except RuntimeError:
  263. self._loop = None
  264. self._last_emit_bytes = 0
  265. self._last_emit_monotonic = 0.0
  266. self._has_emitted = False
  267. def __call__(self, bytes_transferred: int, total_bytes: int) -> None:
  268. if self._loop is None or total_bytes <= 0:
  269. return
  270. now = time.monotonic()
  271. # Mirrors legacy bg-dispatch: emit if first call OR upload complete
  272. # OR 200 ms elapsed OR ≥256 KB transferred since last emit. Two of
  273. # the four matter most: first-call so the user sees something even
  274. # for sub-chunk-size files; uploaded >= total so the bar locks at
  275. # 100% even when the throttle would otherwise eat it.
  276. should_emit = (
  277. not self._has_emitted
  278. or bytes_transferred >= total_bytes
  279. or now - self._last_emit_monotonic >= _DISPATCH_PROGRESS_MIN_INTERVAL_SECS
  280. or bytes_transferred - self._last_emit_bytes >= _DISPATCH_PROGRESS_BYTE_STEP
  281. )
  282. if not should_emit:
  283. return
  284. self._has_emitted = True
  285. self._last_emit_bytes = bytes_transferred
  286. self._last_emit_monotonic = now
  287. try:
  288. asyncio.run_coroutine_threadsafe(
  289. ws_manager.send_queue_item_upload_progress(
  290. user_id=self._user_id,
  291. queue_item_id=self._queue_item_id,
  292. bytes_transferred=bytes_transferred,
  293. total_bytes=total_bytes,
  294. ),
  295. self._loop,
  296. )
  297. except Exception:
  298. pass # progress is best-effort, never block the upload
  299. # Bambu firmware states that mean the project_file has actually been accepted
  300. # and the printer is now processing / running / paused mid-print. Used by the
  301. # dispatch watchdog (#1370): a transition into one of these states means the
  302. # print landed, anything else (e.g. FINISH -> IDLE after the user dismisses
  303. # a post-print prompt) is NOT a valid "command landed" signal even though the
  304. # state value did change. SLICING is included because some firmwares park
  305. # briefly in SLICING between PREPARE and RUNNING while parsing the g-code.
  306. _ACTIVE_PRINT_STATES: frozenset[str] = frozenset({"PREPARE", "SLICING", "RUNNING", "PAUSE"})
  307. # How many times the start-watchdog may revert an item to 'pending' before it
  308. # gives up and fails the row instead (#2555). Each attempt costs a full 3MF
  309. # re-upload plus the watchdog's wait, so a wedged printer left to retry forever
  310. # both never recovers and starves the other printers of dispatch slots. Three
  311. # is chosen to clear the transient causes the watchdog already recovers from —
  312. # a lost MQTT publish on a half-broken session (#887/#936) is fixed by the
  313. # force-reconnect on the very next attempt — while still bounding the loop.
  314. DISPATCH_MAX_ATTEMPTS = 3
  315. # Upload failures that mean the file never reached the printer's storage: the
  316. # file service refused or never answered (#3210). These put the item back in
  317. # the queue instead of failing it, because nothing about the job is wrong --
  318. # and failing it left the printer idle, so the next pass handed it the next
  319. # item, which failed the same way. One P2S whose file service was out of
  320. # connection slots ate 43 queued jobs in ten minutes that way.
  321. #
  322. # Not here: AUTH (a wrong access code needs the user), STORAGE (a full or
  323. # missing card does too), NOT_FOUND (a Bambuddy-side path problem), UNKNOWN,
  324. # and an upload that overran its deadline -- each of those would fail the same
  325. # way on every retry.
  326. _UPLOAD_REQUEUE_KINDS: frozenset[FtpFailureKind] = frozenset(
  327. {FtpFailureKind.HANDSHAKE, FtpFailureKind.COOLOFF, FtpFailureKind.TIMEOUT, FtpFailureKind.NETWORK}
  328. )
  329. # How long a printer whose upload was put back stays out of dispatch (#3210).
  330. # The first window matches the FTP client's own handshake cool-off, so the
  331. # retry lands after the client would talk to the printer again anyway. Each
  332. # further refusal in a row doubles it, up to the cap: #3210's printer refused
  333. # for some 40 hours, and every retry costs a preheat cycle where preheat is on,
  334. # five connection attempts and a page of log. A successful upload resets it.
  335. UPLOAD_FAILURE_BACKOFF_SECONDS = 300
  336. UPLOAD_FAILURE_BACKOFF_MAX_SECONDS = 3600
  337. def _upload_backoff_seconds(refusals: int) -> int:
  338. """Backoff after the *refusals*-th refused upload in a row (1-based)."""
  339. return min(UPLOAD_FAILURE_BACKOFF_SECONDS * 2 ** max(refusals - 1, 0), UPLOAD_FAILURE_BACKOFF_MAX_SECONDS)
  340. @dataclass(slots=True)
  341. class _ModelCandidate:
  342. """One (file, printer model) pair the model-based matcher may try.
  343. Model-based assignment used to have exactly one of these per item, held
  344. directly in the item's own columns. Cross-model queue items (#671) have
  345. several, held in ``print_queue_variants``. Both shapes are normalised into
  346. this so the matching, the cross-model gate and the waiting-reason handling
  347. are written once and an item without variants provably takes the same path
  348. it took before variants existed.
  349. ``variant`` is None for the item's own columns and set for a real variant
  350. row, which is what :meth:`PrintScheduler._resolve_variant` writes onto the
  351. item once that candidate wins.
  352. """
  353. target_model: str | None
  354. sliced_for: str | None
  355. required_filament_types: str | None
  356. filament_overrides: str | None
  357. variant: "PrintQueueVariant | None" = None
  358. def _sliced_for_model(archive, library_file) -> str | None:
  359. """Model a 3MF declares it was sliced for, from whichever source holds it."""
  360. if archive is not None:
  361. return archive.sliced_for_model
  362. if library_file is not None and library_file.file_metadata:
  363. return library_file.file_metadata.get("sliced_for_model")
  364. return None
  365. def _filament_constraints(candidate: _ModelCandidate) -> tuple[list[str] | None, list[dict] | None]:
  366. """The filament a candidate needs, as ``(types, overrides)``.
  367. Both columns are JSON text written by the slicer step. Malformed content is
  368. treated as no constraint rather than as an error: a job whose overrides
  369. cannot be parsed still prints, it just gets no filament-based narrowing.
  370. Overrides carry their own types, so the returned type list is the union of
  371. the two — an override on one slot must not drop the requirements of the
  372. slots it says nothing about.
  373. Shared by the matcher and by the smart-plug wake step so both ask a printer
  374. for the same filament (#2876). Waking a printer the matcher would then
  375. reject on colour is the bug this exists to prevent.
  376. """
  377. required_types = None
  378. if candidate.required_filament_types:
  379. try:
  380. required_types = json.loads(candidate.required_filament_types)
  381. except json.JSONDecodeError:
  382. pass
  383. filament_overrides = None
  384. if candidate.filament_overrides:
  385. try:
  386. filament_overrides = json.loads(candidate.filament_overrides)
  387. except json.JSONDecodeError:
  388. pass
  389. effective_types = required_types
  390. if filament_overrides:
  391. override_types = sorted({o["type"] for o in filament_overrides if "type" in o})
  392. if override_types:
  393. effective_types = sorted(set(required_types or []) | set(override_types))
  394. return effective_types, filament_overrides
  395. def _could_take_printer(item: PrintQueueItem, printer_id: int, printer_model: str | None) -> bool:
  396. """Whether ``item`` was competing for the printer another item just took.
  397. The SJF starvation guard marks a job as jumped when a shorter one lower in
  398. the queue takes a printer it wanted. A pinned job and an "Any <model>" job
  399. compete for the same printer, so the guard has to look across both lanes
  400. (#3200); looking only inside the dispatched item's own lane let a stream of
  401. short model-based jobs hold a longer pinned job back indefinitely.
  402. Location filters are not checked: over-marking only lifts an item that was
  403. going to wait anyway, while under-marking is the starvation this prevents.
  404. """
  405. if item.printer_id is not None:
  406. return item.printer_id == printer_id
  407. if not printer_model:
  408. return False
  409. wanted = (normalize_printer_model(printer_model) or printer_model).lower()
  410. models = [v.target_model for v in item.variants] if item.variants else [item.target_model]
  411. return any(m and (normalize_printer_model(m) or m).lower() == wanted for m in models)
  412. def _candidates_for(item: PrintQueueItem) -> list[_ModelCandidate]:
  413. """Candidate files for ``item``, best first.
  414. An item with no variant rows yields exactly one candidate built from its own
  415. columns — the pre-#671 behaviour, unchanged.
  416. Variants come back least-attempted first, ties broken by the user's
  417. ``position``. On the first pass every count is zero, so this is purely the
  418. user's priority order. After a start-watchdog bounce the printer that failed
  419. drops behind, so the next lap tries the other machine rather than spending the
  420. item's whole retry budget on the one that is wedged. Once every candidate has
  421. been tried equally often they cycle again, which keeps the item-level
  422. ``DISPATCH_MAX_ATTEMPTS`` bound from #2555 intact — a job with alternatives
  423. still gives up, it just does not give up without trying them.
  424. """
  425. if not item.variants:
  426. if not item.archive_id and not item.library_file_id:
  427. # Nothing to print at all. Dispatching would fail deep in the upload
  428. # on "No archive_id or library_file_id"; the caller holds the item
  429. # with an explanation instead.
  430. return []
  431. return [
  432. _ModelCandidate(
  433. target_model=item.target_model,
  434. sliced_for=_sliced_for_model(item.archive, item.library_file),
  435. required_filament_types=item.required_filament_types,
  436. filament_overrides=item.filament_overrides,
  437. )
  438. ]
  439. # Drop candidates whose file is gone or in the trash. Both are reachable and
  440. # neither is covered by the schema: library deletes are soft (the row lives
  441. # on with ``deleted_at`` set, which no foreign key can express), and SQLite
  442. # ships with ``PRAGMA foreign_keys`` off, so the ON DELETE CASCADE never
  443. # fires there and a hard delete leaves the variant row pointing at nothing.
  444. usable = [v for v in item.variants if v.library_file is not None and v.library_file.deleted_at is None]
  445. ordered = sorted(usable, key=lambda v: (v.attempt_count or 0, v.position, v.id))
  446. return [
  447. _ModelCandidate(
  448. target_model=v.target_model,
  449. sliced_for=_sliced_for_model(None, v.library_file),
  450. required_filament_types=v.required_filament_types,
  451. filament_overrides=v.filament_overrides,
  452. variant=v,
  453. )
  454. for v in ordered
  455. ]
  456. def _collapse_waiting_reasons(per_model: list[tuple[str | None, str]]) -> str | None:
  457. """Fold one waiting reason per candidate into a single line for the item.
  458. A cross-model item produces a reason per candidate, and pasting them
  459. together unlabelled reads as gibberish ("No idle printer; PETG not loaded"
  460. — on which machine?). Each reason is prefixed with its model, except in the
  461. single-candidate case where the item already displays its target model and
  462. the prefix would be noise.
  463. Identical reasons collapse rather than repeat, so three idle-less models
  464. read as one clause.
  465. When *every* candidate is merely busy the parts are joined with the ``" | "``
  466. separator :meth:`PrintScheduler._is_busy_only` already parses, and left
  467. unprefixed. That case must keep testing busy-only: a fleet that is simply
  468. printing needs no user action, and labelling the clauses would turn each pass
  469. over a two-model item into a "job waiting" notification.
  470. """
  471. reasons = [(model, reason) for model, reason in per_model if reason]
  472. if not reasons:
  473. return None
  474. if len(reasons) == 1:
  475. return reasons[0][1]
  476. distinct = list(dict.fromkeys(reason for _model, reason in reasons))
  477. if len(distinct) == 1:
  478. return distinct[0]
  479. if all(PrintScheduler._is_busy_only(reason) for _model, reason in reasons):
  480. return " | ".join(distinct)
  481. return "; ".join(f"{model or 'unassigned'}: {reason}" for model, reason in reasons)
  482. def _candidate_model_label(candidates: list[_ModelCandidate]) -> str | None:
  483. """Human label for the models an item is waiting on ("H2S or H2C").
  484. Notifications take a single target model. For a cross-model item the item's
  485. own ``target_model`` is whichever variant happens to be first, which reads as
  486. a lie once it is the H2C that actually runs — so name all of them.
  487. """
  488. models = list(dict.fromkeys(c.target_model for c in candidates if c.target_model))
  489. if not models:
  490. return None
  491. return " or ".join(models)
  492. def _mapping_is_all_unresolved(mapping: list | None) -> bool:
  493. """True if ``mapping`` is a non-empty list whose every entry is the
  494. unresolved sentinel (-1 / None) — i.e. no required slot ever matched a tray.
  495. Such a mapping is a bug artifact: a frontend status-load race can serialize
  496. ``[-1]`` before the printer's AMS trays are known (#2589). It must be
  497. recomputed from live status at dispatch rather than trusted, otherwise it
  498. reaches the print command and is silently downgraded to external-spool mode.
  499. A partially-resolved mapping (``[-1, -1, 5]`` where slot 3 matched, or a
  500. padding ``-1`` for a slot this plate does not print) is NOT unresolved. An
  501. explicit external selection (``>= 254``) is NOT unresolved either — those
  502. keep their meaning.
  503. """
  504. if not isinstance(mapping, list) or not mapping:
  505. return False
  506. return all(t is None or (isinstance(t, int) and t < 0) for t in mapping)
  507. def _is_tray_id(value: object) -> bool:
  508. """True when ``value`` can be read as a global tray ID.
  509. ``bool`` is a subclass of ``int``, so a hand-written API payload carrying
  510. ``true`` would otherwise be read as tray 1 and judged — or dispatched —
  511. against whatever happens to be loaded there.
  512. """
  513. return isinstance(value, int) and not isinstance(value, bool)
  514. # Prefix of the waiting_reason `_block_on_unmatched_filament` writes when it
  515. # stages an item (#2799). The manual_start branch clears every other reason on
  516. # a staged item (#3074); this one is the reason it was staged, so it stays.
  517. _UNMATCHED_HOLD_PREFIX = "Needs "
  518. def _is_unmatched_hold_reason(reason: str | None) -> bool:
  519. """True for the reason the unmatched-filament hold wrote when it staged the item."""
  520. return bool(reason) and reason.startswith(_UNMATCHED_HOLD_PREFIX)
  521. def _unresolved_required(required: list[dict], mapping: list) -> list[dict]:
  522. """The requirements in ``required`` that ``mapping`` gives no tray (#2799).
  523. Only slots the plate prints are judged: a ``-1`` anywhere else is padding
  524. for a filament this plate does not use. A slot past the end of the mapping
  525. is unresolved too.
  526. """
  527. unresolved = []
  528. for req in required:
  529. slot_id = req.get("slot_id") or 0
  530. if slot_id <= 0:
  531. continue
  532. tray = mapping[slot_id - 1] if slot_id <= len(mapping) else None
  533. if not _is_tray_id(tray) or tray < 0:
  534. unresolved.append(req)
  535. return unresolved
  536. # Global tray ids at or above this are the external spool(s), not an AMS slot:
  537. # 254 is the deputy feed and 255 the main one. Mirrors the sentinel documented
  538. # on `_mapping_is_all_unresolved`.
  539. _EXTERNAL_TRAY_ID_MIN = 254
  540. def _consumed_mapping_entries(mapping: list | None, required: list[dict] | None) -> list | None:
  541. """The ``mapping`` entries for the slots this plate actually prints.
  542. ``required`` comes from ``extract_filament_requirements``, which drops any
  543. filament with ``used_g <= 0`` — so a slot_id present there is one the plate
  544. consumes, and one absent from it is padding. That distinction is why this
  545. decision lives here and not in the MQTT command builder: a ``-1`` in the
  546. mapping means either "this plate does not print filament N" or "we never
  547. worked out which tray", and only the plate's own filament list separates
  548. them. The builder sees both as the same byte, which is how a plate whose one
  549. printed filament sat on the external spool went out as `use_ams=true` with a
  550. mapping of nothing but -1 and stalled at preheat until the firmware gave up
  551. with 07FF_8012 (#3087).
  552. Returns None whenever the two cannot be lined up — no mapping, no parsed
  553. requirements, or a requirement the mapping is too short to cover — so every
  554. caller falls back to existing behaviour rather than acting on a guess.
  555. """
  556. if not isinstance(mapping, list) or not mapping or not required:
  557. return None
  558. entries = []
  559. for filament in required:
  560. slot_id = filament.get("slot_id")
  561. if not isinstance(slot_id, int) or not 1 <= slot_id <= len(mapping):
  562. # The mapping and the requirements disagree about how many filaments
  563. # the file has. They came from different reads, so judge nothing.
  564. return None
  565. entries.append(mapping[slot_id - 1])
  566. return entries or None
  567. def _is_external_tray(tray_id) -> bool:
  568. """True for an explicit external-spool selection (254/255), not for an
  569. unresolved slot and not for an AMS tray."""
  570. if tray_id is None:
  571. return False
  572. try:
  573. return int(tray_id) >= _EXTERNAL_TRAY_ID_MIN
  574. except (TypeError, ValueError):
  575. return False
  576. def _might_be_dual_nozzle(printer_model: str | None, status) -> bool:
  577. """Whether this printer could have two extruders, judged generously.
  578. On a dual-nozzle printer ``use_ams`` is nozzle routing rather than an
  579. AMS on/off flag — H2D Pro firmware reads it as an extruder index — which is
  580. why the MQTT command builder skips its own use_ams reconcile there. Anything
  581. that might be dual-nozzle therefore keeps whatever ``use_ams`` it arrived
  582. with, external spools or not.
  583. Deliberately over-eager: a wrong "yes" only means this printer keeps the
  584. behaviour it has always had, while a wrong "no" would rewrite a field that
  585. steers which nozzle prints. The model name is the first answer (it is what
  586. the command builder falls back to as well), then the same live evidence the
  587. dispatcher's extruder annotation uses — a second nozzle reporting a
  588. diameter, a populated ``ams_extruder_map``, or more than one external feed,
  589. since a single-nozzle printer has exactly one.
  590. """
  591. if is_dual_nozzle_model(printer_model):
  592. return True
  593. nozzles = getattr(status, "nozzles", None) or []
  594. if len(nozzles) > 1 and getattr(nozzles[1], "nozzle_diameter", ""):
  595. return True
  596. raw = getattr(status, "raw_data", None) or {}
  597. if raw.get("ams_extruder_map"):
  598. return True
  599. vt_trays = raw.get("vt_tray") or []
  600. return isinstance(vt_trays, list) and len(vt_trays) > 1
  601. def _int_or(value, default: int) -> int:
  602. """``int(value)``, or ``default`` when the field is missing or junk.
  603. AMS telemetry types its ids inconsistently — `"0"` in one firmware, `0` in
  604. the next — and a tray id that fails to parse must not take the whole
  605. derivation down with it.
  606. """
  607. try:
  608. return int(value)
  609. except (TypeError, ValueError):
  610. return default
  611. def _global_tray_id(ams_id: int, tray_id: int) -> int:
  612. """Bambu's flat tray addressing: ``ams_id * 4 + tray_id`` for a four-slot
  613. unit, and the bare unit id for an AMS-HT (ids from 128, one tray each).
  614. Mirrors the calculation in ``_build_loaded_filaments``, which is what
  615. produces the ids stored in ``PrintQueueItem.ams_mapping`` — the two must
  616. agree or a mapping cannot be read back against live tray telemetry.
  617. """
  618. return ams_id if ams_id >= 128 else ams_id * 4 + tray_id
  619. def _used_global_tray_ids(item: PrintQueueItem | None) -> set[int] | None:
  620. """The global tray ids ``item`` actually prints from, or None if unknown.
  621. ``ams_mapping`` is the array the print command carries: position = filament
  622. slot, value = global tray id, ``-1`` / ``None`` for a slot this plate does
  623. not use. None means "no usable statement" — no mapping, unparseable JSON,
  624. an all-unresolved mapping (the artifact ``_mapping_is_all_unresolved``
  625. documents), or one that resolves to no tray at all. Callers must treat None
  626. as "consider every loaded tray" rather than "consider none": narrowing on
  627. an absent mapping would silently drop requirements the print really has.
  628. """
  629. raw = getattr(item, "ams_mapping", None)
  630. if not raw:
  631. return None
  632. if isinstance(raw, str):
  633. try:
  634. mapping = json.loads(raw)
  635. except (json.JSONDecodeError, TypeError):
  636. return None
  637. else:
  638. mapping = raw
  639. if not isinstance(mapping, list) or _mapping_is_all_unresolved(mapping):
  640. return None
  641. used = {t for t in mapping if isinstance(t, int) and not isinstance(t, bool) and t >= 0}
  642. return used or None
  643. def _mqtt_commands_rejected(status) -> bool:
  644. """True when the printer is currently reporting that it refused a command.
  645. ``HMS_MQTT_VERIFY_FAILED`` means the firmware's authorization check rejected
  646. a control command it could not verify. Queries still answer, so the printer
  647. looks connected and idle while project_file, gcode_line and
  648. ams_change_filament are all dropped — no amount of waiting or re-uploading
  649. changes that (#2732).
  650. Tolerates a missing status and errors without a ``full_code`` (the 8-char
  651. ``print_error`` path builds HMSError differently), so this is safe to call on
  652. every watchdog poll.
  653. """
  654. for err in getattr(status, "hms_errors", None) or []:
  655. if getattr(err, "full_code", "") == HMS_MQTT_VERIFY_FAILED:
  656. return True
  657. return False
  658. def _drying_ams_ids(status) -> list[int]:
  659. """AMS unit ids currently running a drying cycle, per firmware telemetry.
  660. ``dry_time`` is minutes remaining, so >0 is the firmware's own statement that
  661. a cycle is active. Used by the dispatch watchdog to say *why* a print never
  662. started (#2758) — it is a diagnostic, not a gate.
  663. Deliberately not used to block or stop drying before dispatch. This printer
  664. class supports drying concurrently with an active print
  665. (``supports_drying_while_printing``), so drying is not incompatible with
  666. printing in general; what #2758 shows is one X2D refusing to *begin* a print
  667. while two AMS units were drying, one of them without its external PSU. Until
  668. it is known whether the blocker is drying itself or the power budget
  669. (``dry_sf_reason`` 1 / 8), acting on this would tear down drying that the
  670. hardware is perfectly happy to continue.
  671. """
  672. ids: list[int] = []
  673. for unit in (getattr(status, "raw_data", None) or {}).get("ams") or []:
  674. if not isinstance(unit, dict):
  675. continue
  676. try:
  677. if int(unit.get("dry_time") or 0) > 0:
  678. ids.append(int(unit.get("id", 0)))
  679. except (TypeError, ValueError):
  680. continue
  681. return ids
  682. def _parse_diameter(raw) -> float | None:
  683. """``"0.4"`` → ``0.4``; anything unparseable or non-positive → None."""
  684. try:
  685. value = float(raw)
  686. except (TypeError, ValueError):
  687. return None
  688. return value if value > 0 else None
  689. def _nozzle_info_by_id(status) -> dict[int, dict]:
  690. """Index ``PrinterState.nozzle_rack`` by nozzle id.
  691. The field name is historical: on the H2 series ``nozzle_info`` carries an
  692. entry for *every* nozzle the printer knows about — the L/R hotends under
  693. ids 0/1 and, on a rack model, the dock positions under ids 16-21. Only the
  694. latter are the rack proper; :func:`_rack_nozzle_diameters` and
  695. :func:`_installed_nozzle_diameters` each take the half they need.
  696. """
  697. by_id: dict[int, dict] = {}
  698. for entry in getattr(status, "nozzle_rack", None) or []:
  699. if not isinstance(entry, dict):
  700. continue
  701. try:
  702. by_id[int(entry.get("id"))] = entry
  703. except (TypeError, ValueError):
  704. continue
  705. return by_id
  706. # What the firmware puts in a nozzle's serial number when the carriage or dock
  707. # is empty. Measured on an H2C at idle, where the rack-side hotend had parked
  708. # its nozzle back in the rack (#2885).
  709. _EMPTY_NOZZLE_SERIAL = "N/A"
  710. def _nozzle_is_mounted(entry: dict | None) -> bool:
  711. """Whether a ``nozzle_info`` entry describes hardware that is actually there.
  712. A hotend entry is reported whether or not a nozzle is mounted, and an empty
  713. carriage keeps the *last* nozzle's diameter — measured on an H2C at idle,
  714. where id 1 read ``diameter "0.4"`` with ``max_temp 0``, ``serial_number
  715. "N/A"`` and ``wear 0`` because that hotend had parked its nozzle back in
  716. the dock. So the diameter is not a presence signal (#2885).
  717. Emptiness has to be stated, not merely unstated. The serial must be the
  718. firmware's explicit ``"N/A"`` marker *and* the temperature rating must be
  719. absent — an empty serial only means the printer didn't say, and a firmware
  720. that reports neither field normalises to exactly that. Treating "didn't
  721. say" as "empty" would silently switch the #1899 guard off on any such
  722. machine, so everything we cannot positively call empty counts as mounted.
  723. """
  724. if entry is None:
  725. return True
  726. serial = str(entry.get("serial_number") or "").strip().upper()
  727. if serial != _EMPTY_NOZZLE_SERIAL:
  728. return True
  729. try:
  730. max_temp = float(entry.get("max_temp") or 0)
  731. except (TypeError, ValueError):
  732. return True
  733. return max_temp > 0
  734. def _installed_nozzle_diameters(status) -> list[float]:
  735. """Parse the mounted nozzle diameters from a PrinterState (#1899).
  736. Returns the diameters the printer actually reports (e.g. [0.4] single-nozzle,
  737. [0.4, 0.6] dual-nozzle), skipping the empty-string defaults that populate a
  738. NozzleInfo before MQTT fills it in. An empty list means "the printer hasn't
  739. told us its nozzle hardware" — callers must treat that as unknown, not as a
  740. mismatch, so we never block a print on missing data.
  741. A hotend whose ``nozzle_info`` entry says nothing is mounted is skipped even
  742. though ``nozzles`` still carries a diameter for it: that value is stale, and
  743. counting it would let a slice match a nozzle the machine does not have
  744. (#2885). Printers that report no ``nozzle_info`` at all are unaffected.
  745. """
  746. info = _nozzle_info_by_id(status)
  747. diameters: list[float] = []
  748. for index, nozzle in enumerate(getattr(status, "nozzles", None) or []):
  749. raw = getattr(nozzle, "nozzle_diameter", "") or ""
  750. try:
  751. value = float(raw)
  752. except (TypeError, ValueError):
  753. continue
  754. if value > 0 and _nozzle_is_mounted(info.get(index)):
  755. diameters.append(value)
  756. return diameters
  757. def _rack_nozzle_diameters(status) -> list[float]:
  758. """Diameters sitting in the tool-changer rack, nearest dock first (#2885).
  759. Keyed off the nozzle ids themselves rather than the printer model: only a
  760. rack machine ever reports ids 16-21, so there is no model registry to keep
  761. in sync. An empty dock is simply absent from the payload — measured on an
  762. H2C whose R2 (id 17) was empty and unlisted while R1/R3/R4/R5/R6 were all
  763. present — so appearing here already means "a nozzle is in that dock".
  764. ``stat`` is deliberately not interpreted: its values aren't known, and
  765. reading it wrongly could hide a nozzle the printer would happily fetch.
  766. """
  767. diameters: list[float] = []
  768. for nozzle_id, entry in sorted(_nozzle_info_by_id(status).items()):
  769. if nozzle_id not in _RACK_NOZZLE_IDS:
  770. continue
  771. # PrinterState spells it "diameter"; the REST schema renames it to
  772. # "nozzle_diameter". Accept either so a caller holding the serialised
  773. # shape gets the same answer.
  774. value = _parse_diameter(entry.get("diameter") or entry.get("nozzle_diameter"))
  775. if value is not None:
  776. diameters.append(value)
  777. return diameters
  778. def _format_diameters(diameters: list[float]) -> str:
  779. """``[0.4, 0.4, 0.6]`` → ``"0.4mm / 0.6mm"``, in first-seen order.
  780. Deduplicated because a loaded rack holds several nozzles of the same size,
  781. and "0.4mm / 0.4mm / 0.4mm / 0.6mm / 0.2mm" tells the reader nothing the
  782. short form doesn't. Matching still runs over the full list.
  783. """
  784. return " / ".join(f"{d:g}mm" for d in dict.fromkeys(diameters))
  785. def _nozzle_mismatch_message(
  786. sliced_nozzle: float | None,
  787. installed: list[float],
  788. rack: list[float] | None = None,
  789. ) -> str | None:
  790. """Return an actionable error message when the sliced nozzle can't be
  791. printed on any nozzle the machine can reach, else None (#1899).
  792. Fail-safe: returns None whenever we lack the data to judge — no sliced
  793. diameter, or the printer reported no nozzles — so a print is only ever
  794. blocked on a POSITIVE mismatch. On dual-nozzle printers a match against
  795. EITHER installed nozzle passes (a 0.6 slice is fine if one hotend is 0.6).
  796. The 0.05 tolerance absorbs float noise while staying well inside the 0.2
  797. gap between adjacent nozzle sizes (0.2/0.4/0.6/0.8).
  798. *rack* holds the diameters parked in a tool-changer dock (H2C). Those count
  799. as reachable: the printer fetches one as part of starting the print, so a
  800. slice that matches a docked nozzle is not a mismatch. Without this the
  801. guard blocked every job whose nozzle happened not to be on a hotend at
  802. dispatch time — on a rack loaded with 0.2/0.4/0.6 that meant only the
  803. diameter already mounted could ever print, and the user had to fetch the
  804. nozzle by hand on the printer's own UI first (#2885).
  805. """
  806. reachable = [*installed, *(rack or [])]
  807. if not sliced_nozzle or not reachable:
  808. return None
  809. if any(abs(d - sliced_nozzle) < 0.05 for d in reachable):
  810. return None
  811. where = f"{_format_diameters(installed)} installed" if installed else "no nozzle mounted"
  812. if rack:
  813. where += f" and {_format_diameters(rack)} in the nozzle rack"
  814. return (
  815. f"File sliced for a {sliced_nozzle:g}mm nozzle, but the printer has "
  816. f"{where}. Re-slice for an available nozzle, or fit the matching "
  817. f"nozzle before printing."
  818. )
  819. def _describe_filament(entry: dict, nozzle_key: str) -> str:
  820. """One-line "PETG #000000 (left nozzle)" for an error message (#2771).
  821. Shared by the required and loaded sides, which name their extruder
  822. differently: a 3MF requirement carries ``nozzle_id``, a loaded tray carries
  823. ``extruder_id``. Both are MQTT extruder ids — 0 is the right/main nozzle,
  824. 1 the left/deputy — and both are absent on single-nozzle printers, where
  825. naming a nozzle would be noise.
  826. """
  827. parts = [(entry.get("type") or "filament").upper()]
  828. if entry.get("color"):
  829. parts.append(str(entry["color"]))
  830. nozzle = entry.get(nozzle_key)
  831. if nozzle == 0:
  832. parts.append("(right nozzle)")
  833. elif nozzle == 1:
  834. parts.append("(left nozzle)")
  835. return " ".join(parts)
  836. def _unmatched_filament_message(required: list[dict], loaded: list[dict]) -> str:
  837. """Explain that nothing loaded matches what the file needs (#2771).
  838. Only ever built for a printer with no AMS, where the loaded list is short
  839. enough to quote in full and there is no "load another spool and hit Resume"
  840. recovery — the external spool holder is all there is, so the user needs to
  841. be told which filament to put on it.
  842. """
  843. want = ", ".join(_describe_filament(r, "nozzle_id") for r in required)
  844. have = ", ".join(_describe_filament(f, "extruder_id") for f in loaded)
  845. return (
  846. f"No filament loaded on this printer matches the file. It needs {want}; "
  847. f"the printer has {have} and no AMS. Load the required filament on the "
  848. f"external spool holder, or send this job to a printer that has it."
  849. )
  850. def _effective_plate_id(explicit_plate_id: int | None, file_path: Path) -> int:
  851. """The plate to dispatch, resolved once in ``_start_print`` and reused at
  852. every call site below it: G-code injection, usage registration, rack-plan
  853. lookup, slot-extruder lookup, the external-spool check, and the actual
  854. print command.
  855. A positive explicit ``plate_id`` on the queue item always wins. A
  856. non-positive one is treated as "not set" and resolved from the archive,
  857. the way the rest of the queue code already reads it (``if item.plate_id:``
  858. in ``api/routes/print_queue.py``): the schemas put no lower bound on the
  859. field, and passing a 0 straight through would build a print command for
  860. ``Metadata/plate_0.gcode``, which is the same wedge this function exists
  861. to prevent. The ``item.plate_id or 1`` this replaced mapped 0 to 1.
  862. Falling back to a bare ``1`` instead of reading the archive assumes a
  863. single-plate file's one G-code is numbered 1, which only holds for a
  864. plate exported on its own: one cut out of a larger project keeps its
  865. ORIGINAL plate number, so a printer asked to print "plate 1" of a file
  866. whose only G-code is ``plate_2.gcode`` accepts the command, can't find
  867. the file, throws an HMS error, and sits wedged in IDLE until
  868. power-cycled (#2947).
  869. The call sites agreed on a fallback only by accident before this:
  870. with G-code injection on, ``inject_gcode_into_3mf`` already falls back to
  871. the archive's own default plate internally whenever the plate id it's
  872. handed isn't in the file, so it could silently inject into a different
  873. plate than the one the print command itself asked for.
  874. Falls back to 1 when the archive can't be read, holds no G-code member at
  875. all, or its default member doesn't follow the ``plate_N`` naming
  876. convention (a slicer that doesn't use it has no number to dispatch).
  877. An explicit plate the archive doesn't hold is logged and then sent
  878. anyway. It wedges the printer exactly like #2947 did, but redirecting it
  879. to a plate that is in the file would print a model nobody asked for,
  880. which is the worse of the two.
  881. """
  882. try:
  883. with zipfile.ZipFile(file_path, "r") as zf:
  884. names = zf.namelist()
  885. except (OSError, zipfile.BadZipFile) as exc:
  886. logger.warning(
  887. "Dispatch plate: cannot read %s (%s), so the archive's own plate numbering "
  888. "is unavailable; dispatching plate %s",
  889. file_path,
  890. exc,
  891. explicit_plate_id if explicit_plate_id is not None and explicit_plate_id > 0 else 1,
  892. )
  893. names = None
  894. if explicit_plate_id is not None and explicit_plate_id > 0:
  895. if (
  896. names is not None
  897. and default_plate_number(names) is not None
  898. and select_plate_gcode_name(names, explicit_plate_id) is None
  899. ):
  900. logger.warning(
  901. "Dispatch plate: %s was queued for plate %s but holds no G-code for it "
  902. "(it has %s). Sending plate %s as asked; expect the printer to reject the "
  903. "file, since printing a different plate would print the wrong model (#2947)",
  904. file_path,
  905. explicit_plate_id,
  906. ", ".join(sorted(n for n in names if n.endswith(".gcode"))),
  907. explicit_plate_id,
  908. )
  909. return explicit_plate_id
  910. resolved = default_plate_number(names) if names is not None else None
  911. return resolved if resolved is not None else 1
  912. class PrintScheduler:
  913. """Background scheduler that processes the print queue."""
  914. # Built-in drying presets per filament type (from BambuStudio filament profiles)
  915. # Format: { n3f_temp, n3s_temp, n3f_hours, n3s_hours }
  916. DEFAULT_DRYING_PRESETS: dict[str, dict[str, int]] = {
  917. "PLA": {"n3f": 45, "n3s": 45, "n3f_hours": 12, "n3s_hours": 12},
  918. "PETG": {"n3f": 65, "n3s": 65, "n3f_hours": 12, "n3s_hours": 12},
  919. "TPU": {"n3f": 65, "n3s": 75, "n3f_hours": 12, "n3s_hours": 18},
  920. "ABS": {"n3f": 65, "n3s": 80, "n3f_hours": 12, "n3s_hours": 8},
  921. "ASA": {"n3f": 65, "n3s": 80, "n3f_hours": 12, "n3s_hours": 8},
  922. "PA": {"n3f": 65, "n3s": 85, "n3f_hours": 12, "n3s_hours": 12},
  923. "PC": {"n3f": 65, "n3s": 80, "n3f_hours": 12, "n3s_hours": 8},
  924. "PVA": {"n3f": 65, "n3s": 85, "n3f_hours": 12, "n3s_hours": 18},
  925. }
  926. def __init__(self):
  927. self._running = False
  928. self._check_interval = 30 # seconds
  929. # After a pass that actually dispatched something, loop again almost
  930. # immediately instead of sleeping the full interval (#2555). A dispatch
  931. # changes printer state — a batch launch fans out over several passes as
  932. # printers free up, a wedged head-of-line job reverts to pending, an
  933. # upload slot opens — and the next batch of ready work should not have to
  934. # wait 30 s behind an idle sleep. When a pass dispatches nothing (all
  935. # pending items are behind printers that are genuinely busy printing),
  936. # there is nothing to react to, so we fall back to the normal interval;
  937. # that also means this can never tight-loop, since fast ticks only
  938. # continue while dispatches keep happening and the queue is draining.
  939. self._fast_check_interval = 3 # seconds
  940. self._power_on_wait_time = 180 # seconds to wait for printer after power on (3 min)
  941. self._power_on_check_interval = 10 # seconds between connection checks
  942. # Printers whose class-target power-on failed, mapped to the monotonic
  943. # time their cool-off expires (#2786).
  944. #
  945. # Without this, one printer with an unreachable plug starves every
  946. # sibling of its model forever: the wake step walks candidates in id
  947. # order, spends the pass's single attempt on the same broken printer
  948. # every time, and the healthy one two slots down is never reached. It
  949. # also costs a full ``_power_on_wait_time`` out of every 30 s pass,
  950. # which delays the whole queue, not just this job.
  951. #
  952. # Entries expire on read rather than being cleared on success: a printer
  953. # inside its cool-off is skipped before the power-on is reached, so a
  954. # live entry can never be overwritten by a success anyway. A printer
  955. # that comes back by any other route stops being a wake candidate the
  956. # moment it connects.
  957. self._wake_failures: dict[int, float] = {}
  958. self._wake_failure_cooloff = 600 # seconds
  959. # Printers whose last upload never reached them, mapped to the monotonic
  960. # time they may be dispatched to again (#3210). Same expire-on-read shape
  961. # as `_wake_failures`. Without it, a printer that cannot take files is
  962. # idle on every pass, so it is picked on every pass.
  963. self._upload_backoff: dict[int, float] = {}
  964. # Refused uploads in a row per printer, which sets the next backoff
  965. # window. Cleared by a successful upload to that printer.
  966. self._upload_refusals: dict[int, int] = {}
  967. # Printers whose current run of refusals has already sent its one
  968. # "job waiting" notification. Each retry clears the item's waiting
  969. # reason and the next refusal sets it again, which `hold_item` reads as
  970. # a new reason; without this, every retry would notify. Cleared with
  971. # `_upload_refusals`.
  972. self._upload_refusal_notified: set[int] = set()
  973. # Track which printers are currently auto-drying (printer_id -> start timestamp)
  974. self._drying_in_progress: dict[int, float] = {}
  975. # Per-AMS memory of the auto-drying cycles WE armed, keyed by
  976. # (printer_id, ams_id) (#2770). Entries only exist between arming a
  977. # cycle and the humidity finally coming down, so the normal steady
  978. # state is an empty dict. Fields:
  979. # running — a cycle we armed is (or should be) on the firmware
  980. # ended_at — monotonic when we observed that cycle end
  981. # unproductive — consecutive armed cycles that ended with the reading
  982. # still above the threshold
  983. # suspended — we have stopped arming this unit and said so
  984. self._auto_dry_units: dict[tuple[int, int], dict[str, object]] = {}
  985. # Sustained-humidity streaks for the ambient-drying wait (#2518). Keyed
  986. # like _auto_dry_units but deliberately a SEPARATE dict: membership in
  987. # _auto_dry_units means "Bambuddy armed a cycle on this unit", and the
  988. # print-takes-priority stop, the manual-cycle adoption guard, and the
  989. # arming setdefault all act on that meaning -- a unit that is merely
  990. # waiting out its streak must not become stoppable or judgeable.
  991. # since -- monotonic stamp when the current continuous above-threshold
  992. # streak began
  993. # last -- monotonic stamp of the last pass that observed the streak
  994. self._auto_dry_above: dict[tuple[int, int], dict[str, float]] = {}
  995. # Slots already notified as low on filament:
  996. # {(printer_id, ams_id, tray_id, spool_id)}.
  997. # Cleared when the slot goes back above its threshold rather than on a
  998. # timer, mirroring _notified_hms_errors — a spool hovering at the
  999. # boundary must not produce an alert on every pass, and a slot that has
  1000. # been refilled has to be able to alert again. See #2913.
  1001. #
  1002. # The spool id is part of the key because the slot alone cannot clear
  1003. # itself. Pull a low spool and put a different part-used one in the same
  1004. # slot: the key survives the pass where the slot resolves to nothing --
  1005. # deliberately, so a brief Spoolman outage does not re-alert everything
  1006. # when it returns -- and the replacement never goes above the threshold,
  1007. # so it can never re-arm. Keyed with the spool, the new spool is simply a
  1008. # different key and the stale one is inert.
  1009. self._notified_filament_low: set[tuple[int, int, int, int]] = set()
  1010. # Earliest monotonic time the next low-filament check may run (#2913).
  1011. self._filament_low_next_check: float = 0.0
  1012. # SKU key -> (condition, events told) for a SKU that is alerting (#2955): the
  1013. # condition is "reorder" or "break", the events are those that had a
  1014. # subscriber when it was sent. Absent means not alerting. Cleared when the
  1015. # condition clears, so it can alert again, and when nobody wants either
  1016. # event. Mirrored to one settings row (_STOCK_ALERTS_SETTING_KEY), because the
  1017. # first check runs as soon as Bambuddy starts and a restart would otherwise
  1018. # re-send one message per SKU that is still low.
  1019. self._notified_stock_alerts: dict[stock_forecast.SkuKey, tuple[str, frozenset[str]]] = {}
  1020. # The JSON last read from or written to that row; None until it has been read.
  1021. self._stock_alerts_persisted: str | None = None
  1022. # Earliest monotonic time the next stock forecast check may run (#2955).
  1023. self._stock_forecast_next_check: float = 0.0
  1024. # Printers with a "running" scheduled drying row (#2638). Rebuilt from the
  1025. # DB on every _check_scheduled_dryings call so route-side cancels show up.
  1026. # Auto-drying's stop-all branches must not stop or untrack these printers;
  1027. # both features share _drying_in_progress.
  1028. self._scheduled_drying_printer_ids: set[int] = set()
  1029. # Monotonic stamp of the last scheduled-drying prune. None = never, so
  1030. # the first pass after a restart reaps anything left behind.
  1031. self._last_scheduled_drying_prune: float | None = None
  1032. # Defensive in-memory dispatch hold (#1157): a printer that just received
  1033. # a project_file command must not get a second dispatch until either it
  1034. # transitions out of pre_state OR the hard timeout expires. The H2D Pro
  1035. # can take 80–210 s to flip FINISH→PREPARE after project_file, and
  1036. # during that window the DB busy_printers seed is empirically unreliable
  1037. # (multi-plate batches double-/triple-dispatched onto the same printer
  1038. # 30 s apart). Keyed by printer_id; cleared by the watchdog on success
  1039. # or revert.
  1040. # printer_id -> (monotonic_started_at, pre_state, pre_subtask_id)
  1041. self._dispatch_holds: dict[int, tuple[float, str, str | None]] = {}
  1042. # Minimum cooldown between dispatches to the same printer (covers the
  1043. # H2D's project_file digestion window).
  1044. self._dispatch_min_cooldown = 60.0
  1045. # Hard timeout — drop the hold even if we never observed a transition,
  1046. # so a lost MQTT session can't lock a printer out of the queue forever.
  1047. # Matches the watchdog timeout (90 s) plus a safety margin so the
  1048. # watchdog runs first on the unhappy path.
  1049. self._dispatch_max_hold = 180.0
  1050. # Refillable upload pool (#2602). Items whose FTP upload was launched by
  1051. # an earlier pass and is still running. `_start_print` flips the row
  1052. # pending -> printing only *after* the upload completes, so until then
  1053. # the row stays `pending`: each tick, check_queue excludes these
  1054. # item_ids from re-selection and their printers from new dispatch /
  1055. # auto-drying, and launches only `limit - len(_inflight)` new uploads so
  1056. # freed slots refill on the next fast tick. check_queue is the sole,
  1057. # sequential caller and the prune done-callbacks run in the same
  1058. # event-loop thread, so this dict needs no lock.
  1059. # item_id -> (task, printer_id)
  1060. self._inflight: dict[int, tuple[asyncio.Task, int | None]] = {}
  1061. # Expected prints registered by `_start_print` that have not yet had a
  1062. # print command sent. Populated at registration, dropped once
  1063. # `start_print()` succeeds, and rolled back by `_dispatch_one` on every
  1064. # other exit. Same threading argument as `_inflight` above: one
  1065. # sequential caller, callbacks on the same loop, so no lock.
  1066. # item_id -> (printer_id, remote_filename, archive_id)
  1067. self._unconfirmed_expected_print: dict[int, tuple[int, str, int]] = {}
  1068. # Budget reservations created for a dispatch whose print command has
  1069. # not been confirmed yet. `_dispatch_one` releases these on every
  1070. # unsuccessful exit; a successful start removes the item id and leaves
  1071. # the reservation for finance_billing to consume with the archive.
  1072. self._unconfirmed_budget_reservations: set[int] = set()
  1073. # Chamber temperature history for smart soak-time reduction.
  1074. # printer_id -> deque of (monotonic_timestamp, celsius) sampled each scheduler tick.
  1075. # Entries older than _chamber_history_ttl are pruned on write.
  1076. self._chamber_history: dict[int, deque[tuple[float, float]]] = {}
  1077. self._chamber_history_ttl = _CHAMBER_HISTORY_TTL_SECONDS
  1078. # Per-printer keep-warm state (see `_KeepWarmEntry` at module top).
  1079. # Populated on engagement in `_apply_keep_warm`, cleared by
  1080. # `_sweep_keep_warm` when the printer leaves the candidate set (or
  1081. # when a gate setting toggles off mid-hold — the release publishes
  1082. # bed → 0 first).
  1083. self._keep_warm: dict[int, _KeepWarmEntry] = {}
  1084. # Preheat rollback registry: printer_id -> subset of
  1085. # {"bed", "chamber", "airduct"} listing which preheat commands
  1086. # actually fired for the in-flight dispatch. `_dispatch_one` unwinds
  1087. # every entry still present at exit unless the print successfully
  1088. # started, so a failed upload / cancel / exception never leaves the
  1089. # printer heating for a job that isn't happening.
  1090. self._preheat_pin: dict[int, set[str]] = {}
  1091. # Bed target (°C) that the pinned "bed" entry above actually set, so the
  1092. # rollback can tell its own target from one someone else has since
  1093. # chosen — the same guard `_release_keep_warm` applies to a keep-warm
  1094. # hold. Written wherever `"bed"` joins the pin, evicted alongside it.
  1095. self._preheat_pin_bed: dict[int, int] = {}
  1096. # Item ids whose in-flight dispatch has been cancelled or deleted while
  1097. # its preheat was still holding at temperature. Set by
  1098. # `notify_dispatch_cancelled` from the queue routes, consumed by
  1099. # `_preheat_sleep`, and cleared when the dispatch exits.
  1100. self._cancelled_dispatches: set[int] = set()
  1101. # printer_id -> monotonic time it was first seen terminal while one of
  1102. # its queue rows was still 'printing'. Reset by any non-terminal
  1103. # observation, so it measures an unbroken run rather than a total.
  1104. # In-memory on purpose: a restart re-arms the grace period, which only
  1105. # delays a recovery that is already the exceptional path (#2829).
  1106. self._terminal_since: dict[int, float] = {}
  1107. # Per-pass memo for `_get_filament_requirements` (#2799 review). Two
  1108. # gates parse the same 3MF per dispatch attempt, and a one-file fan-out
  1109. # across idle printers repeats that for every one of them in a single
  1110. # pass. Cleared at the top of each pass so a re-sliced file is never
  1111. # served from a previous tick.
  1112. self._filament_req_memo: dict[tuple, list[dict] | None] = {}
  1113. async def run(self):
  1114. """Main loop - check queue every interval."""
  1115. self._running = True
  1116. logger.info("Print scheduler started")
  1117. await self._clear_stale_dispatch_claims(at_startup=True)
  1118. while self._running:
  1119. dispatched = False
  1120. try:
  1121. self._sample_chamber_temps()
  1122. # No-op while any upload is in flight; on a quiet tick it releases
  1123. # a claim whose best-effort clear failed (e.g. the database was
  1124. # briefly unreachable), instead of leaving the row wedged until
  1125. # the next restart.
  1126. await self._clear_stale_dispatch_claims()
  1127. await self._close_stranded_printing_items()
  1128. dispatched = await self.check_queue()
  1129. except Exception as e:
  1130. logger.error("Scheduler error: %s", e)
  1131. # Re-check quickly after a productive pass so a draining batch does
  1132. # not stall behind the idle interval; otherwise sleep normally (#2555).
  1133. await asyncio.sleep(self._fast_check_interval if dispatched else self._check_interval)
  1134. async def _close_stranded_printing_items(self) -> None:
  1135. """Close a ``printing`` row the completion event never closed (#2829).
  1136. ``on_print_complete`` refuses to close a row when the completion's
  1137. subtask name disagrees with the file the row was dispatched with, so a
  1138. completion meant for something else (the printer's own
  1139. ``auto_pa_line_calib_mode`` run, say) cannot end someone's job early.
  1140. The refusal has no way back, though: nothing else ever closes the row,
  1141. and ``check_queue`` treats every ``printing`` row as a busy printer, so
  1142. one bad comparison wedges that printer's queue until a human presses
  1143. cancel. That is what #2829's reporters hit, and the guard's own
  1144. docstring already called stranding the worse of the two failures.
  1145. This is the way back. When a row has been ``printing`` while its
  1146. printer sat in a terminal state for the whole grace period, the print
  1147. is over however the event was read, and the row is closed with the
  1148. status the printer's own state implies.
  1149. Deliberately conservative:
  1150. * Only a connected printer counts. A disconnected one has a stale
  1151. ``state`` and proves nothing.
  1152. * The clock is reset by any non-terminal observation, so this cannot
  1153. fire on a printer that is merely between stages.
  1154. * The grace period is far longer than the gap between a printer
  1155. finishing and its completion arriving, so the normal path always
  1156. wins the race and this only ever sees genuine strandings.
  1157. What it does *not* do is replay the completion's side effects --
  1158. notifications, billing, auto-off. It restores the queue, which is the
  1159. harm being undone; the archive was updated by the normal path
  1160. regardless, since only the queue block refuses. A recovery that
  1161. silently re-fired notifications minutes late would be its own bug.
  1162. """
  1163. try:
  1164. async with async_session() as db:
  1165. result = await db.execute(
  1166. select(PrintQueueItem)
  1167. .where(PrintQueueItem.status == "printing")
  1168. .where(PrintQueueItem.printer_id.is_not(None))
  1169. )
  1170. items = list(result.scalars().all())
  1171. if not items:
  1172. self._terminal_since.clear()
  1173. return
  1174. now = time.monotonic()
  1175. seen_printers: set[int] = set()
  1176. closed = False
  1177. for item in items:
  1178. printer_id = item.printer_id
  1179. seen_printers.add(printer_id)
  1180. state = printer_manager.get_status(printer_id)
  1181. status = _terminal_queue_status(state)
  1182. if status is None:
  1183. self._terminal_since.pop(printer_id, None)
  1184. continue
  1185. since = self._terminal_since.setdefault(printer_id, now)
  1186. if now - since < _STRANDED_PRINTING_GRACE_SECONDS:
  1187. continue
  1188. item.status = status
  1189. item.completed_at = datetime.now(timezone.utc)
  1190. closed = True
  1191. logger.warning(
  1192. "Queue item %s was still 'printing' after printer %s reported %s for %.0fs — "
  1193. "closing it as %s. Its completion event was never matched to it, which blocks "
  1194. "every later job for this printer (#2829).",
  1195. item.id,
  1196. printer_id,
  1197. getattr(state, "state", None),
  1198. now - since,
  1199. status,
  1200. )
  1201. if closed:
  1202. await db.commit()
  1203. for printer_id in list(self._terminal_since):
  1204. if printer_id not in seen_printers:
  1205. del self._terminal_since[printer_id]
  1206. except Exception as e:
  1207. # Best-effort, same as the claim sweep beside it: a recovery path
  1208. # that can itself break the scheduler loop is worse than the strand.
  1209. logger.error("Stranded-item sweep failed: %s", e)
  1210. async def _clear_stale_dispatch_claims(self, *, at_startup: bool = False) -> None:
  1211. """Clear dispatch claims with no live dispatch coroutine behind them (#2615).
  1212. A claim is only ever held by a live dispatch coroutine, so when this
  1213. process has nothing in ``_inflight`` every ``dispatching_at`` in the table
  1214. is stale. At startup that is trivially true — no coroutine survives a
  1215. restart. It is equally true on any later tick where no upload is running,
  1216. which is what makes this safe to repeat rather than only run once.
  1217. Repeating it matters because ``_clear_dispatch_claim`` is best-effort: if
  1218. the database is briefly unreachable at exactly the moment dispatch ends,
  1219. the claim survives and the row is wedged out of the selection query. That
  1220. used to last until the next restart (#2702 follow-up, seen when
  1221. PostgreSQL refused a connection mid-dispatch).
  1222. ``_inflight`` is populated when the task is spawned, before the coroutine
  1223. claims its row, and pruned by a done-callback that cannot run before the
  1224. coroutine's own ``finally`` — so "claim present, nothing in flight" has no
  1225. race window and needs no age threshold. A size-derived upload deadline
  1226. (``max(600s, size/25KB/s)``) has no safe fixed bound anyway.
  1227. """
  1228. if self._inflight:
  1229. return
  1230. try:
  1231. async with async_session() as db:
  1232. res = await db.execute(
  1233. update(PrintQueueItem).where(PrintQueueItem.dispatching_at.is_not(None)).values(dispatching_at=None)
  1234. )
  1235. await db.commit()
  1236. if res.rowcount:
  1237. logger.info(
  1238. "Cleared %d orphaned dispatch claim(s)%s (#2615)",
  1239. res.rowcount,
  1240. " at startup" if at_startup else "",
  1241. )
  1242. except Exception as exc:
  1243. logger.error("Failed to clear orphaned dispatch claims: %s", exc)
  1244. def stop(self):
  1245. """Stop the scheduler."""
  1246. self._running = False
  1247. logger.info("Print scheduler stopped")
  1248. async def check_queue(self) -> bool:
  1249. """Check for prints ready to start.
  1250. Returns True if this pass dispatched at least one item, so the caller
  1251. can loop again quickly instead of sleeping the full interval (#2555).
  1252. """
  1253. # Scoped to one pass: a file re-sliced between ticks must be re-read.
  1254. self._filament_req_memo.clear()
  1255. async with async_session() as db:
  1256. # Check if shortest-job-first scheduling is enabled
  1257. sjf_enabled = await self._get_bool_setting(db, "queue_shortest_first")
  1258. # Get all pending items in the one order the queue page shows them
  1259. # in (#3200). The first eligible item takes a printer, so this order
  1260. # decides who wins a printer that a pinned job and an "Any <model>"
  1261. # job both want. It used to start with ``printer_id``, which made the
  1262. # lane outrank the position: SQLite sorts NULL first, so model-based
  1263. # jobs always won; PostgreSQL sorts it last, so pinned jobs did.
  1264. # Neither is what the user dragged into place.
  1265. if sjf_enabled:
  1266. # SJF: items already jumped get top priority (starvation guard),
  1267. # then sort by print_time ascending. Items with no print time go
  1268. # last, and position breaks ties.
  1269. result = await db.execute(
  1270. select(PrintQueueItem)
  1271. .where(PrintQueueItem.status == "pending")
  1272. # Never re-select a row a dispatch worker has already claimed
  1273. # (#2615) — belt-and-suspenders with the _inflight exclusion
  1274. # below, and the guard that lets an orphaned claim be ignored
  1275. # until startup reconciliation clears it.
  1276. .where(PrintQueueItem.dispatching_at.is_(None))
  1277. # archive/library_file are read by the cross-model gate
  1278. # (#2578); eager-load once per pass instead of a lazy-load
  1279. # (which would raise in async) per item.
  1280. .options(
  1281. selectinload(PrintQueueItem.archive),
  1282. selectinload(PrintQueueItem.library_file),
  1283. # Cross-model candidates (#671), plus each candidate's file
  1284. # for the same cross-model gate. Lazy-loading either would
  1285. # raise in async.
  1286. selectinload(PrintQueueItem.variants).selectinload(PrintQueueVariant.library_file),
  1287. )
  1288. .order_by(
  1289. PrintQueueItem.been_jumped.desc(),
  1290. PrintQueueItem.print_time_seconds.asc().nullslast(),
  1291. PrintQueueItem.position,
  1292. PrintQueueItem.id,
  1293. )
  1294. )
  1295. else:
  1296. result = await db.execute(
  1297. select(PrintQueueItem)
  1298. .where(PrintQueueItem.status == "pending")
  1299. # Skip rows already claimed by a dispatch worker (#2615).
  1300. .where(PrintQueueItem.dispatching_at.is_(None))
  1301. .options(
  1302. selectinload(PrintQueueItem.archive),
  1303. selectinload(PrintQueueItem.library_file),
  1304. # Cross-model candidates (#671), plus each candidate's file
  1305. # for the same cross-model gate. Lazy-loading either would
  1306. # raise in async.
  1307. selectinload(PrintQueueItem.variants).selectinload(PrintQueueVariant.library_file),
  1308. )
  1309. .order_by(PrintQueueItem.position, PrintQueueItem.id)
  1310. )
  1311. items = list(result.scalars().all())
  1312. # Drop rows whose upload is still in flight from an earlier pass
  1313. # (#2602). They stay `pending` until the upload finishes, so without
  1314. # this a fast tick would re-select and re-dispatch the same row.
  1315. # Belt-and-suspenders with the printer exclusion below.
  1316. if self._inflight:
  1317. inflight_ids = set(self._inflight)
  1318. items = [it for it in items if it.id not in inflight_ids]
  1319. # Read plate-clear setting once per queue check. Default MUST be
  1320. # False to match the schema (SettingsSchema.require_plate_clear
  1321. # defaults False) and the frontend (toggle + card badge both treat a
  1322. # missing value as off). When no settings row exists, a True default
  1323. # here re-enabled the plate-clear gate the UI showed as disabled,
  1324. # blocking dispatch to FINISH-state printers forever with no UI path
  1325. # to clear it (#1865).
  1326. require_plate_clear = await self._get_bool_setting(db, "require_plate_clear", default=False)
  1327. # Dispatch and track scheduled drying runs (#2638)
  1328. await self._check_scheduled_dryings(db)
  1329. if not items:
  1330. # No dispatchable pending items — still check auto-drying on idle
  1331. # printers, but keep any printer with an upload still in flight
  1332. # from an earlier pass out of it (#2602): its print is imminent,
  1333. # so it must not be auto-dried in the gap before the row flips to
  1334. # printing. Report the pass as productive while uploads run so the
  1335. # loop stays on the fast interval.
  1336. #
  1337. # Also release any keep-warm holds that got orphaned by the queue
  1338. # emptying — the normal sweep in `_apply_keep_warm` is skipped by
  1339. # this early return, so call it directly with an empty candidate
  1340. # set. Without this, a printer whose queued item was cancelled or
  1341. # deleted would keep its bed at target until the max-duration
  1342. # timeout expired.
  1343. self._sweep_keep_warm(active_candidates=set(), dispatched=set())
  1344. inflight_printers = {pid for (_task, pid) in self._inflight.values() if pid is not None}
  1345. await self._check_auto_drying(db, [], inflight_printers)
  1346. await self._check_filament_low(db)
  1347. await self._check_stock_forecast(db)
  1348. return bool(self._inflight)
  1349. logger.info(
  1350. "Queue check: found %d pending items: %s",
  1351. len(items),
  1352. [(i.id, i.printer_id, i.archive_id, i.library_file_id) for i in items],
  1353. )
  1354. # Seed busy_printers with printers that already have an item in 'printing'
  1355. # status. _is_printer_idle() alone is not sufficient as a dispatch gate —
  1356. # on H2D / P1 series the MQTT state transition from IDLE to RUNNING can
  1357. # lag several seconds behind the print command, so the next check_queue
  1358. # tick still sees IDLE and would double-dispatch onto the same printer.
  1359. # Without this guard, two pending items targeting the same printer
  1360. # (e.g. a batch with quantity>1) both end up in 'printing' status —
  1361. # surfaced via the "BUG: Multiple queue items" warning in on_print_complete.
  1362. busy_result = await db.execute(
  1363. select(PrintQueueItem.printer_id)
  1364. .where(PrintQueueItem.status == "printing")
  1365. .where(PrintQueueItem.printer_id.is_not(None))
  1366. )
  1367. busy_printers: set[int] = {pid for (pid,) in busy_result.all() if pid is not None}
  1368. # Why each printer left this pass, recorded where the decision is
  1369. # made rather than re-derived when the summary is logged. #3018's
  1370. # bundle shows what the old summary produced: "printer 1 not
  1371. # available -- connected=True, state=IDLE" immediately followed by a
  1372. # dispatch to printer 1. Two things went wrong at once. The line read
  1373. # live state at log time, which by then no longer matched the state
  1374. # the decision was made on; and `busy_printers` holds both printers
  1375. # that cannot take work and printers this pass has claimed for it,
  1376. # which are opposite facts. It is the first line anyone greps for
  1377. # "why did my item not go out", so it has to say which.
  1378. busy_reasons: dict[int, str] = dict.fromkeys(busy_printers, "an item is already printing on it")
  1379. # Printers this pass is dispatching to. They are in busy_printers so
  1380. # nothing else in the pass targets them -- that is a reservation, not
  1381. # an obstruction, and the summary says so.
  1382. claimed_printers: set[int] = set()
  1383. def mark_busy(printer_id: int, reason: str) -> None:
  1384. """Take ``printer_id`` out of this pass, recording why.
  1385. First reason wins: a printer already excluded by a stronger fact
  1386. -- a print running on it -- must not be relabelled by a weaker
  1387. check that ran later and would have excluded it anyway.
  1388. """
  1389. busy_printers.add(printer_id)
  1390. busy_reasons.setdefault(printer_id, reason)
  1391. def claim_printer(printer_id: int) -> None:
  1392. """Reserve ``printer_id`` for an item this pass is dispatching."""
  1393. claimed_printers.add(printer_id)
  1394. mark_busy(printer_id, "selected for dispatch in this pass")
  1395. # The user-facing half of `busy_reasons` (#3074). The same decisions
  1396. # worded for a different audience: the log wants "still inside its
  1397. # post-dispatch hold window", the queue row wants to know the printer
  1398. # is taken and nothing is broken. Only the cases that would read wrong
  1399. # as a plain "Busy" are recorded here; the rest fall back to it.
  1400. item_hold_reasons: dict[int, str] = {}
  1401. # Names and models for the printers this pass may have to write a
  1402. # waiting reason about, read once rather than per skip per tick. The
  1403. # model-based branch already has its names from `_printers_for_model`.
  1404. pinned_printer_ids = {i.printer_id for i in items if i.printer_id}
  1405. pinned_printers: dict[int, tuple[str, str]] = {}
  1406. if pinned_printer_ids:
  1407. pinned_rows = await db.execute(
  1408. select(Printer.id, Printer.name, Printer.model).where(Printer.id.in_(pinned_printer_ids))
  1409. )
  1410. pinned_printers = {pid: (name or "", model or "") for pid, name, model in pinned_rows.all()}
  1411. def printer_label(printer_id: int) -> str:
  1412. """What to call this printer in a queue row."""
  1413. entry = pinned_printers.get(printer_id)
  1414. return (entry[0] if entry else "") or f"printer {printer_id}"
  1415. async def hold_item(item: PrintQueueItem, reason: str | None, *, notify: bool = True) -> None:
  1416. """Record why *item* is not going out, in the words the queue row shows.
  1417. The fixed-printer branch's single writer for ``waiting_reason``
  1418. (#3074). Before this, the sensor interlock was the only thing that
  1419. wrote the field there, so an item pinned to a printer that was
  1420. merely printing sat at `pending` with nothing to show for it —
  1421. indistinguishable from a queue that had stopped working — while
  1422. the same job queued as "Any <model>" explained itself.
  1423. Every exit from that branch now calls this, which is also what
  1424. replaced the interlock's old habit of clearing the field up front:
  1425. a lifted hold cannot leave "Waiting on Enclosure Door" standing,
  1426. because whichever exit runs next overwrites it and the dispatch
  1427. path clears it.
  1428. A notification goes out when this item starts asking for
  1429. something, and only then: the new reason needs the user, and what
  1430. it replaced did not. "What it replaced did not" has to include a
  1431. busy-only reason, not just an empty one. The sequence this branch
  1432. actually produces is a print running (``Busy: X1C-01``) and then
  1433. the plate it left behind (``Waiting for plate confirmation``), and
  1434. testing "was the field empty" would call that no transition at all
  1435. and stay quiet through the one case worth saying out loud.
  1436. The cost is that a printer dropping offline, coming back busy and
  1437. dropping again asks twice rather than once. That is the honest
  1438. reading — it went wrong twice — and the alternative was a rule
  1439. that never fired for the case this was built for.
  1440. *notify* is how a caller opts out. The sensor interlock does: it
  1441. has never sent this notification, and a change about what the
  1442. queue *displays* is not the place to start (#1148).
  1443. """
  1444. if item.waiting_reason == reason:
  1445. return
  1446. # Busy-only and empty are the same thing here: neither is the
  1447. # queue asking the user for anything.
  1448. was_asking = bool(item.waiting_reason) and not self._is_busy_only(item.waiting_reason)
  1449. item.waiting_reason = reason
  1450. await db.commit()
  1451. if not notify or not reason or was_asking or self._is_busy_only(reason):
  1452. return
  1453. try:
  1454. job_name = await self._get_job_name(db, item)
  1455. entry = pinned_printers.get(item.printer_id) if item.printer_id else None
  1456. await notification_service.on_queue_job_waiting(
  1457. job_name=job_name,
  1458. target_model=(entry[1] if entry else "") or "",
  1459. waiting_reason=reason,
  1460. db=db,
  1461. )
  1462. except Exception as e:
  1463. # A queue that cannot say why it is waiting is the bug being
  1464. # fixed here; a queue that stops dispatching because a
  1465. # notification provider is down would be a worse one.
  1466. logger.debug("Waiting notification failed for item %s: %s", item.id, e)
  1467. async def hold_for_printer(
  1468. item: PrintQueueItem, printer_id: int, log_reason: str, item_reason: str
  1469. ) -> None:
  1470. """Take *printer_id* out of this pass and tell the item's owner why."""
  1471. mark_busy(printer_id, log_reason)
  1472. item_hold_reasons.setdefault(printer_id, item_reason)
  1473. await hold_item(item, item_reason)
  1474. # Defense-in-depth (#1157): augment busy_printers with any printer
  1475. # still in its post-dispatch hold window. Empirically, the DB seed
  1476. # above can miss in-flight items in a multi-plate batch — same-file
  1477. # plates were being dispatched 30 s apart while the H2D was still
  1478. # digesting the first project_file. The hold is keyed in-memory and
  1479. # released by the watchdog on the success path, so it adds a layer
  1480. # that doesn't depend on DB row visibility or completion-callback
  1481. # timing.
  1482. for held_printer_id in list(self._dispatch_holds.keys()):
  1483. if self._printer_in_dispatch_hold(held_printer_id):
  1484. mark_busy(held_printer_id, "still inside its post-dispatch hold window")
  1485. # Exclude printers whose upload is still in flight from an earlier
  1486. # pass (#2602). The row is `pending` until the upload finishes and
  1487. # the printing-state seed / dispatch hold above only arm once the
  1488. # upload completes, so this is what holds the printer (and, via
  1489. # busy_printers, its auto-drying) out of the pass during the upload.
  1490. for _task, inflight_pid in self._inflight.values():
  1491. if inflight_pid is not None:
  1492. mark_busy(inflight_pid, "an upload to it is still in flight")
  1493. # Snapshot taken here, before the item loop adds anything (#2801).
  1494. #
  1495. # The three sources above all mean the same thing: a print on this
  1496. # printer is running or imminent. Everything the loop adds below
  1497. # means only "the queue could not dispatch to it this pass", which
  1498. # is a different statement -- a printer waiting on a plate-clear
  1499. # acknowledgment, an offline printer, one with no matching file.
  1500. #
  1501. # Auto-drying must only see the first kind. Reading the whole set
  1502. # as "is currently printing" is what put a plate-held printer down
  1503. # the mid-print path: it capped the drying temperature, logged the
  1504. # cycle as (mid-print), and skipped the very gate that was supposed
  1505. # to hold it. The interlock block below already documents the same
  1506. # hazard and works around it by staying out of busy_printers; this
  1507. # generalises that workaround instead of repeating it per case.
  1508. dispatching_printers: set[int] = set(busy_printers)
  1509. # Printers whose last upload never reached them (#3210). After the
  1510. # snapshot above on purpose: such a printer is idle, not about to
  1511. # print, and auto-drying must keep treating it as idle.
  1512. now_mono = time.monotonic()
  1513. # In backoff this pass. Kept apart from busy_printers so keep-warm
  1514. # can leave them out: there is no print to keep a bed warm for.
  1515. backoff_printers: set[int] = set()
  1516. # In backoff, and the "job waiting" notification for this run of
  1517. # refusals has gone out already. Their holds are written silently.
  1518. silent_hold_printers: set[int] = set()
  1519. for backoff_pid, retry_at in list(self._upload_backoff.items()):
  1520. if now_mono >= retry_at:
  1521. del self._upload_backoff[backoff_pid]
  1522. continue
  1523. backoff_printers.add(backoff_pid)
  1524. if backoff_pid in self._upload_refusal_notified:
  1525. silent_hold_printers.add(backoff_pid)
  1526. else:
  1527. self._upload_refusal_notified.add(backoff_pid)
  1528. mark_busy(
  1529. backoff_pid,
  1530. f"its file service refused the last upload; retrying in {retry_at - now_mono:.0f}s",
  1531. )
  1532. item_hold_reasons.setdefault(
  1533. backoff_pid,
  1534. f"{printer_label(backoff_pid)} is not accepting files — Bambuddy will retry automatically",
  1535. )
  1536. # Printers held by a Home Assistant sensor interlock (#1148) — an
  1537. # enclosure door left open, say. The fixed-printer branch turns
  1538. # this into a waiting_reason the user can act on; the model-based
  1539. # branch hides these printers from the matcher so an "Any <model>"
  1540. # job runs on a sibling instead of queueing behind the held one.
  1541. #
  1542. # Deliberately NOT merged into busy_printers, even though that set
  1543. # already means "unavailable this pass". _check_auto_drying reads
  1544. # it as "is currently printing" and would put an idle-but-held
  1545. # printer down the mid-print drying path, which caps the drying
  1546. # temperature and skips the queue-only gating. A held printer is
  1547. # idle; it should dry exactly as it did before.
  1548. #
  1549. # Only sensors we actually read and found alerting appear here; see
  1550. # ha_sensor_manager.blocked_printers. A Home Assistant that is down
  1551. # holds nothing.
  1552. interlocked: dict[int, str] = {}
  1553. try:
  1554. interlocked = await ha_sensor_manager.blocked_printers(db)
  1555. except Exception as e:
  1556. # Never let the interlock stop the queue running. A broken
  1557. # lookup means no holds, not no dispatches.
  1558. logger.warning("Home Assistant interlock check failed: %s", e)
  1559. interlocked = {}
  1560. # Printers a smart plug can bring back, read once for the whole pass
  1561. # (#2786). Used by the model-based branch both to word "Offline" in
  1562. # the waiting reason and to decide what the wake step may switch on.
  1563. wakeable_printer_ids = await self._wakeable_printer_ids(db)
  1564. # At most one power-on per queue check. Each one blocks this loop
  1565. # for the boot wait, so a queue of ten class-targeted jobs must not
  1566. # switch on ten printers inside a single pass — the next pass wakes
  1567. # the next one (#2786).
  1568. power_on_attempted = False
  1569. # Log skip reasons once per queue check (not per item)
  1570. skip_reasons: dict[str, int] = {}
  1571. # Items selected for dispatch in this pass, one per printer. The
  1572. # loop below only *decides* — the uploads happen afterwards, in
  1573. # parallel (#2555). See _dispatch_selected().
  1574. dispatch_ids: list[int] = []
  1575. # Library rows queued with `cleanup_library_after_dispatch` (the
  1576. # printer-card "upload and print" flow) are CONSUMED by the dispatch
  1577. # that prints them: the row is deleted and the 3MF is unlinked from
  1578. # disk. That was safe only because dispatch was serial. Run two of
  1579. # them against the same row at once and the second DELETE matches no
  1580. # row (StaleDataError), and the winner's unlink can pull the file out
  1581. # from under the loser's in-flight upload.
  1582. #
  1583. # Only the cleanup flag mutates the row. An ordinary library print
  1584. # just reads it, so the common fan-out — one file, many printers,
  1585. # which is exactly the reporter's workload — still goes out fully in
  1586. # parallel. Narrow the guard to the mutating case; do not serialise
  1587. # the case the whole fix exists for.
  1588. dispatch_libs: set[int] = set()
  1589. consumed_libs: set[int] = set()
  1590. def _library_row_conflict(candidate: PrintQueueItem) -> bool:
  1591. """True if dispatching `candidate` now would race another item's cleanup."""
  1592. lib_id = candidate.library_file_id
  1593. if lib_id is None:
  1594. return False
  1595. if candidate.cleanup_library_after_dispatch:
  1596. # We would delete a row someone else in this pass is reading.
  1597. return lib_id in dispatch_libs
  1598. # Someone else in this pass will delete the row out from under us.
  1599. return lib_id in consumed_libs
  1600. def _claim_library_row(candidate: PrintQueueItem) -> None:
  1601. lib_id = candidate.library_file_id
  1602. if lib_id is None:
  1603. return
  1604. dispatch_libs.add(lib_id)
  1605. if candidate.cleanup_library_after_dispatch:
  1606. consumed_libs.add(lib_id)
  1607. # Printer scope of each job's creator (#1727), so an "any <model>"
  1608. # job only lands on a printer its creator may use. Loaded lazily,
  1609. # once per creator per pass.
  1610. from backend.app.core.auth import is_auth_enabled
  1611. scope_auth_enabled = await is_auth_enabled(db)
  1612. creator_scopes: dict[int, PrinterScope] = {}
  1613. async def _creator_scope(user_id: int | None) -> PrinterScope:
  1614. if not scope_auth_enabled or user_id is None:
  1615. return ALL_PRINTERS
  1616. if user_id not in creator_scopes:
  1617. creator_scopes[user_id] = await resolve_user_id_printer_scope(db, user_id)
  1618. return creator_scopes[user_id]
  1619. for item in items:
  1620. # Check scheduled time first (scheduled_time is stored in UTC from ISO string)
  1621. if item.scheduled_time:
  1622. sched = item.scheduled_time
  1623. if sched.tzinfo is None:
  1624. sched = sched.replace(tzinfo=timezone.utc)
  1625. if sched > datetime.now(timezone.utc):
  1626. # Waiting on the clock, not on a printer.
  1627. await hold_item(item, None)
  1628. skip_reasons["scheduled_future"] = skip_reasons.get("scheduled_future", 0) + 1
  1629. continue
  1630. # Skip items that require manual start
  1631. if item.manual_start:
  1632. # Waiting on the user, not on a printer. Cleared here because
  1633. # this is the last pass that will look at the row: a staged
  1634. # item never reaches the branches below again, so a reason
  1635. # left from before it was staged would stand forever (#3074).
  1636. # The one exception is the unmatched-filament hold, whose
  1637. # reason was written at staging and is why the item waits:
  1638. # it names the filament to load (#2799). Pressing start
  1639. # clears manual_start, and the branches below overwrite it.
  1640. keep = item.waiting_reason if _is_unmatched_hold_reason(item.waiting_reason) else None
  1641. await hold_item(item, keep)
  1642. skip_reasons["manual_start"] = skip_reasons.get("manual_start", 0) + 1
  1643. continue
  1644. if item.printer_id:
  1645. # Its creator may no longer use this printer (#1727): an
  1646. # admin took it away from their group after the job was
  1647. # queued, say to keep it free for a training session. Held,
  1648. # not failed, so it starts if access comes back or the user
  1649. # moves it to a printer they still have.
  1650. if not (await _creator_scope(item.created_by_id)).allows(item.printer_id):
  1651. await hold_item(
  1652. item,
  1653. "Its owner no longer has access to this printer — move it to another printer",
  1654. notify=False,
  1655. )
  1656. skip_reasons["printer_out_of_scope"] = skip_reasons.get("printer_out_of_scope", 0) + 1
  1657. continue
  1658. # Held by a sensor interlock (#1148). Checked before the
  1659. # busy_printers test that would otherwise swallow it
  1660. # silently — "waiting for a printer" and "waiting for you
  1661. # to shut the enclosure" need to read differently, and only
  1662. # one of them is something the user can fix.
  1663. #
  1664. # It used to be the only thing that wrote a waiting_reason on
  1665. # this branch, and it cleared the field up front so a lifted
  1666. # hold could not leave a shut door reading "Waiting on
  1667. # Enclosure Door". `hold_item` carries that guarantee now —
  1668. # every exit below writes — so the clear is gone and the
  1669. # interlock is an ordinary hold like the rest (#3074).
  1670. interlock_reason = interlocked.get(item.printer_id)
  1671. if interlock_reason:
  1672. # Silent, exactly as it has always been. #1148 built this
  1673. # as a hold that shows on the row, never as an alert, and
  1674. # routing it through the shared writer must not quietly
  1675. # turn every open door into a notification.
  1676. await hold_item(item, f"Waiting on {interlock_reason}", notify=False)
  1677. skip_reasons["sensor_interlock"] = skip_reasons.get("sensor_interlock", 0) + 1
  1678. continue
  1679. # Specific printer assignment (existing behavior)
  1680. if item.printer_id in busy_printers:
  1681. # Whatever took the printer out of this pass — a print
  1682. # already running on it, a post-dispatch hold, an upload
  1683. # still in flight, an item ahead of this one in the same
  1684. # pass — reads the same way from the queue: the printer is
  1685. # taken and this item is in line for it. The exceptions
  1686. # that do not (an offline printer, say) recorded their own
  1687. # wording in `item_hold_reasons` when they held it.
  1688. await hold_item(
  1689. item,
  1690. item_hold_reasons.get(item.printer_id) or f"Busy: {printer_label(item.printer_id)}",
  1691. notify=item.printer_id not in silent_hold_printers,
  1692. )
  1693. continue
  1694. # Check if printer is idle
  1695. printer_idle = self._is_printer_idle(item.printer_id, require_plate_clear)
  1696. printer_connected = printer_manager.is_connected(item.printer_id)
  1697. # If printer not connected, try to power on via smart plug
  1698. if not printer_connected:
  1699. plugs = await self._get_smart_plugs(db, item.printer_id)
  1700. auto_on_plugs = [p for p in plugs if p.auto_on and p.enabled]
  1701. if auto_on_plugs:
  1702. logger.info("Printer %s offline, attempting to power on via smart plug(s)", item.printer_id)
  1703. # Power on using the plug that actually feeds the printer, and
  1704. # wait for it to boot on that one only (#2629).
  1705. primary_plug = self._pick_power_plug(auto_on_plugs)
  1706. powered_on = await self._power_on_and_wait(primary_plug, item.printer_id, db)
  1707. if powered_on:
  1708. # Also turn on any remaining auto_on plugs (e.g., filter)
  1709. for extra_plug in [p for p in auto_on_plugs if p.id != primary_plug.id]:
  1710. try:
  1711. service = await smart_plug_manager.get_service_for_plug(extra_plug, db)
  1712. await service.turn_on(extra_plug)
  1713. logger.info(
  1714. "Also powered on plug '%s' for printer %s", extra_plug.name, item.printer_id
  1715. )
  1716. except Exception as e:
  1717. logger.warning("Failed to power on extra plug '%s': %s", extra_plug.name, e)
  1718. printer_connected = True
  1719. printer_idle = self._is_printer_idle(item.printer_id, require_plate_clear)
  1720. else:
  1721. logger.warning("Could not power on printer %s via smart plug", item.printer_id)
  1722. await hold_for_printer(
  1723. item,
  1724. item.printer_id,
  1725. "smart-plug power-on failed",
  1726. f"Offline: {printer_label(item.printer_id)} — the smart plug could not power it on",
  1727. )
  1728. continue
  1729. else:
  1730. # No plug or auto_on disabled. Worded exactly as the
  1731. # model-based branch words it (#2786): this is the one
  1732. # entry on that list the user has to act on, because
  1733. # Bambuddy will never switch this printer on itself.
  1734. await hold_for_printer(
  1735. item,
  1736. item.printer_id,
  1737. "offline, with no smart plug to power it on",
  1738. f"Offline, no Auto On smart plug: {printer_label(item.printer_id)}",
  1739. )
  1740. continue
  1741. # Check if printer is idle (busy with another print)
  1742. if not printer_idle:
  1743. await hold_for_printer(
  1744. item,
  1745. item.printer_id,
  1746. "not idle",
  1747. self._pinned_hold_reason(
  1748. item.printer_id, printer_label(item.printer_id), require_plate_clear
  1749. ),
  1750. )
  1751. continue
  1752. # Drying blocks the queue, if the user asked it to. A hold
  1753. # is a skip like any other, so it belongs here with the
  1754. # rest of the availability checks.
  1755. # A parked timer (#2896) never ends, so it must not hold the
  1756. # queue; it stays tracked so the stop paths still reach it.
  1757. if (
  1758. self._drying_in_progress.get(item.printer_id)
  1759. and not self._drying_is_only_parked(item.printer_id)
  1760. and await self._get_bool_setting(db, "queue_drying_block")
  1761. ):
  1762. # Busy-shaped on purpose: the cycle ends on its own and
  1763. # the job goes out, so there is nothing to alert about.
  1764. await hold_for_printer(
  1765. item,
  1766. item.printer_id,
  1767. "drying, and drying is set to block the queue",
  1768. f"Busy: {printer_label(item.printer_id)} (drying)",
  1769. )
  1770. continue
  1771. # Check condition (previous print success)
  1772. if item.require_previous_success:
  1773. if not await self._check_previous_success(db, item):
  1774. item.status = "skipped"
  1775. item.error_message = "Previous print failed or was aborted"
  1776. item.completed_at = datetime.now(timezone.utc)
  1777. # Not pending any more, so not waiting for anything.
  1778. item.waiting_reason = None
  1779. await db.commit()
  1780. logger.info("Skipped queue item %s - previous print failed", item.id)
  1781. # Send notification
  1782. job_name = await self._get_job_name(db, item)
  1783. printer = await self._get_printer(db, item.printer_id)
  1784. await notification_service.on_queue_job_skipped(
  1785. job_name=job_name,
  1786. printer_id=item.printer_id,
  1787. printer_name=printer.name if printer else "Unknown",
  1788. reason="Previous print failed or was aborted",
  1789. db=db,
  1790. )
  1791. continue
  1792. # Resolve the AMS mapping when it's missing OR unresolved
  1793. # (all -1). A stored all-[-1] mapping is a bug artifact — a
  1794. # frontend status-load race can persist [-1] (#2589) — and
  1795. # must be recomputed from live trays rather than trusted.
  1796. unmappable = await self._ensure_ams_mapping(db, item.printer_id, item)
  1797. if unmappable:
  1798. await self._fail_unmappable_item(db, item, item.printer_id, unmappable)
  1799. continue
  1800. # Filament-deficit pre-dispatch check (#1496). If the
  1801. # assigned spool can't satisfy any required slot grams,
  1802. # promote the item to manual_start so the user must
  1803. # acknowledge via the ▶ button (which re-checks live).
  1804. if await self._block_on_filament_deficit(db, item):
  1805. # Now staged for the user to start by hand, and the row
  1806. # shows the filament-short badge instead. Cleared because
  1807. # a staged item never reaches this branch again.
  1808. await hold_item(item, None)
  1809. continue
  1810. # Unmatched-filament pre-dispatch check (#2799). Hold rather
  1811. # than let the printer pick a substitute for a slot the
  1812. # matcher could not resolve on this printer.
  1813. if await self._block_on_unmatched_filament(db, item):
  1814. continue
  1815. # Hold this item back for the next pass rather than racing
  1816. # another dispatch over the same transient library row. The
  1817. # printer is still marked busy so a later item does not jump
  1818. # its place in this printer's queue.
  1819. if _library_row_conflict(item):
  1820. skip_reasons["library_row_in_use"] = skip_reasons.get("library_row_in_use", 0) + 1
  1821. await hold_for_printer(
  1822. item,
  1823. item.printer_id,
  1824. "holding its place while another item releases a library row",
  1825. f"Busy: {printer_label(item.printer_id)}",
  1826. )
  1827. continue
  1828. # Print takes priority: stop a cycle Bambuddy armed, now
  1829. # that this item is definitely going out.
  1830. #
  1831. # Placement is the whole point (#2801). This used to sit up
  1832. # with the availability checks, inside the not-idle branch
  1833. # -- so it fired only on the passes where the print was NOT
  1834. # going to start, and never on the ones where it was.
  1835. # Drying is not one of the things `_is_printer_idle` looks
  1836. # at, so a stop could never have unblocked that printer
  1837. # anyway; the cycle was spent for nothing, auto-drying
  1838. # re-armed on the next tick, and a plate left
  1839. # unacknowledged turned that into a loop on the scheduler
  1840. # interval. Every skip between there and here -- a failed
  1841. # previous print, an unmappable item, a filament deficit, a
  1842. # contested library row -- is another way to lose a cycle
  1843. # for a print that never happens, which is why this waits
  1844. # until the decision is actually made.
  1845. if self._drying_in_progress.get(
  1846. item.printer_id
  1847. ) and not await self._drying_may_continue_through_print(db, item.printer_id):
  1848. await self._stop_drying(item.printer_id)
  1849. # Queue the dispatch instead of running it here — see
  1850. # _dispatch_selected(). busy_printers still gets the printer
  1851. # immediately, so nothing else in this pass can target it.
  1852. #
  1853. # The reason goes first: this item is not waiting for anything
  1854. # any more, and the model-based branch clears its own at the
  1855. # equivalent moment (#3074).
  1856. await hold_item(item, None)
  1857. _claim_library_row(item)
  1858. dispatch_ids.append(item.id)
  1859. claim_printer(item.printer_id)
  1860. # SJF starvation guard: mark items that were jumped
  1861. if sjf_enabled and item.print_time_seconds is not None:
  1862. pinned_model = pinned_printers.get(item.printer_id, ("", ""))[1]
  1863. for other in items:
  1864. if (
  1865. other.id != item.id
  1866. and other.id not in dispatch_ids
  1867. and other.status == "pending"
  1868. and _could_take_printer(other, item.printer_id, pinned_model)
  1869. and not other.been_jumped
  1870. and other.position < item.position
  1871. and (
  1872. other.print_time_seconds is None
  1873. or other.print_time_seconds > item.print_time_seconds
  1874. )
  1875. ):
  1876. other.been_jumped = True
  1877. await db.commit()
  1878. elif item.target_model or item.variants:
  1879. # Model-based assignment - find any idle printer of matching model.
  1880. # A plain model-based item has exactly one candidate, built from
  1881. # its own columns. A cross-model item (#671) has one per sliced
  1882. # variant and takes the first that matches, walking them in the
  1883. # user's priority order so the pick is reproducible when more
  1884. # than one printer is free in the same pass.
  1885. candidates = _candidates_for(item)
  1886. item_scope = await _creator_scope(item.created_by_id)
  1887. printer_id = None
  1888. chosen: _ModelCandidate | None = None
  1889. per_model_reasons: list[tuple[str | None, str]] = []
  1890. # Candidates that cleared the cross-model gate below. The
  1891. # smart-plug wake step may only consider these — waking a
  1892. # printer for a file that can never legally run on it is
  1893. # worse than not waking at all (#2786).
  1894. wakeable_candidates: list[_ModelCandidate] = []
  1895. if not candidates:
  1896. # Every candidate file has been deleted or trashed out from
  1897. # under this item. Hold it with something the user can act
  1898. # on rather than letting it look dispatchable forever.
  1899. per_model_reasons.append(
  1900. (
  1901. item.target_model,
  1902. "Every file for this job has been deleted — add a file back or remove the item",
  1903. )
  1904. )
  1905. for candidate in candidates:
  1906. effective_types, filament_overrides = _filament_constraints(candidate)
  1907. # Cross-model safety gate (#2578): never hand a 3MF sliced
  1908. # for an incompatible model to a printer, no matter how the
  1909. # row got into the DB (old rows, direct API writes). Held
  1910. # as pending with an actionable waiting_reason — the user
  1911. # fixes it by editing the item's target model.
  1912. if not is_gcode_compatible(candidate.sliced_for, candidate.target_model):
  1913. per_model_reasons.append(
  1914. (
  1915. candidate.target_model,
  1916. f"File was sliced for {candidate.sliced_for}, which is not compatible with "
  1917. f"{candidate.target_model} — edit the item and fix its target model",
  1918. )
  1919. )
  1920. skip_reasons["sliced_model_mismatch"] = skip_reasons.get("sliced_model_mismatch", 0) + 1
  1921. continue
  1922. wakeable_candidates.append(candidate)
  1923. match_id, match_reason = await self._find_idle_printer_for_model(
  1924. db,
  1925. candidate.target_model,
  1926. # Sensor-held printers are unavailable to the
  1927. # matcher but stay out of busy_printers itself
  1928. # (#1148) — see where `interlocked` is built.
  1929. busy_printers | interlocked.keys(),
  1930. effective_types,
  1931. item.target_location,
  1932. filament_overrides=filament_overrides,
  1933. require_plate_clear=require_plate_clear,
  1934. wakeable_ids=wakeable_printer_ids,
  1935. printer_scope=item_scope,
  1936. )
  1937. if match_id:
  1938. printer_id = match_id
  1939. chosen = candidate
  1940. break
  1941. per_model_reasons.append((candidate.target_model, match_reason or ""))
  1942. # Nothing is available and nothing has been woken this pass:
  1943. # switch one matching printer on. Assignment is left to the
  1944. # next pass, which sees the booted printer's live state
  1945. # instead of guessing at it seconds after connect (#2786).
  1946. if printer_id is None and not power_on_attempted and wakeable_candidates:
  1947. woken_id, attempted_id = await self._wake_printer_for_model(
  1948. db,
  1949. wakeable_candidates,
  1950. item.target_location,
  1951. busy_printers | interlocked.keys(),
  1952. wakeable_printer_ids,
  1953. require_plate_clear,
  1954. printer_scope=item_scope,
  1955. )
  1956. # An attempt spends the pass's one wake whether or not
  1957. # it worked: it has already blocked the queue loop for
  1958. # the boot wait. A failed printer is held out of later
  1959. # passes by its own cool-off, deliberately NOT by
  1960. # busy_printers — it is off, not busy, and labelling it
  1961. # busy would both misdescribe it in every later item's
  1962. # waiting reason and suppress the notification, since
  1963. # an all-busy reason is treated as needing no action.
  1964. power_on_attempted = attempted_id is not None
  1965. if woken_id is not None:
  1966. # Hold this item back rather than dispatching onto a
  1967. # printer whose AMS has not reported yet.
  1968. skip_reasons["powered_on_printer"] = skip_reasons.get("powered_on_printer", 0) + 1
  1969. continue
  1970. waiting_reason = None if printer_id else _collapse_waiting_reasons(per_model_reasons)
  1971. # Fold the winning variant's file and settings onto the item
  1972. # before anything else looks at them — the guards below and
  1973. # every step of the dispatch read the item's own columns.
  1974. if chosen is not None:
  1975. self._resolve_variant(item, chosen)
  1976. # Update waiting_reason if changed and send notification when first waiting
  1977. if item.waiting_reason != waiting_reason:
  1978. was_waiting = item.waiting_reason is not None
  1979. item.waiting_reason = waiting_reason
  1980. await db.commit()
  1981. # Send waiting notification only when transitioning to waiting state
  1982. # and the reason requires user action (not just "all printers busy")
  1983. if waiting_reason and not was_waiting and not self._is_busy_only(waiting_reason):
  1984. job_name = await self._get_job_name(db, item)
  1985. await notification_service.on_queue_job_waiting(
  1986. job_name=job_name,
  1987. target_model=_candidate_model_label(candidates) or item.target_model,
  1988. waiting_reason=waiting_reason,
  1989. db=db,
  1990. )
  1991. if printer_id:
  1992. # Before claiming the printer: hold back rather than race
  1993. # another dispatch over the same transient library row.
  1994. # Checked here so a held item does not get a printer
  1995. # assigned and then sit on it. See _library_row_conflict().
  1996. #
  1997. # No busy_printers.add() here, unlike the fixed-printer
  1998. # branch above: that one protects its printer's own queue
  1999. # ordering, but this item was never assigned to `printer_id`
  2000. # — the matcher merely offered it. Marking it busy would
  2001. # strand an idle printer for the rest of the pass.
  2002. if _library_row_conflict(item):
  2003. skip_reasons["library_row_in_use"] = skip_reasons.get("library_row_in_use", 0) + 1
  2004. continue
  2005. # Check condition (previous print success) before assigning
  2006. if item.require_previous_success:
  2007. if not await self._check_previous_success(db, item):
  2008. item.status = "skipped"
  2009. item.error_message = "Previous print failed or was aborted"
  2010. item.completed_at = datetime.now(timezone.utc)
  2011. await db.commit()
  2012. logger.info("Skipped queue item %s - previous print failed", item.id)
  2013. # Send notification
  2014. job_name = await self._get_job_name(db, item)
  2015. printer = await self._get_printer(db, printer_id)
  2016. await notification_service.on_queue_job_skipped(
  2017. job_name=job_name,
  2018. printer_id=printer_id,
  2019. printer_name=printer.name if printer else "Unknown",
  2020. reason="Previous print failed or was aborted",
  2021. db=db,
  2022. )
  2023. continue
  2024. # Assign printer and start - clear waiting reason
  2025. item.printer_id = printer_id
  2026. item.waiting_reason = None
  2027. logger.info("Model-based assignment: queue item %s assigned to printer %s", item.id, printer_id)
  2028. # Send assignment notification
  2029. job_name = await self._get_job_name(db, item)
  2030. printer = await self._get_printer(db, printer_id)
  2031. await notification_service.on_queue_job_assigned(
  2032. job_name=job_name,
  2033. printer_id=printer_id,
  2034. printer_name=printer.name if printer else "Unknown",
  2035. target_model=item.target_model,
  2036. db=db,
  2037. )
  2038. # A mapping on this item was not made for the printer just
  2039. # picked: it came with a job moved here from a fixed
  2040. # printer, or from a variant. Its tray IDs can name a
  2041. # tray of the right type in another colour here, which
  2042. # the fit check lets through (#3239). Match afresh, by
  2043. # type and colour.
  2044. if item.ams_mapping:
  2045. logger.info(
  2046. "Queue item %s: dropping stored ams_mapping %s, not made for printer %s",
  2047. item.id,
  2048. item.ams_mapping,
  2049. printer_id,
  2050. )
  2051. item.ams_mapping = None
  2052. # Resolve the AMS mapping for the assigned printer. It is
  2053. # always missing here, so this computes it, and it also
  2054. # self-heals a bogus stored [-1] (#2589).
  2055. unmappable = await self._ensure_ams_mapping(db, printer_id, item)
  2056. if unmappable:
  2057. await self._fail_unmappable_item(db, item, printer_id, unmappable)
  2058. continue
  2059. # Filament-deficit pre-dispatch check (#1496).
  2060. if await self._block_on_filament_deficit(db, item):
  2061. continue
  2062. # Unmatched-filament pre-dispatch check (#2799). Model-based
  2063. # selection already filters on filament type, so this is a
  2064. # backstop for an AMS that changed between assignment and
  2065. # dispatch, and for a type loaded on the wrong nozzle.
  2066. # The assignment made above is released with the hold —
  2067. # this item asked for a model, not this printer.
  2068. if await self._block_on_unmatched_filament(db, item, release_assignment=True):
  2069. continue
  2070. _claim_library_row(item)
  2071. dispatch_ids.append(item.id)
  2072. claim_printer(printer_id)
  2073. # SJF starvation guard: mark items that were jumped
  2074. if sjf_enabled and item.print_time_seconds is not None:
  2075. for other in items:
  2076. if (
  2077. other.id != item.id
  2078. and other.id not in dispatch_ids
  2079. and other.status == "pending"
  2080. and _could_take_printer(other, printer_id, item.target_model)
  2081. and not other.been_jumped
  2082. and other.position < item.position
  2083. and (
  2084. other.print_time_seconds is None
  2085. or other.print_time_seconds > item.print_time_seconds
  2086. )
  2087. ):
  2088. other.been_jumped = True
  2089. await db.commit()
  2090. # Log the decisions BEFORE dispatching. The dispatch below blocks for
  2091. # as long as the slowest upload takes (minutes on a big 3MF), and a
  2092. # skip summary that only lands after the transfers have finished is
  2093. # useless for working out why an item did not go out.
  2094. if skip_reasons:
  2095. logger.info("Queue skip summary: %s", skip_reasons)
  2096. for pid in sorted(busy_printers):
  2097. reason = busy_reasons.get(pid, "no reason recorded")
  2098. if pid in claimed_printers:
  2099. logger.info("Queue: printer %d reserved — %s", pid, reason)
  2100. continue
  2101. # The three live fields stay, because they are what someone
  2102. # reading a bundle wants next -- but they are labelled as read
  2103. # now, not as the state the decision was made on, which is what
  2104. # made the old line contradict itself.
  2105. state = printer_manager.get_status(pid)
  2106. logger.info(
  2107. "Queue: printer %d unavailable — %s (now: connected=%s, state=%s, awaiting_plate_clear=%s)",
  2108. pid,
  2109. reason,
  2110. printer_manager.is_connected(pid),
  2111. state.state if state else "NO_STATUS",
  2112. printer_manager.is_awaiting_plate_clear(pid),
  2113. )
  2114. # Keep-warm is a comfort feature; dispatch is not. It sits between
  2115. # selection and `_launch_uploads`, so anything raising here would
  2116. # discard this tick's selections — computed AMS mappings and all —
  2117. # and, on a persistent fault, stop the queue dispatching entirely.
  2118. # Same reasoning as the deficit check's guard below: never let an
  2119. # auxiliary check wedge the queue. The bed simply stays wherever it
  2120. # was, and the next tick tries again.
  2121. try:
  2122. await self._apply_keep_warm(
  2123. db, items, dispatch_ids, busy_printers - backoff_printers, require_plate_clear
  2124. )
  2125. except Exception as e:
  2126. logger.warning("Keep-warm pass failed, continuing with dispatch: %s", e, exc_info=True)
  2127. # Read the concurrency limit BEFORE the commit below, not inside
  2128. # _dispatch_selected(). A SELECT on this session after the commit
  2129. # implicitly opens a fresh transaction that nothing then closes, and
  2130. # it would stay open for the whole dispatch — minutes of "idle in
  2131. # transaction" on Postgres (pinned MVCC snapshot, vacuum blocked),
  2132. # and on SQLite a pinned WAL read snapshot that stops the WAL being
  2133. # checkpointed while every dispatch is writing to it.
  2134. upload_limit = max(1, await self._get_int_setting(db, "queue_max_concurrent_uploads", default=4))
  2135. # Selection is done; every decision above is recorded on `db`
  2136. # (model-based printer assignment, computed ams_mapping). Flush it
  2137. # before the dispatch tasks open their own sessions, or they will
  2138. # read a row that still says printer_id=None. This also releases the
  2139. # connection back to the pool for the duration of the dispatch.
  2140. await db.commit()
  2141. if dispatch_ids:
  2142. item_printers = {it.id: it.printer_id for it in items}
  2143. self._launch_uploads(dispatch_ids, item_printers, upload_limit)
  2144. # Auto-drying: start drying on idle printers that have no pending queue items
  2145. await self._check_auto_drying(db, items, dispatching_printers)
  2146. # Low filament: alert on assigned spools that have crossed their
  2147. # low-stock threshold (#2913). Runs on both paths out of this method
  2148. # for the same reason auto-drying does — an empty queue does not mean
  2149. # the spools in the printers stopped mattering.
  2150. await self._check_filament_low(db)
  2151. # Stock forecast: alert when a filament SKU reaches its reorder point or
  2152. # is about to run out before a replenishment could arrive (#2955).
  2153. await self._check_stock_forecast(db)
  2154. # Keep the loop on the fast interval while any upload is in flight so
  2155. # a slot freed mid-tick refills within seconds rather than after the
  2156. # 30 s idle sleep (#2602). Selecting anything this pass (launched or
  2157. # deferred because the pool was full) also counts as productive.
  2158. return bool(dispatch_ids) or bool(self._inflight)
  2159. def _launch_uploads(self, item_ids: list[int], item_printers: dict[int, int | None], limit: int) -> None:
  2160. """Launch selected uploads as a refillable pool, capped at ``limit`` (#2602).
  2161. Dispatch used to happen inline in the selection loop: ``await
  2162. _start_print(db, item)`` per item in turn. Since ``_start_print``
  2163. performs the FTP upload, that serialized every printer behind every
  2164. other printer's transfer even though the printers are independent
  2165. machines; #2555 moved it to a parallel ``asyncio.gather()``. But that
  2166. gather was awaited before ``check_queue`` returned, so the run loop
  2167. stayed blocked until the *slowest* upload in the batch finished — a
  2168. 513 s upload left 15 of 16 configured slots idle for 8.5 minutes on a
  2169. 93-printer farm even as other printers came free (#2602).
  2170. Each upload now runs as an independent background task tracked in
  2171. ``self._inflight``. check_queue excludes in-flight item_ids (still
  2172. `pending` until their upload completes) and their printers from the
  2173. next pass's selection, and this method launches at most
  2174. ``limit - len(self._inflight)`` new uploads, so a freed slot refills on
  2175. the next fast tick instead of waiting out the whole batch. The bound
  2176. exists because the printers are independent but the host is not: each
  2177. in-flight upload holds a thread in the FTP pool, a TLS session and a
  2178. file handle.
  2179. The no-overlapping-dispatch invariant the batch-await used to provide
  2180. is now carried by the in-flight exclusion in check_queue. Everything
  2181. else — the pending->printing CAS, the busy-printer guard (#2598), the
  2182. per-printer hold, and each item's independent failure handling — still
  2183. lives in ``_start_print`` and runs per task exactly as before.
  2184. Synchronous on purpose: it registers every launched task into
  2185. ``self._inflight`` before returning, so the next (sequential) tick sees
  2186. an accurate in-flight count with no interleaving await.
  2187. """
  2188. free = limit - len(self._inflight)
  2189. if free <= 0:
  2190. logger.info(
  2191. "Upload pool full (%d/%d in flight) — deferring %d item(s) to a later tick: %s",
  2192. len(self._inflight),
  2193. limit,
  2194. len(item_ids),
  2195. item_ids,
  2196. )
  2197. return
  2198. to_launch = item_ids[:free]
  2199. deferred = item_ids[free:]
  2200. logger.info(
  2201. "Launching %d upload(s) (pool %d/%d in flight)%s",
  2202. len(to_launch),
  2203. len(self._inflight),
  2204. limit,
  2205. f" — deferring {deferred} to a later tick" if deferred else "",
  2206. )
  2207. for item_id in to_launch:
  2208. task = spawn_background_task(
  2209. self._dispatch_one(item_id, item_printers.get(item_id)),
  2210. name=f"queue-upload-{item_id}",
  2211. )
  2212. self._inflight[item_id] = (task, item_printers.get(item_id))
  2213. # Prune on completion so the freed slot is refillable next tick.
  2214. # spawn_background_task already logs any uncaught exception; this
  2215. # only reclaims the pool slot (fires on success, failure, or cancel).
  2216. task.add_done_callback(lambda _t, iid=item_id: self._inflight.pop(iid, None))
  2217. async def _dispatch_one(self, item_id: int, selected_printer_id: int | None = None) -> None:
  2218. """Upload + start one queue item in its own session (pool worker, #2602).
  2219. Its own session: pool workers run concurrently and an AsyncSession is
  2220. not safe to share across tasks; it also keeps a slow upload from pinning
  2221. the scheduler's session (and, on SQLite, its transaction) open for the
  2222. transfer's duration.
  2223. ``selected_printer_id`` is the printer this item was selected for, taken
  2224. from the same snapshot the caller used. It exists so the preheat pin can
  2225. be unwound on the paths that never reach the ``finally`` below — see the
  2226. claim failure a few lines down. Optional so the direct-call tests keep
  2227. working; when it is absent those paths simply behave as they did before.
  2228. """
  2229. async with async_session() as item_db:
  2230. # Claim the row for dispatch BEFORE reading the printer snapshot or
  2231. # touching any slow I/O (#2615). The claim is an atomic CAS on
  2232. # (status='pending', dispatching_at IS NULL); while it's held the edit
  2233. # routes reject reassignment (409), so printer_id can't change out from
  2234. # under the in-flight upload and split the queue row from the
  2235. # archive/expected-print/physical command.
  2236. if not await self._claim_for_dispatch(item_db, item_id):
  2237. logger.info(
  2238. "Queue item %s not claimable for dispatch (cancelled, removed, or already claimed) — skipping",
  2239. item_id,
  2240. )
  2241. # This return is outside the try/finally below, so the rollback
  2242. # has to happen here. Selecting this item already handed any
  2243. # keep-warm hold on its printer over to the preheat pin
  2244. # (`_sweep_keep_warm`), on the promise that this dispatch would
  2245. # unwind it. Bailing without doing so leaves the bed hot with
  2246. # nothing tracking it: the keep-warm entry is gone, so the
  2247. # max-duration cap no longer applies, and if this was the
  2248. # printer's last pending item nothing else will ever turn it
  2249. # off. Reachable whenever a cancel or delete lands between
  2250. # selection and the claim.
  2251. if selected_printer_id is not None:
  2252. self._rollback_preheat_pin(item_id, selected_printer_id)
  2253. return
  2254. # Seeded from the caller's snapshot so the `item vanished` return
  2255. # below still unwinds the pin; overwritten with the row's own
  2256. # printer_id as soon as we have it.
  2257. item_printer_id: int | None = selected_printer_id
  2258. try:
  2259. item = await item_db.get(PrintQueueItem, item_id)
  2260. if not item:
  2261. logger.info("Queue item %s vanished after claim — skipping", item_id)
  2262. return
  2263. item_printer_id = item.printer_id
  2264. await self._start_print(item_db, item)
  2265. finally:
  2266. # Undo an expected-print registration whose print command never
  2267. # went out. One choke point covers every way `_start_print` can
  2268. # end without sending: a raised exception (a DB failure mid-
  2269. # dispatch is the reported case), an early return, a cancel
  2270. # winning the #1853 CAS, or `start_print()` returning False.
  2271. # A confirmed send removes the entry itself, so this is a no-op
  2272. # on the happy path.
  2273. self._rollback_unconfirmed_expected_print(item_id)
  2274. # Mirror the pre-#1625 background-dispatch lifecycle: a
  2275. # reservation survives only after start_print() accepted the
  2276. # command. Failure, cancellation, deferral, and exceptions all
  2277. # release it here.
  2278. await asyncio.shield(self._release_unconfirmed_budget_reservation(item_id))
  2279. # Unwind preheat state (bed/chamber/airduct) if the
  2280. # dispatch aborted before the print's own gcode took over.
  2281. # `_start_print` clears the pin on successful `start_print()`;
  2282. # anything still present here is by definition an aborted
  2283. # dispatch and gets rolled back so the printer isn't left
  2284. # heating for a job that isn't happening.
  2285. if item_printer_id is not None:
  2286. self._rollback_preheat_pin(item_id, item_printer_id)
  2287. # The cancellation flag only has meaning while this dispatch is
  2288. # running; drop it so the set cannot grow without bound and a
  2289. # re-queued item never inherits a stale cancellation.
  2290. self._cancelled_dispatches.discard(item_id)
  2291. # Release the claim on every exit. Once dispatch has finished the
  2292. # row's status carries the lock (printing/failed/cancelled are all
  2293. # != pending), so the token is only needed for the duration of the
  2294. # upload. A row left pending (e.g. busy-printer deferral) becomes
  2295. # dispatchable again on the next tick.
  2296. await self._clear_dispatch_claim(item_db, item_id)
  2297. async def _requeue_after_upload_refused(
  2298. self,
  2299. db: AsyncSession,
  2300. item: PrintQueueItem,
  2301. printer: Printer,
  2302. error_msg: str,
  2303. toast_uid: int | None,
  2304. ) -> None:
  2305. """Put an item back in the queue after its file never reached the printer (#3210).
  2306. The item keeps its printer. Its AMS mapping was computed against that
  2307. printer's trays, and nothing on the row says whether a mapping was
  2308. computed or set by the user, so moving it to a sibling could print from
  2309. the wrong slots. The printer goes into `_upload_backoff` instead, which
  2310. keeps every other item away from it; this one waits there and is the
  2311. only thing that knocks again, once per backoff window. The window
  2312. grows with each refusal in a row; see ``_upload_backoff_seconds``.
  2313. `dispatch_attempts` is not charged: that budget bounds a printer that
  2314. takes the file and then never starts, which this is not.
  2315. """
  2316. refusals = self._upload_refusals.get(printer.id, 0) + 1
  2317. self._upload_refusals[printer.id] = refusals
  2318. backoff = _upload_backoff_seconds(refusals)
  2319. self._upload_backoff[printer.id] = time.monotonic() + backoff
  2320. # The row is still `pending` -- it only moves to `printing` after a
  2321. # successful upload -- so there is no status to write back. Writing one
  2322. # anyway would undo a cancel that landed during the upload.
  2323. if item.error_message:
  2324. item.error_message = None
  2325. await db.commit()
  2326. logger.warning(
  2327. "Queue item %s: upload to printer %s (%s) never reached it — %s Kept in the queue; "
  2328. "refusal %d in a row, so the printer is out of dispatch for %ds.",
  2329. item.id,
  2330. printer.id,
  2331. printer.name,
  2332. error_msg,
  2333. refusals,
  2334. backoff,
  2335. )
  2336. try:
  2337. # Closes the dispatch toast for this attempt. Without it the toast
  2338. # keeps spinning on an upload that has ended.
  2339. await ws_manager.send_queue_item_failed(
  2340. user_id=toast_uid,
  2341. queue_item_id=item.id,
  2342. printer_id=item.printer_id,
  2343. reason="upload_failed",
  2344. )
  2345. except Exception:
  2346. pass # toast is best-effort
  2347. def _rollback_unconfirmed_expected_print(self, item_id: int) -> None:
  2348. """Drop an expectation for a print command that was never sent.
  2349. Best-effort and never raises: this runs in the ``finally`` of dispatch,
  2350. where the interesting exception is usually the one already propagating.
  2351. """
  2352. pending = self._unconfirmed_expected_print.pop(item_id, None)
  2353. if pending is None:
  2354. return
  2355. printer_id, remote_filename, archive_id = pending
  2356. try:
  2357. from backend.app.main import unregister_expected_print
  2358. unregister_expected_print(printer_id, remote_filename, archive_id)
  2359. except Exception:
  2360. logger.warning(
  2361. "Queue item %s: failed to unregister expected print (printer=%s, file=%s, archive=%s)",
  2362. item_id,
  2363. printer_id,
  2364. remote_filename,
  2365. archive_id,
  2366. exc_info=True,
  2367. )
  2368. async def _release_unconfirmed_budget_reservation(self, item_id: int) -> None:
  2369. """Release a queue reservation without touching the dispatch session."""
  2370. if item_id not in self._unconfirmed_budget_reservations:
  2371. return
  2372. for attempt in range(1, 4):
  2373. async with async_session() as cleanup_db:
  2374. try:
  2375. await release_budget_reservation(
  2376. cleanup_db,
  2377. source_type="print_queue",
  2378. source_id=item_id,
  2379. status="released",
  2380. )
  2381. await cleanup_db.commit()
  2382. self._unconfirmed_budget_reservations.discard(item_id)
  2383. return
  2384. except Exception as exc:
  2385. try:
  2386. await cleanup_db.rollback()
  2387. except Exception:
  2388. pass
  2389. if attempt == 3:
  2390. logger.error(
  2391. "Queue item %s: failed to release budget reservation after %d attempts: %s",
  2392. item_id,
  2393. attempt,
  2394. exc,
  2395. )
  2396. return
  2397. await asyncio.sleep(0.5 * attempt)
  2398. @staticmethod
  2399. def _reported_bed_target(printer_id: int) -> int | None:
  2400. """The bed target firmware currently reports, or None if it can't be read.
  2401. None means "no evidence", not "zero" — callers must not treat it as a
  2402. temperature. Deliberately total: this feeds cleanup paths that run in a
  2403. ``finally``, where a malformed status must not become the exception the
  2404. caller sees.
  2405. """
  2406. try:
  2407. state = printer_manager.get_status(printer_id)
  2408. if state is None:
  2409. return None
  2410. temps = state.temperatures
  2411. if not isinstance(temps, dict):
  2412. return None
  2413. return int(float(temps.get("bed_target", 0) or 0))
  2414. except (TypeError, ValueError, AttributeError):
  2415. return None
  2416. def _rollback_preheat_pin(self, item_id: int, printer_id: int) -> None:
  2417. """Unwind everything preheat set when dispatch did NOT hand off to a running print.
  2418. Turns the bed heater off, the chamber heater off, and opens the
  2419. airduct flap back to cooling — for whichever of those preheat
  2420. actually applied. `_start_print` clears the pin on successful
  2421. `start_print()`; anything still present when `_dispatch_one` exits
  2422. is by definition an aborted dispatch and gets rolled back here.
  2423. The bed is the one action that can be declined: if firmware has since
  2424. been given a target other than the one we pinned, it belongs to someone
  2425. else and is left alone. See the comment at that branch.
  2426. Also called directly from `_dispatch_one`'s claim-failure return, which
  2427. never reaches the ``finally``.
  2428. Best-effort and never raises — this runs in the ``finally`` of dispatch.
  2429. """
  2430. pin = self._preheat_pin.pop(printer_id, set())
  2431. pinned_bed = self._preheat_pin_bed.pop(printer_id, None)
  2432. if not pin:
  2433. return
  2434. client = printer_manager.get_client(printer_id)
  2435. if client is None:
  2436. logger.info(
  2437. "Dispatch item %s (printer %d): preheat rollback skipped — no client",
  2438. item_id,
  2439. printer_id,
  2440. )
  2441. return
  2442. if "bed" in pin:
  2443. # Only undo our own target. If firmware reports something else, the
  2444. # user or another writer owns the bed now and zeroing it would
  2445. # clobber their choice -- the same guard `_release_keep_warm`
  2446. # applies to a keep-warm hold.
  2447. #
  2448. # Every uncertain case switches the bed off rather than leaving it:
  2449. # no recorded target (a pin written before this bookkeeping, or a
  2450. # setter that raised after pinning) and an unreadable status both
  2451. # fall through. A bed left hot with no owner is the worse failure,
  2452. # and this runs in a `finally` where raising would mask the real
  2453. # exception.
  2454. cur_bed_target = self._reported_bed_target(printer_id) if pinned_bed is not None else None
  2455. if cur_bed_target is not None and cur_bed_target != pinned_bed:
  2456. logger.info(
  2457. "Dispatch item %s (printer %d): rollback skipped bed → 0 (firmware target %d != pinned %d)",
  2458. item_id,
  2459. printer_id,
  2460. cur_bed_target,
  2461. pinned_bed,
  2462. )
  2463. else:
  2464. try:
  2465. client.set_bed_temperature(0)
  2466. except Exception as exc:
  2467. logger.warning("Dispatch item %s: rollback bed → 0 failed: %s", item_id, exc)
  2468. if "chamber" in pin:
  2469. try:
  2470. client.set_chamber_temperature(0)
  2471. except Exception as exc:
  2472. logger.warning("Dispatch item %s: rollback chamber → 0 failed: %s", item_id, exc)
  2473. if "airduct" in pin:
  2474. try:
  2475. client.set_airduct_mode("cooling")
  2476. except Exception as exc:
  2477. logger.warning("Dispatch item %s: rollback airduct → cooling failed: %s", item_id, exc)
  2478. logger.info(
  2479. "Dispatch item %s (printer %d): preheat rollback → %s",
  2480. item_id,
  2481. printer_id,
  2482. sorted(pin),
  2483. )
  2484. async def _claim_for_dispatch(self, db: AsyncSession, item_id: int) -> bool:
  2485. """Atomically stamp ``dispatching_at`` on a still-pending, unclaimed row.
  2486. Returns True if this call won the claim, False if the row was already
  2487. claimed, no longer pending (cancelled mid-tick), or removed. The CAS is
  2488. the load-bearing guard against reassign-during-dispatch (#2615)."""
  2489. res = await db.execute(
  2490. update(PrintQueueItem)
  2491. .where(PrintQueueItem.id == item_id)
  2492. .where(PrintQueueItem.status == "pending")
  2493. .where(PrintQueueItem.dispatching_at.is_(None))
  2494. .values(dispatching_at=datetime.now(timezone.utc))
  2495. )
  2496. await db.commit()
  2497. return res.rowcount > 0
  2498. async def _clear_dispatch_claim(self, db: AsyncSession, item_id: int) -> None:
  2499. """Clear the dispatch claim (#2615). Best-effort: a failure here must not
  2500. mask the dispatch outcome.
  2501. Retried, because the failure mode in practice is transient and narrow: a
  2502. database that is momentarily unreachable — PostgreSQL out of connection
  2503. slots is the observed case — refuses this write for a second or two while
  2504. the dispatch that just ended is still holding the row out of the selection
  2505. query. One attempt was enough to wedge the item; a couple of spaced
  2506. attempts clear it. Each attempt rolls back first, since a failed write
  2507. leaves the session needing it before it can be reused.
  2508. If every attempt fails, ``_clear_stale_dispatch_claims`` picks the row up
  2509. on the next quiet tick.
  2510. """
  2511. for attempt in range(1, 4):
  2512. try:
  2513. await db.execute(update(PrintQueueItem).where(PrintQueueItem.id == item_id).values(dispatching_at=None))
  2514. await db.commit()
  2515. return
  2516. except Exception as exc:
  2517. try:
  2518. await db.rollback()
  2519. except Exception:
  2520. pass
  2521. if attempt == 3:
  2522. logger.warning(
  2523. "Queue item %s: failed to clear dispatch claim after %d attempts: %s "
  2524. "— a later quiet tick will release it",
  2525. item_id,
  2526. attempt,
  2527. exc,
  2528. )
  2529. return
  2530. await asyncio.sleep(0.5 * attempt)
  2531. async def _printers_for_model(
  2532. self,
  2533. db: AsyncSession,
  2534. model: str,
  2535. target_location: str | None = None,
  2536. printer_scope: PrinterScope = ALL_PRINTERS,
  2537. ) -> list[Printer]:
  2538. """Active printers of *model*, optionally narrowed to one location.
  2539. Shared by the matcher and by the smart-plug wake step (#2786) so both
  2540. answer "which printers can this job run on" from one query — a job can
  2541. only be woken onto a printer the matcher would also have considered.
  2542. ``printer_scope`` is the job creator's (#1727).
  2543. """
  2544. normalized_model = normalize_printer_model(model) or model
  2545. query = (
  2546. select(Printer)
  2547. .where(func.lower(Printer.model) == normalized_model.lower())
  2548. .where(Printer.is_active == True) # noqa: E712
  2549. )
  2550. if target_location:
  2551. query = query.where(Printer.location == target_location)
  2552. if (clause := printer_scope.where_strict(Printer.id)) is not None:
  2553. query = query.where(clause)
  2554. result = await db.execute(query)
  2555. return list(result.scalars().all())
  2556. async def _wakeable_printer_ids(self, db: AsyncSession) -> set[int]:
  2557. """Printer IDs that at least one enabled ``auto_on`` plug can power on.
  2558. Read once per queue check rather than per printer: it decides both
  2559. whether the wake step has anything to do and how an offline printer is
  2560. worded in the waiting reason — "Offline" and "offline with no Auto On
  2561. plug" are different problems, and the second is the one the user has to
  2562. fix themselves (#2786).
  2563. """
  2564. result = await db.execute(
  2565. select(SmartPlug.printer_id)
  2566. .where(SmartPlug.printer_id.is_not(None))
  2567. .where(SmartPlug.enabled == True) # noqa: E712
  2568. .where(SmartPlug.auto_on == True) # noqa: E712
  2569. )
  2570. return {pid for (pid,) in result.all() if pid is not None}
  2571. def _wake_recently_failed(self, printer_id: int) -> bool:
  2572. """True while this printer's failed power-on is still cooling off (#2786)."""
  2573. deadline = self._wake_failures.get(printer_id)
  2574. if deadline is None:
  2575. return False
  2576. if time.monotonic() >= deadline:
  2577. del self._wake_failures[printer_id]
  2578. return False
  2579. return True
  2580. async def _wake_printer_for_model(
  2581. self,
  2582. db: AsyncSession,
  2583. candidates: list[_ModelCandidate],
  2584. target_location: str | None,
  2585. exclude_ids: set[int],
  2586. wakeable_ids: set[int],
  2587. require_plate_clear: bool,
  2588. printer_scope: PrinterScope = ALL_PRINTERS,
  2589. ) -> tuple[int | None, int | None]:
  2590. """Power on one offline printer a model-based item could run on (#2786).
  2591. The fixed-printer branch has powered a printer on since smart plugs
  2592. existed. The model-based branch never could: its matcher drops an
  2593. offline printer into the "Offline:" waiting reason and nothing looks at
  2594. its plugs, so a class-targeted job with every matching printer switched
  2595. off sat pending forever. The reporter's log is the controlled
  2596. experiment — the same item, same plug, same Auto On setting, dispatched
  2597. the moment they edited it onto a specific printer.
  2598. Returns ``(woken_id, attempted_id)``. ``attempted_id`` is set whenever a
  2599. power-on was actually tried, so the caller can tell "nothing here was
  2600. wakeable" (both None — cheap, other items may still find something)
  2601. from "we tried and it did not come up" (only ``attempted_id`` — the
  2602. boot timeout has already been spent).
  2603. A printer whose last known trays cannot satisfy the job is passed over
  2604. rather than woken (#2876): the colours are readable while it is off, so
  2605. switching a farm on one machine at a time to discover them wakes
  2606. printers that could never have taken the job.
  2607. Deliberately does NOT go on to match the job once a printer is up: AMS
  2608. trays arrive with the first status push after connect, so a filament
  2609. check against a printer that booted seconds ago can reject the printer
  2610. we just woke. The next queue pass matches it with live state.
  2611. At most one printer per pass. Each wake blocks the queue loop for the
  2612. boot wait, and a queue of ten class-targeted jobs must not switch on
  2613. ten printers inside one check.
  2614. """
  2615. for candidate in candidates:
  2616. if not candidate.target_model:
  2617. continue
  2618. required_types, filament_overrides = _filament_constraints(candidate)
  2619. printers = await self._printers_for_model(db, candidate.target_model, target_location, printer_scope)
  2620. for printer in sorted(printers, key=lambda p: p.id):
  2621. if printer.id in exclude_ids or printer.id not in wakeable_ids:
  2622. continue
  2623. if printer_manager.is_connected(printer.id):
  2624. continue
  2625. if self._wake_recently_failed(printer.id):
  2626. # Its plug did not bring it back a moment ago. Move on to a
  2627. # sibling instead of spending this pass — and every pass —
  2628. # on the same printer.
  2629. continue
  2630. if require_plate_clear and printer_manager.is_awaiting_plate_clear(printer.id):
  2631. # Waking this one buys nothing: it would boot into IDLE and
  2632. # then be held by the plate-clear gate, which is exactly
  2633. # what the reporter's log shows happening for 80 minutes
  2634. # after a fixed-printer wake. The flag is Bambuddy-side and
  2635. # persisted, so it is readable while the printer is off.
  2636. logger.info(
  2637. "Not powering on printer %s for a %s job: it is awaiting plate-clear acknowledgment",
  2638. printer.id,
  2639. candidate.target_model,
  2640. )
  2641. continue
  2642. shortfall = self._cached_filament_shortfall(printer.id, required_types, filament_overrides)
  2643. if shortfall:
  2644. logger.info(
  2645. "Not powering on printer %s for a %s job: last known filament cannot satisfy it (needs %s)",
  2646. printer.id,
  2647. candidate.target_model,
  2648. ", ".join(shortfall),
  2649. )
  2650. continue
  2651. plugs = await self._get_smart_plugs(db, printer.id)
  2652. auto_on_plugs = [p for p in plugs if p.auto_on and p.enabled]
  2653. if not auto_on_plugs:
  2654. # wakeable_ids said otherwise — the plug changed under us
  2655. # mid-pass. Nothing to do but move on.
  2656. continue
  2657. logger.info(
  2658. "No %s printer available for a queued job; powering on offline printer %s via smart plug(s)",
  2659. candidate.target_model,
  2660. printer.id,
  2661. )
  2662. primary_plug = self._pick_power_plug(auto_on_plugs)
  2663. if not await self._power_on_and_wait(primary_plug, printer.id, db):
  2664. logger.warning(
  2665. "Could not power on printer %s via smart plug; not trying it again for %ss",
  2666. printer.id,
  2667. self._wake_failure_cooloff,
  2668. )
  2669. self._wake_failures[printer.id] = time.monotonic() + self._wake_failure_cooloff
  2670. return None, printer.id
  2671. for extra_plug in [p for p in auto_on_plugs if p.id != primary_plug.id]:
  2672. try:
  2673. service = await smart_plug_manager.get_service_for_plug(extra_plug, db)
  2674. await service.turn_on(extra_plug)
  2675. logger.info("Also powered on plug '%s' for printer %s", extra_plug.name, printer.id)
  2676. except Exception as e:
  2677. logger.warning("Failed to power on extra plug '%s': %s", extra_plug.name, e)
  2678. return printer.id, printer.id
  2679. return None, None
  2680. async def _find_idle_printer_for_model(
  2681. self,
  2682. db: AsyncSession,
  2683. model: str,
  2684. exclude_ids: set[int],
  2685. required_filament_types: list[str] | None = None,
  2686. target_location: str | None = None,
  2687. filament_overrides: list[dict] | None = None,
  2688. require_plate_clear: bool = True,
  2689. wakeable_ids: set[int] | None = None,
  2690. printer_scope: PrinterScope = ALL_PRINTERS,
  2691. ) -> tuple[int | None, str | None]:
  2692. """Find an idle, connected printer matching the model with compatible filaments.
  2693. Args:
  2694. db: Database session
  2695. model: Printer model to match (e.g., "X1C", "P1S")
  2696. exclude_ids: Printer IDs to exclude (already busy)
  2697. required_filament_types: Optional list of filament types needed (e.g., ["PLA", "PETG"])
  2698. If provided, only printers with all required types loaded will match.
  2699. target_location: Optional location filter. If provided, only printers in this location are considered.
  2700. filament_overrides: Optional list of override dicts. Each entry may include
  2701. ``force_color_match: true`` to require an exact type+color match
  2702. on the printer for that slot. Without the flag the existing
  2703. colour-preference logic applies.
  2704. wakeable_ids: Printers a smart plug can power on (#2786). Only changes how an
  2705. offline printer is worded: one Bambuddy will switch on reads
  2706. differently from one the user has to go and switch on themselves.
  2707. Returns:
  2708. Tuple of (printer_id, waiting_reason):
  2709. - (printer_id, None) if a matching printer was found
  2710. - (None, reason) if no printer is available, with explanation
  2711. """
  2712. normalized_model = normalize_printer_model(model) or model
  2713. printers = await self._printers_for_model(db, model, target_location, printer_scope)
  2714. location_suffix = f" in {target_location}" if target_location else ""
  2715. if not printers:
  2716. return None, f"No active {normalized_model} printers{location_suffix} configured"
  2717. # Separate force-matched overrides from preference-only overrides
  2718. force_overrides = [o for o in (filament_overrides or []) if o.get("force_color_match")]
  2719. pref_overrides = [o for o in (filament_overrides or []) if not o.get("force_color_match")]
  2720. # Track reasons for skipping printers
  2721. printers_busy = []
  2722. printers_offline = []
  2723. printers_offline_no_plug = []
  2724. printers_missing_filament: list[tuple[str, list[str]]] = []
  2725. candidates: list[tuple[int, int]] = [] # (printer_id, color_match_count)
  2726. for printer in printers:
  2727. if printer.id in exclude_ids:
  2728. # Printer is already claimed by another job in this scheduling run.
  2729. # For force-color jobs, still check if the color would match — if not,
  2730. # report it as a color mismatch rather than plain "Busy" so the user
  2731. # knows the job needs a filament change, not just to wait for availability.
  2732. if force_overrides and not pref_overrides:
  2733. missing_colors = self._get_missing_force_color_slots(printer.id, force_overrides)
  2734. if missing_colors:
  2735. printers_missing_filament.append((printer.name, missing_colors))
  2736. continue
  2737. printers_busy.append(printer.name)
  2738. continue
  2739. is_connected = printer_manager.is_connected(printer.id)
  2740. is_idle = self._is_printer_idle(printer.id, require_plate_clear) if is_connected else False
  2741. if not is_connected:
  2742. # An offline printer whose last known filament cannot run this
  2743. # job is reported as needing filament rather than as offline
  2744. # (#2876). It is also the printer the smart-plug step will now
  2745. # decline to switch on, and "Offline:" on its own would leave
  2746. # that decision looking like nothing happening at all.
  2747. shortfall = self._cached_filament_shortfall(printer.id, required_filament_types, filament_overrides)
  2748. if shortfall:
  2749. printers_missing_filament.append((printer.name, shortfall))
  2750. elif wakeable_ids is not None and printer.id not in wakeable_ids:
  2751. printers_offline_no_plug.append(printer.name)
  2752. else:
  2753. printers_offline.append(printer.name)
  2754. continue
  2755. if not is_idle:
  2756. # Printer is currently printing. For force-color jobs, check whether the
  2757. # loaded color would satisfy the requirement — if not, surface it as a
  2758. # color-mismatch reason rather than plain "Busy" so the user understands
  2759. # that the job is waiting for a filament change, not just printer availability.
  2760. if force_overrides and not pref_overrides:
  2761. missing_colors = self._get_missing_force_color_slots(printer.id, force_overrides)
  2762. if missing_colors:
  2763. printers_missing_filament.append((printer.name, missing_colors))
  2764. logger.debug(
  2765. "Printer %s (%s) is busy but also has wrong force-color: %s",
  2766. printer.id,
  2767. printer.name,
  2768. missing_colors,
  2769. )
  2770. continue
  2771. printers_busy.append(printer.name)
  2772. continue
  2773. # Validate filament compatibility if required types are specified
  2774. if required_filament_types:
  2775. missing = self._get_missing_filament_types(printer.id, required_filament_types)
  2776. if missing:
  2777. # When force_overrides are present, enrich missing entries with color info
  2778. # so the "Waiting on" message includes "TYPE (color)" instead of just "TYPE"
  2779. if force_overrides:
  2780. force_color_map = {
  2781. (o.get("type") or "").upper(): o.get("color_name") or o.get("color", "?")
  2782. for o in force_overrides
  2783. }
  2784. missing_enriched = [
  2785. f"{t} ({force_color_map[t_upper]})" if (t_upper := t.upper()) in force_color_map else t
  2786. for t in missing
  2787. ]
  2788. printers_missing_filament.append((printer.name, missing_enriched))
  2789. else:
  2790. printers_missing_filament.append((printer.name, missing))
  2791. logger.debug("Skipping printer %s (%s) - missing filaments: %s", printer.id, printer.name, missing)
  2792. continue
  2793. # Force color match: ALL flagged slots must have an exact type+color match
  2794. if force_overrides:
  2795. missing_colors = self._get_missing_force_color_slots(printer.id, force_overrides)
  2796. if missing_colors:
  2797. printers_missing_filament.append((printer.name, missing_colors))
  2798. logger.debug(
  2799. "Skipping printer %s (%s) - missing force-matched colors: %s",
  2800. printer.id,
  2801. printer.name,
  2802. missing_colors,
  2803. )
  2804. continue
  2805. # If preference-only overrides exist, rank by color matches (existing behaviour)
  2806. if pref_overrides:
  2807. color_matches = self._count_override_color_matches(printer.id, pref_overrides)
  2808. if color_matches > 0:
  2809. candidates.append((printer.id, color_matches))
  2810. else:
  2811. override_colors = [f"{o.get('type', '?')} ({o.get('color', '?')})" for o in pref_overrides]
  2812. printers_missing_filament.append((printer.name, override_colors))
  2813. logger.debug("Skipping printer %s (%s) - no matching override colors", printer.id, printer.name)
  2814. continue
  2815. elif force_overrides:
  2816. # Passed all force checks — immediately eligible (no preference ordering needed)
  2817. return printer.id, None
  2818. else:
  2819. # No overrides at all - take first available (existing behavior)
  2820. return printer.id, None
  2821. # If we have candidates from preference override matching, pick the one with most color matches
  2822. if candidates:
  2823. candidates.sort(key=lambda c: c[1], reverse=True)
  2824. return candidates[0][0], None
  2825. # Build waiting reason from what we found
  2826. reasons = []
  2827. if printers_missing_filament:
  2828. # Filament/color mismatch is most actionable - show first
  2829. if force_overrides and not pref_overrides:
  2830. # All mismatches are force-color failures — use descriptive message only;
  2831. # but only if there are no busy printers that DO have the matching color.
  2832. # If a printer has the right color but is busy, surface "Busy" instead so
  2833. # the user knows the job will start automatically once that printer is free.
  2834. # Same for a printer that is merely offline: Bambuddy switches that one on
  2835. # by itself, so the job is not actually waiting on anybody to change a
  2836. # spool (#2876 — offline printers reach this list now that a switched-off
  2837. # printer's own filament is read).
  2838. if not printers_busy and not printers_offline:
  2839. all_missing = sorted({c for _, cols in printers_missing_filament for c in cols})
  2840. return None, f"No matching material/color. Waiting on {', '.join(all_missing)}"
  2841. # else: fall through — the self-resolving entries are appended below
  2842. else:
  2843. names_and_missing = [
  2844. f"{name} (needs {', '.join(missing)})" for name, missing in printers_missing_filament
  2845. ]
  2846. reasons.append(f"Waiting for filament: {'; '.join(names_and_missing)}")
  2847. if printers_busy:
  2848. reasons.append(f"Busy: {', '.join(printers_busy)}")
  2849. if printers_offline:
  2850. reasons.append(f"Offline: {', '.join(printers_offline)}")
  2851. if printers_offline_no_plug:
  2852. # Named separately because it is the one entry on this list the
  2853. # user has to act on: no enabled Auto On plug means Bambuddy will
  2854. # never power this printer on for the queue (#2786).
  2855. reasons.append(f"Offline, no Auto On smart plug: {', '.join(printers_offline_no_plug)}")
  2856. return None, " | ".join(reasons) if reasons else f"No available {model} printers{location_suffix}"
  2857. @staticmethod
  2858. def _is_busy_only(waiting_reason: str) -> bool:
  2859. """Check if the waiting reason only contains 'Busy' entries.
  2860. When all matching printers are simply busy printing, the queued job
  2861. will start automatically once a printer finishes — no user action
  2862. is required, so we skip the notification.
  2863. """
  2864. parts = [p.strip() for p in waiting_reason.split(" | ")]
  2865. return all(p.startswith("Busy:") for p in parts)
  2866. def _get_missing_force_color_slots(
  2867. self, printer_id: int, force_overrides: list[dict], raw_data: dict | None = None
  2868. ) -> list[str]:
  2869. """Return descriptive strings for force_color_match slots not satisfied by the printer.
  2870. Each entry in ``force_overrides`` must have ``type`` and ``color`` fields and is expected
  2871. to carry ``force_color_match: True``. The printer must have **every** such slot loaded
  2872. with an exact type+color match.
  2873. When both the override and a candidate tray carry a ``tray_info_idx``, they must also
  2874. match on it: Bambu reports every PLA variant as ``tray_type == "PLA"``, so the
  2875. Basic/Matte/Silk distinction lives only in ``tray_info_idx`` (GFA00/GFA01/GFA06/...).
  2876. Without this, a job sliced for PLA Matte matched every white PLA regardless of variant
  2877. (#2650). If either side lacks an idx (custom/third-party spools report a blank one, and
  2878. older 3MFs carry none) we fall back to the historical type+colour behaviour so those
  2879. setups are unaffected.
  2880. Returns:
  2881. List of ``"TYPE (color)"`` strings for unmatched slots (empty list means all match).
  2882. """
  2883. if raw_data is None:
  2884. status = printer_manager.get_status(printer_id)
  2885. if not status:
  2886. return [f"{o.get('type', '?')} ({o.get('color_name') or o.get('color', '?')})" for o in force_overrides]
  2887. raw_data = status.raw_data
  2888. # Build loaded (type, colour, tray_info_idx) triples from AMS and external spool.
  2889. loaded: list[tuple[str, str, str]] = []
  2890. for ams_unit in raw_data.get("ams", []):
  2891. for tray in ams_unit.get("tray", []):
  2892. tray_type = tray.get("tray_type")
  2893. if tray_type:
  2894. color_norm = (tray.get("tray_color", "") or "").replace("#", "").lower()[:6]
  2895. loaded.append((canonical_filament_type(tray_type), color_norm, tray.get("tray_info_idx", "") or ""))
  2896. for vt in raw_data.get("vt_tray") or []:
  2897. vt_type = vt.get("tray_type")
  2898. if vt_type:
  2899. color_norm = (vt.get("tray_color", "") or "").replace("#", "").lower()[:6]
  2900. loaded.append((canonical_filament_type(vt_type), color_norm, vt.get("tray_info_idx", "") or ""))
  2901. missing = []
  2902. for o in force_overrides:
  2903. o_type = canonical_filament_type(o.get("type") or "")
  2904. o_color = (o.get("color") or "").replace("#", "").lower()[:6]
  2905. o_idx = o.get("tray_info_idx") or ""
  2906. satisfied = any(
  2907. t_type == o_type and t_color == o_color and (not o_idx or not t_idx or o_idx == t_idx)
  2908. for t_type, t_color, t_idx in loaded
  2909. )
  2910. if not satisfied:
  2911. color_label = o.get("color_name") or o.get("color", "?")
  2912. missing.append(f"{o_type} ({color_label})")
  2913. return missing
  2914. def _get_missing_filament_types(
  2915. self, printer_id: int, required_types: list[str], raw_data: dict | None = None
  2916. ) -> list[str]:
  2917. """Get the list of required filament types that are not loaded on the printer.
  2918. Args:
  2919. printer_id: The printer ID
  2920. required_types: List of filament types needed (e.g., ["PLA", "PETG"])
  2921. Returns:
  2922. List of missing filament types (empty if all are loaded)
  2923. """
  2924. if raw_data is None:
  2925. status = printer_manager.get_status(printer_id)
  2926. if not status:
  2927. return required_types # Can't determine, assume all missing
  2928. raw_data = status.raw_data
  2929. # Collect all filament types loaded on this printer (AMS units + external spool)
  2930. # Use canonical types so equivalence groups (e.g. PA-CF/PA12-CF/PAHT-CF) match.
  2931. loaded_types: set[str] = set()
  2932. # Check AMS units (stored in raw_data["ams"])
  2933. ams_data = raw_data.get("ams", [])
  2934. if ams_data:
  2935. for ams_unit in ams_data:
  2936. for tray in ams_unit.get("tray", []):
  2937. tray_type = tray.get("tray_type")
  2938. if tray_type:
  2939. loaded_types.add(canonical_filament_type(tray_type))
  2940. # Check external spool(s) (virtual tray, stored in raw_data["vt_tray"] as list)
  2941. for vt in raw_data.get("vt_tray") or []:
  2942. vt_type = vt.get("tray_type")
  2943. if vt_type:
  2944. loaded_types.add(canonical_filament_type(vt_type))
  2945. # Find which required types are missing (using canonical type for equivalence)
  2946. missing = []
  2947. for req_type in required_types:
  2948. if canonical_filament_type(req_type) not in loaded_types:
  2949. missing.append(req_type)
  2950. return missing
  2951. def _count_override_color_matches(
  2952. self, printer_id: int, overrides: list[dict], raw_data: dict | None = None
  2953. ) -> int:
  2954. """Count how many filament overrides have an exact color match on the printer.
  2955. Used to prefer printers that already have the desired override colors loaded.
  2956. """
  2957. if raw_data is None:
  2958. status = printer_manager.get_status(printer_id)
  2959. if not status:
  2960. return 0
  2961. raw_data = status.raw_data
  2962. # Collect loaded filaments' type+color pairs
  2963. loaded: set[tuple[str, str]] = set()
  2964. for ams_unit in raw_data.get("ams", []):
  2965. for tray in ams_unit.get("tray", []):
  2966. tray_type = tray.get("tray_type")
  2967. # `or ""`, not a dict default: a slot can carry the key with a
  2968. # null value, and this now runs against switched-off printers
  2969. # too, where nobody is watching for the AttributeError.
  2970. tray_color = tray.get("tray_color") or ""
  2971. if tray_type:
  2972. color_norm = tray_color.replace("#", "").lower()[:6]
  2973. loaded.add((tray_type.upper(), color_norm))
  2974. for vt in raw_data.get("vt_tray") or []:
  2975. vt_type = vt.get("tray_type")
  2976. if vt_type:
  2977. color_norm = (vt.get("tray_color", "") or "").replace("#", "").lower()[:6]
  2978. loaded.add((vt_type.upper(), color_norm))
  2979. matches = 0
  2980. for o in overrides:
  2981. o_type = (o.get("type") or "").upper()
  2982. o_color = (o.get("color") or "").replace("#", "").lower()[:6]
  2983. if (o_type, o_color) in loaded:
  2984. matches += 1
  2985. return matches
  2986. @staticmethod
  2987. def _tray_reading(printer_id: int) -> dict:
  2988. """The best tray reading available for a printer that is not printing.
  2989. Live status first: a printer keeps its last status after the power
  2990. goes, because ``mark_power_off`` blanks ``connected`` and ``state`` and
  2991. leaves ``raw_data`` alone. The manager's own record is the fallback,
  2992. for when the client itself has been dropped and taken its status with
  2993. it — which is what every power-on attempt does.
  2994. An empty result means "we have never heard", not "nothing is loaded":
  2995. the two are indistinguishable from here, and only the second would be
  2996. safe to act on.
  2997. """
  2998. status = printer_manager.get_status(printer_id)
  2999. raw = (status.raw_data if status else None) or {}
  3000. for ams_unit in raw.get("ams") or []:
  3001. if any(tray.get("tray_type") for tray in ams_unit.get("tray", [])):
  3002. return raw
  3003. if any(vt.get("tray_type") for vt in raw.get("vt_tray") or []):
  3004. return raw
  3005. return printer_manager.last_known_trays(printer_id)
  3006. def _cached_filament_shortfall(
  3007. self,
  3008. printer_id: int,
  3009. required_types: list[str] | None,
  3010. filament_overrides: list[dict] | None,
  3011. ) -> list[str]:
  3012. """What a switched-off printer's last known filament cannot provide (#2876).
  3013. The smart-plug wake step used to consider only the model, so a job for a
  3014. colour loaded on the last printer in ID order switched on every earlier
  3015. one in turn, evaluated it, rejected it on colour and left it running.
  3016. Bambuddy knew those colours the whole time. This asks the same three
  3017. questions the matcher asks a live printer — required types, forced
  3018. colours, preferred colours — of the trays it last reported, and returns
  3019. the answers in the same shape the "Waiting for filament" reason uses.
  3020. Empty means the printer may still be able to take the job.
  3021. Fails open, and deliberately: with no tray reading (never connected
  3022. since Bambuddy started, or the cache dropped by a reconnect) this
  3023. returns nothing to report and the printer is treated as it was before.
  3024. A farm restarted while its printers were off must not conclude that
  3025. none of them can print.
  3026. """
  3027. if not required_types and not filament_overrides:
  3028. return []
  3029. raw_data = self._tray_reading(printer_id)
  3030. if not raw_data:
  3031. return []
  3032. force_overrides = [o for o in (filament_overrides or []) if o.get("force_color_match")]
  3033. pref_overrides = [o for o in (filament_overrides or []) if not o.get("force_color_match")]
  3034. if required_types:
  3035. missing = self._get_missing_filament_types(printer_id, required_types, raw_data)
  3036. if missing:
  3037. # Same enrichment the live path applies: a bare "PLA" is not
  3038. # much help when what is missing is a particular PLA.
  3039. force_color_map = {
  3040. (o.get("type") or "").upper(): o.get("color_name") or o.get("color", "?") for o in force_overrides
  3041. }
  3042. return [
  3043. f"{t} ({force_color_map[t_upper]})" if (t_upper := t.upper()) in force_color_map else t
  3044. for t in missing
  3045. ]
  3046. if force_overrides:
  3047. missing_colors = self._get_missing_force_color_slots(printer_id, force_overrides, raw_data)
  3048. if missing_colors:
  3049. return missing_colors
  3050. # Preference overrides read as a preference but the matcher treats zero
  3051. # matches as a skip, so a printer with none of the wanted colours is
  3052. # rejected there too. Waking it would only produce that same rejection.
  3053. if pref_overrides and self._count_override_color_matches(printer_id, pref_overrides, raw_data) == 0:
  3054. return [f"{o.get('type', '?')} ({o.get('color_name') or o.get('color', '?')})" for o in pref_overrides]
  3055. return []
  3056. def _resolve_variant(self, item: PrintQueueItem, candidate: _ModelCandidate) -> None:
  3057. """Fold the winning candidate's file and settings onto the queue row (#671).
  3058. This is the whole trick that keeps cross-model items cheap: the many-to-many
  3059. never escapes the selection loop. By the time the pass commits, the row
  3060. looks exactly like an ordinary single-file model-based item, so the upload,
  3061. archive creation, expected-print registration, print history and reprint
  3062. paths need no knowledge that variants exist.
  3063. No-ops for a non-variant candidate, which is already the item's own columns.
  3064. Safe to run and re-run: the item's file columns are only ever *read* when it
  3065. has no variants, so an item that gets resolved and then skipped (library-row
  3066. conflict, previous-print gate) is simply resolved again on the next pass.
  3067. """
  3068. variant = candidate.variant
  3069. if variant is None:
  3070. return
  3071. item.library_file_id = variant.library_file_id
  3072. item.library_file = variant.library_file
  3073. # The dispatcher checks archive_id first and would print that instead of
  3074. # the file we just picked. Creation refuses to combine the two, so this
  3075. # only ever fires on a hand-written row — clear it rather than silently
  3076. # dispatch something the matcher never considered.
  3077. item.archive_id = None
  3078. item.archive = None
  3079. item.target_model = variant.target_model
  3080. item.plate_id = variant.plate_id
  3081. item.ams_mapping = variant.ams_mapping
  3082. item.nozzle_mapping = variant.nozzle_mapping
  3083. item.nozzle_rack_choice = variant.nozzle_rack_choice
  3084. item.filament_overrides = variant.filament_overrides
  3085. item.required_filament_types = variant.required_filament_types
  3086. if variant.print_time_seconds is not None:
  3087. # The row carried the shortest candidate's estimate so SJF could order
  3088. # it before a printer was known; now that one is chosen, record what is
  3089. # actually going to run so history and the ETA agree with reality.
  3090. item.print_time_seconds = variant.print_time_seconds
  3091. async def _ensure_ams_mapping(self, db: AsyncSession, printer_id: int, item: PrintQueueItem) -> str | None:
  3092. """Ensure the queue item carries a usable AMS mapping before dispatch.
  3093. Recomputes from live printer status when the stored mapping is missing OR
  3094. unresolved (all -1). A stored all-[-1] mapping is a bug artifact — a
  3095. frontend status-load race can serialize [-1] before the printer's AMS
  3096. trays are known (#2589) — and must not be trusted: downstream it would be
  3097. silently downgraded to external-spool mode and print against an empty
  3098. feed. A resolved mapping is re-checked against the target printer's
  3099. live trays before it is trusted (#2799) and recomputed when it does not
  3100. fit. When it does fit, its resolved slots are kept as they are and only
  3101. the slots the plate prints that are still unresolved are matched again;
  3102. all of it is left alone when the user has acknowledged it with "Print
  3103. Anyway".
  3104. When recompute cannot resolve it either (no compatible tray loaded), the
  3105. bogus [-1] is cleared to None so it is not later mistaken for an explicit
  3106. external selection. On a printer that reported loaded trays,
  3107. ``_block_on_unmatched_filament`` holds the item rather than let it go out
  3108. mapping-less; where it cannot judge, the print command keeps use_ams=True
  3109. and the firmware surfaces a clear AMS-mapping error instead of silently
  3110. printing to the empty external feed.
  3111. Returns an actionable message when that firmware error is the only
  3112. possible outcome — the matcher ran, matched nothing, and the printer has
  3113. no AMS to load a different spool into (#2771). The caller fails the item
  3114. on it instead of spending an upload on a print that cannot start.
  3115. Returns None everywhere else, including every case where we simply lack
  3116. the data to judge, so dispatch is only ever blocked on a positive
  3117. finding.
  3118. """
  3119. stored_mapping: list | None = None
  3120. if item.ams_mapping:
  3121. try:
  3122. stored_mapping = json.loads(item.ams_mapping)
  3123. except (json.JSONDecodeError, TypeError):
  3124. stored_mapping = None
  3125. # Present and resolved. Global tray IDs only mean something relative to
  3126. # the printer they were resolved against, so "resolved" is not the same
  3127. # as "resolved *here*" — check the mapping still fits the printer that
  3128. # is about to run this item before trusting it (#2799). "Print anyway"
  3129. # is the user overriding exactly this judgement, so it short-circuits.
  3130. if item.ams_mapping and not _mapping_is_all_unresolved(stored_mapping):
  3131. if item.skip_filament_check:
  3132. return None
  3133. conflict = await self._stored_mapping_conflict(db, printer_id, item, stored_mapping)
  3134. if conflict is None:
  3135. # It fits, but a slot the plate prints may still be empty: the
  3136. # spool was missing when it was mapped, and has perhaps been
  3137. # loaded since. Nothing else would ever look at that slot again,
  3138. # so a job held for it would be held again on every Start.
  3139. filled = await self._fill_unresolved_slots(db, printer_id, item, stored_mapping)
  3140. if filled is not None:
  3141. item.ams_mapping = json.dumps(filled)
  3142. logger.info(
  3143. "Queue item %s: filled unresolved slots of %s on printer %s: %s",
  3144. item.id,
  3145. stored_mapping,
  3146. printer_id,
  3147. filled,
  3148. )
  3149. await db.commit()
  3150. return None
  3151. logger.warning(
  3152. "Queue item %s: stored ams_mapping %s does not fit printer %s (%s) — recomputing",
  3153. item.id,
  3154. stored_mapping,
  3155. printer_id,
  3156. conflict,
  3157. )
  3158. # Drop it before recomputing so a failed recompute cannot fall back
  3159. # to the mapping we just rejected.
  3160. item.ams_mapping = None
  3161. stored_mapping = None
  3162. await db.commit()
  3163. computed_mapping = await self._compute_ams_mapping_for_printer(db, printer_id, item)
  3164. if computed_mapping and not _mapping_is_all_unresolved(computed_mapping):
  3165. item.ams_mapping = json.dumps(computed_mapping)
  3166. logger.info(
  3167. "Queue item %s: Computed AMS mapping for printer %s: %s",
  3168. item.id,
  3169. printer_id,
  3170. computed_mapping,
  3171. )
  3172. await db.commit()
  3173. return None
  3174. external_only = await self._external_spool_only_mapping(db, printer_id, item)
  3175. if external_only is not None:
  3176. item.ams_mapping = json.dumps(external_only)
  3177. logger.info(
  3178. "Queue item %s: printer %s has no AMS and its external spool has no filament set; "
  3179. "mapping every filament to it: %s",
  3180. item.id,
  3181. printer_id,
  3182. external_only,
  3183. )
  3184. await db.commit()
  3185. return None
  3186. if _mapping_is_all_unresolved(stored_mapping):
  3187. logger.warning(
  3188. "Queue item %s: stored ams_mapping %s is unresolved and could not be recomputed "
  3189. "from live status on printer %s; clearing it so dispatch does not treat it as external",
  3190. item.id,
  3191. stored_mapping,
  3192. printer_id,
  3193. )
  3194. item.ams_mapping = None
  3195. await db.commit()
  3196. return await self._unmappable_without_ams_message(db, printer_id, item, computed_mapping)
  3197. async def _external_spool_only_mapping(
  3198. self, db: AsyncSession, printer_id: int, item: PrintQueueItem
  3199. ) -> list[int] | None:
  3200. """Every filament on the external spool, for a printer that has nothing else (#3239).
  3201. A printer without an AMS prints from its external spool, and an external
  3202. spool whose filament was never set reports no type, so the matcher has
  3203. nothing to match. Sent without a mapping, the print goes out with the AMS
  3204. on and the firmware rejects it with 0700_8012. A stored ``[254]`` used
  3205. to carry such a job through; a job placed by model or location no longer
  3206. keeps one, and one queued that way never had one.
  3207. Only on a positive report: the printer has said it has no AMS, it has a
  3208. single external feed (a dual-nozzle printer's feeds steer nozzles, which
  3209. is not ours to pick), and no feed has a filament set — with one set, the
  3210. matcher has already given its answer. Not for a job that asked for its
  3211. colours to be matched strictly: a spool without a filament set has no
  3212. colour to check. Returns None otherwise.
  3213. """
  3214. if item.filament_overrides:
  3215. try:
  3216. overrides = json.loads(item.filament_overrides)
  3217. except (json.JSONDecodeError, TypeError):
  3218. return None
  3219. if not isinstance(overrides, list) or any(
  3220. isinstance(o, dict) and o.get("force_color_match") for o in overrides
  3221. ):
  3222. return None
  3223. status = printer_manager.get_status(printer_id)
  3224. if status is None or not isinstance(status.raw_data, dict):
  3225. return None
  3226. ams_units = status.raw_data.get("ams")
  3227. if not isinstance(ams_units, list) or ams_units:
  3228. return None
  3229. vt_trays = status.raw_data.get("vt_tray")
  3230. if not isinstance(vt_trays, list) or len(vt_trays) != 1 or not isinstance(vt_trays[0], dict):
  3231. return None
  3232. if self._build_loaded_filaments(status):
  3233. return None
  3234. printer = await self._get_printer(db, printer_id)
  3235. if _might_be_dual_nozzle(printer.model if printer else None, status):
  3236. return None
  3237. external = _int_or(vt_trays[0].get("id"), _EXTERNAL_TRAY_ID_MIN)
  3238. if external < _EXTERNAL_TRAY_ID_MIN:
  3239. return None
  3240. required = await self._get_filament_requirements(db, item)
  3241. slot_ids = [r.get("slot_id") for r in required or []]
  3242. slot_ids = [s for s in slot_ids if isinstance(s, int) and s > 0]
  3243. if not slot_ids:
  3244. return None
  3245. mapping = [-1] * max(slot_ids)
  3246. for slot_id in slot_ids:
  3247. mapping[slot_id - 1] = external
  3248. return mapping
  3249. async def _fill_unresolved_slots(
  3250. self,
  3251. db: AsyncSession,
  3252. printer_id: int,
  3253. item: PrintQueueItem,
  3254. stored_mapping: list | None,
  3255. ) -> list | None:
  3256. """``stored_mapping`` with its unresolved required slots matched on live trays (#2799).
  3257. Every resolved entry is kept, so a tray the user picked by hand stays
  3258. picked. Only the plate's unresolved slots are matched, and only against
  3259. trays the mapping does not already use: the matcher never gives two
  3260. slots one tray, and filling a gap with a tray the user assigned to
  3261. another slot would print that slot's filament twice.
  3262. Returns None when there is nothing to fill or nothing could be filled.
  3263. """
  3264. if not isinstance(stored_mapping, list) or not stored_mapping:
  3265. return None
  3266. required = await self._get_filament_requirements(db, item)
  3267. if not required:
  3268. return None
  3269. self._apply_filament_overrides(item, required)
  3270. gaps = {req["slot_id"] for req in _unresolved_required(required, stored_mapping)}
  3271. if not gaps:
  3272. return None
  3273. reserved = {t for t in stored_mapping if _is_tray_id(t) and t >= 0}
  3274. computed = await self._compute_ams_mapping_for_printer(
  3275. db, printer_id, item, only_slots=gaps, reserved_trays=reserved
  3276. )
  3277. if not computed:
  3278. return None
  3279. merged = list(stored_mapping) + [-1] * max(0, len(computed) - len(stored_mapping))
  3280. filled = False
  3281. for slot_id in gaps:
  3282. tray = computed[slot_id - 1] if slot_id <= len(computed) else None
  3283. if _is_tray_id(tray) and tray >= 0:
  3284. merged[slot_id - 1] = tray
  3285. filled = True
  3286. return merged if filled else None
  3287. async def missing_filament_for_start(self, db: AsyncSession, item: PrintQueueItem) -> list[str] | None:
  3288. """What a staged item would be held for if it were started now (#2799).
  3289. The Start button asks this before releasing an item, so that a job whose
  3290. filament is still not loaded offers "Print Anyway" instead of being
  3291. released, held again by the scheduler and leaving no way past the hold.
  3292. It reaches the same answer the dispatch path would: a stored mapping
  3293. that fits is kept and only its gaps are matched again, and one that
  3294. does not fit (or is missing) is replaced by a fresh match.
  3295. Returns the missing filaments, described the way the queue row
  3296. describes them, or None when nothing is missing or there is not enough
  3297. evidence to say: no printer yet (a model-based item picks one at
  3298. dispatch), no status, no trays reported, or no readable 3MF.
  3299. """
  3300. if item.skip_filament_check or not item.printer_id:
  3301. return None
  3302. status = printer_manager.get_status(item.printer_id)
  3303. if status is None or not self._build_loaded_filaments(status):
  3304. return None
  3305. # Called from a request, not from a pass: drop whatever the last pass
  3306. # memoised so this reads the file as it is now. A pass running at the
  3307. # same time only loses its cache.
  3308. self._filament_req_memo.clear()
  3309. mapping: list | None = None
  3310. if item.ams_mapping:
  3311. try:
  3312. mapping = json.loads(item.ams_mapping)
  3313. except (json.JSONDecodeError, TypeError):
  3314. mapping = None
  3315. if not isinstance(mapping, list) or _mapping_is_all_unresolved(mapping):
  3316. mapping = None
  3317. if mapping is not None:
  3318. if await self._stored_mapping_conflict(db, item.printer_id, item, mapping) is None:
  3319. mapping = await self._fill_unresolved_slots(db, item.printer_id, item, mapping) or mapping
  3320. else:
  3321. mapping = None
  3322. if mapping is None:
  3323. mapping = await self._compute_ams_mapping_for_printer(db, item.printer_id, item) or []
  3324. required = await self._get_filament_requirements(db, item)
  3325. if not required:
  3326. return None
  3327. self._apply_filament_overrides(item, required)
  3328. missing = _unresolved_required(required, mapping)
  3329. return [_describe_filament(req, "nozzle_id") for req in missing] or None
  3330. async def _stored_mapping_conflict(
  3331. self,
  3332. db: AsyncSession,
  3333. printer_id: int,
  3334. item: PrintQueueItem,
  3335. stored_mapping: list | None,
  3336. ) -> str | None:
  3337. """Describe why ``stored_mapping`` cannot be trusted on ``printer_id`` (#2799).
  3338. A mapping is a list of global tray IDs, and those are only meaningful
  3339. relative to the printer they were resolved against — the same rule
  3340. ``print_queue`` documents for tray identity. Two ways a stored mapping
  3341. arrives at a printer it was not resolved for:
  3342. * the print dialog stamps one mapping onto every selected printer, so a
  3343. mapping computed against the first printer is dispatched verbatim to
  3344. the rest, whose AMS slot order differs;
  3345. * a spool is moved between queueing and dispatch.
  3346. Both end the same way: the slot index still resolves, so nothing looks
  3347. wrong, and the printer obeys it — an explicit ``ams_mapping`` bypasses
  3348. the firmware's own type check, so a PETG slot happily prints in ASA.
  3349. Returns a short reason when the mapping names a tray this printer does
  3350. not have loaded, points a slot at a tray holding a different filament
  3351. type, or points it at an external spool this printer reports empty while
  3352. its AMS holds that slot's filament (#3239). Returns None when the
  3353. mapping fits, and — deliberately — whenever we lack the evidence to
  3354. judge, so a recompute only ever follows a positive finding.
  3355. An unresolved (``-1``) required slot is NOT a conflict: it says the
  3356. matcher had nothing, not that the mapping belongs to another printer,
  3357. and destroying a partially hand-resolved mapping over it would lose the
  3358. slots the user did resolve. ``_fill_unresolved_slots`` matches those
  3359. slots again on their own, and ``_block_on_unmatched_filament`` holds the
  3360. item when that finds nothing.
  3361. """
  3362. if not isinstance(stored_mapping, list) or not stored_mapping:
  3363. return None
  3364. status = printer_manager.get_status(printer_id)
  3365. if status is None:
  3366. return None
  3367. loaded = self._build_loaded_filaments(status)
  3368. if not loaded:
  3369. # Nothing reported yet (reconnect, first push pending). Saying
  3370. # "tray not loaded" here would recompute against an AMS we cannot
  3371. # see, which is how #2589 produced a bogus all-[-1] in the first
  3372. # place.
  3373. return None
  3374. by_tray = {f["global_tray_id"]: f for f in loaded}
  3375. # Cheap pass first, on live status alone: every tray the mapping names
  3376. # has to exist here. This catches a foreign mapping without opening the
  3377. # 3MF, which is worth doing because the parse below is neither cached
  3378. # nor free.
  3379. for index, tray in enumerate(stored_mapping):
  3380. if not _is_tray_id(tray) or tray < 0:
  3381. continue
  3382. if tray in by_tray:
  3383. continue
  3384. if tray >= 254:
  3385. # An external feed we have not heard about is absence of
  3386. # evidence about *that slot* — the rest of the mapping is still
  3387. # worth judging, so skip it rather than abandoning the pass.
  3388. continue
  3389. return f"slot {index + 1} points at tray {tray}, which this printer does not have loaded"
  3390. # Only now is the 3MF worth opening: the type check needs to know what
  3391. # each slot actually asked for. The external spool is checked the same
  3392. # way as an AMS tray — `_build_loaded_filaments` reports its type, and
  3393. # two printers with different filament in the external feed is the same
  3394. # failure this method exists to catch.
  3395. required = await self._get_filament_requirements(db, item)
  3396. if not required:
  3397. return None
  3398. self._apply_filament_overrides(item, required)
  3399. # External feeds the printer reports and reports as empty. That is
  3400. # evidence, unlike an external feed it says nothing about (#3239).
  3401. empty_external: set[int] = set()
  3402. vt_trays = status.raw_data.get("vt_tray") if isinstance(status.raw_data, dict) else None
  3403. for vt in vt_trays if isinstance(vt_trays, list) else []:
  3404. if isinstance(vt, dict) and not vt.get("tray_type"):
  3405. try:
  3406. empty_external.add(int(vt.get("id", 254)))
  3407. except (TypeError, ValueError):
  3408. continue
  3409. for req in required:
  3410. slot_id = req.get("slot_id") or 0
  3411. if slot_id <= 0:
  3412. continue
  3413. if slot_id > len(stored_mapping):
  3414. return f"slot {slot_id} is not covered by the mapping"
  3415. tray = stored_mapping[slot_id - 1]
  3416. if not _is_tray_id(tray) or tray < 0:
  3417. continue
  3418. loaded_tray = by_tray.get(tray)
  3419. if loaded_tray is None:
  3420. # A mapping made for a printer that feeds this slot from its
  3421. # external spool, sent to one whose external spool is empty and
  3422. # whose AMS holds the filament (#3239). Only then: an external
  3423. # spool can be loaded without its filament set, and on a printer
  3424. # with nothing else to offer that job printed before. The colour
  3425. # has to match too: with only another colour in the AMS, the
  3426. # printer asking for the spool beats printing in that colour.
  3427. want = canonical_filament_type(req.get("type"))
  3428. if tray >= 254 and tray in empty_external and want:
  3429. ams_tray = next(
  3430. (
  3431. f
  3432. for f in loaded
  3433. if not f.get("is_external")
  3434. and canonical_filament_type(f.get("type")) == want
  3435. and self._colors_are_similar(f.get("color"), req.get("color"))
  3436. ),
  3437. None,
  3438. )
  3439. if ams_tray is not None:
  3440. return (
  3441. f"slot {slot_id} points at the external spool, which is empty, "
  3442. f"while tray {ams_tray['global_tray_id']} holds {ams_tray.get('type')}"
  3443. )
  3444. continue
  3445. want = canonical_filament_type(req.get("type"))
  3446. have = canonical_filament_type(loaded_tray.get("type"))
  3447. if want and have and want != have:
  3448. return f"slot {slot_id} needs {req.get('type')} but tray {tray} holds {loaded_tray.get('type')}"
  3449. return None
  3450. async def _unmappable_without_ams_message(
  3451. self,
  3452. db: AsyncSession,
  3453. printer_id: int,
  3454. item: PrintQueueItem,
  3455. computed_mapping: list[int] | None,
  3456. ) -> str | None:
  3457. """Message for a mapping that resolved nothing on an AMS-less printer (#2771).
  3458. A print dispatched with no mapping goes out as ``use_ams: true`` with no
  3459. ``ams_mapping`` and no ``ams_mapping2``, which the firmware rejects with
  3460. 0700_8012 "Failed to get AMS mapping table" — after Bambuddy has already
  3461. uploaded several megabytes and burned its dispatch retries. This method
  3462. speaks only for the AMS-less case: there is nothing to resume into — the
  3463. external spool holder is the whole inventory — so the useful answer is to
  3464. say which filament is missing and stop. With an AMS attached it returns
  3465. None: where live status reported loaded trays,
  3466. ``_block_on_unmatched_filament`` holds the item instead, and where it
  3467. reported none the print goes out and the firmware error is the answer —
  3468. the user can load the right spool and press Resume.
  3469. Fail-safe by construction, mirroring the nozzle-diameter guard (#1899):
  3470. every branch that lacks the evidence to be sure returns None.
  3471. """
  3472. # None means the matcher never ran (no requirements parsed from the 3MF,
  3473. # or nothing loaded at all) rather than "ran and matched nothing". Those
  3474. # dispatch as they always have.
  3475. if not _mapping_is_all_unresolved(computed_mapping):
  3476. return None
  3477. status = printer_manager.get_status(printer_id)
  3478. if status is None:
  3479. return None
  3480. # "No AMS" has to be a fact the printer stated, not the absence of a
  3481. # statement. `raw_data["ams"]` is written only once an AMS push has been
  3482. # handled and is preserved across partial pushes thereafter, so a missing
  3483. # key means we have not heard yet — most likely a reconnect, where the
  3484. # trays of a fully loaded AMS would be invisible for a few seconds. An
  3485. # empty list is the positive report of a printer with no AMS.
  3486. ams_units = status.raw_data.get("ams")
  3487. if not isinstance(ams_units, list) or ams_units:
  3488. return None
  3489. required = await self._get_filament_requirements(db, item)
  3490. loaded = self._build_loaded_filaments(status)
  3491. if not required or not loaded:
  3492. # Both were non-empty moments ago or the matcher could not have run.
  3493. # If the picture changed under us, say nothing rather than fail an
  3494. # item on stale evidence.
  3495. return None
  3496. self._apply_filament_overrides(item, required)
  3497. return _unmatched_filament_message(required, loaded)
  3498. async def _fail_unmappable_item(
  3499. self, db: AsyncSession, item: PrintQueueItem, printer_id: int, message: str
  3500. ) -> None:
  3501. """Fail a queue item whose filament mapping cannot resolve (#2771).
  3502. This replaces a failure, not a success: without it the item is uploaded,
  3503. rejected by the firmware with 0700_8012, retried twice more and failed
  3504. anyway with "never started the print after N dispatch attempts". So this
  3505. applies on the model-based path too, even though it means an "Any <model>"
  3506. job stops at the first printer offered rather than trying its siblings —
  3507. deferring instead would need the check to move inside
  3508. ``_find_printer_for_model``'s candidate loop, since un-assigning here just
  3509. re-assigns the same printer on the next tick.
  3510. """
  3511. item.status = "failed"
  3512. item.error_message = message
  3513. item.completed_at = datetime.now(timezone.utc)
  3514. item.waiting_reason = None
  3515. await db.commit()
  3516. logger.warning(
  3517. "Queue item %s: no usable AMS mapping on printer %s — %s",
  3518. item.id,
  3519. printer_id,
  3520. message,
  3521. )
  3522. job_name = await self._get_job_name(db, item)
  3523. printer = await self._get_printer(db, printer_id)
  3524. await notification_service.on_queue_job_failed(
  3525. job_name=job_name,
  3526. printer_id=printer_id,
  3527. printer_name=printer.name if printer else "Unknown",
  3528. reason=message,
  3529. db=db,
  3530. )
  3531. try:
  3532. await ws_manager.send_queue_item_failed(
  3533. user_id=item.created_by_id,
  3534. queue_item_id=item.id,
  3535. printer_id=printer_id,
  3536. reason="filament_unmappable",
  3537. )
  3538. except Exception:
  3539. pass
  3540. async def _compute_ams_mapping_for_printer(
  3541. self,
  3542. db: AsyncSession,
  3543. printer_id: int,
  3544. item: PrintQueueItem,
  3545. *,
  3546. only_slots: set[int] | None = None,
  3547. reserved_trays: set[int] | None = None,
  3548. ) -> list[int] | None:
  3549. """Compute AMS mapping for a printer based on filament requirements.
  3550. Called when a queue item has no ams_mapping set — either for model-based
  3551. items after printer assignment, or printer-specific items (e.g. from VP).
  3552. Args:
  3553. db: Database session
  3554. printer_id: The assigned printer ID
  3555. item: The queue item (contains archive_id or library_file_id)
  3556. only_slots: Match only these filament slots; every other slot comes
  3557. back unresolved (-1). Used to fill the gaps of a stored mapping.
  3558. reserved_trays: Global tray IDs to leave out of the match, because
  3559. the stored mapping already gives them to another slot.
  3560. Returns:
  3561. AMS mapping array or None if no mapping needed/possible
  3562. """
  3563. # Get printer status
  3564. status = printer_manager.get_status(printer_id)
  3565. if not status:
  3566. logger.warning("Cannot compute AMS mapping: printer %s status unavailable", printer_id)
  3567. return None
  3568. # Filament Track Switch (FTS): when installed it routes any AMS slot to
  3569. # either extruder, so the per-nozzle hard filter below must NOT apply.
  3570. # Otherwise a print on one nozzle can't use a spool physically loaded in
  3571. # an AMS on the *other* nozzle, and the matcher falls through to a
  3572. # same-type wrong-colour spool on the target nozzle — the H2C + FTS
  3573. # wrong-filament bug (#2186). Mirrors the frontend skip added for #1162.
  3574. fts_installed = bool(getattr(getattr(status, "fila_switch", None), "installed", False))
  3575. # Get filament requirements from source file
  3576. filament_reqs = await self._get_filament_requirements(db, item)
  3577. if not filament_reqs:
  3578. # When the 3MF can't be read but force-color overrides are present, build a
  3579. # direct mapping from the overrides so the printer uses the correct AMS slot.
  3580. if item.filament_overrides:
  3581. try:
  3582. overrides = json.loads(item.filament_overrides)
  3583. force_overrides = [o for o in overrides if o.get("force_color_match")]
  3584. if force_overrides:
  3585. logger.info(
  3586. "Queue item %s: No filament reqs from 3MF; building AMS mapping from %d "
  3587. "force-color override(s)",
  3588. item.id,
  3589. len(force_overrides),
  3590. )
  3591. return self._build_override_direct_mapping(force_overrides, status)
  3592. except (json.JSONDecodeError, KeyError, TypeError) as e:
  3593. logger.warning("Queue item %s: Force-color fallback mapping failed: %s", item.id, e)
  3594. logger.debug("No filament requirements found for queue item %s", item.id)
  3595. return None
  3596. self._apply_filament_overrides(item, filament_reqs)
  3597. if only_slots is not None:
  3598. filament_reqs = [req for req in filament_reqs if req.get("slot_id") in only_slots]
  3599. if not filament_reqs:
  3600. return None
  3601. # Build loaded filaments from printer status
  3602. loaded_filaments = self._build_loaded_filaments(status)
  3603. if reserved_trays:
  3604. loaded_filaments = [f for f in loaded_filaments if f["global_tray_id"] not in reserved_trays]
  3605. if not loaded_filaments:
  3606. logger.debug("No filaments loaded on printer %s", printer_id)
  3607. return None
  3608. # Check if user prefers lowest remaining filament when multiple spools match
  3609. prefer_lowest = await self._get_bool_setting(db, "prefer_lowest_filament")
  3610. # Gate prefer_lowest on the printer's AMS Filament Backup state (#1766).
  3611. # Without backup, the printer will not switch to a second spool when the
  3612. # picked one runs out — so sorting toward the lowest leaves the print
  3613. # at risk of running dry mid-job. None (unknown / A1 family) preserves
  3614. # today's behaviour intentionally.
  3615. if prefer_lowest and status.ams_filament_backup is False:
  3616. logger.info("[prefer-lowest] skipped (AMS Backup OFF on printer %s)", printer_id)
  3617. prefer_lowest = False
  3618. # When the preference is on, surface Bambuddy's inventory-side
  3619. # remaining for each slot that's bound to a tracked spool, so the
  3620. # sort beats the MQTT-only blind spot (#1508). Skip the lookup
  3621. # entirely when the preference is off — no behaviour change for
  3622. # users who haven't opted in.
  3623. inventory_remain_overrides: dict[int, float] | None = None
  3624. if prefer_lowest:
  3625. inventory_remain_overrides = await self._build_inventory_remain_overrides(db, printer_id, loaded_filaments)
  3626. # Compute mapping: match required filaments to available slots
  3627. return self._match_filaments_to_slots(
  3628. filament_reqs, loaded_filaments, prefer_lowest, inventory_remain_overrides, fts_installed
  3629. )
  3630. def _apply_filament_overrides(self, item: PrintQueueItem, filament_reqs: list[dict]) -> None:
  3631. """Rewrite ``filament_reqs`` in place with the item's per-slot overrides.
  3632. Extracted from ``_compute_ams_mapping_for_printer`` so the unmappable
  3633. diagnosis (#2771) describes the filament the matcher actually looked
  3634. for, not the one the 3MF was sliced with — naming the pre-override
  3635. filament in a user-facing error would send the user to load the wrong
  3636. spool.
  3637. """
  3638. if not item.filament_overrides:
  3639. return
  3640. try:
  3641. overrides = json.loads(item.filament_overrides)
  3642. override_map = {o["slot_id"]: o for o in overrides}
  3643. for req in filament_reqs:
  3644. if req["slot_id"] in override_map:
  3645. override = override_map[req["slot_id"]]
  3646. req["type"] = override["type"]
  3647. req["color"] = override["color"]
  3648. # A manual/preference override SWAPS the slot's filament, so the
  3649. # 3MF's original tray_info_idx now points at the old spool and must
  3650. # be cleared — matching then falls back to type+colour. A
  3651. # force_color_match override is not a swap: it carries the 3MF's
  3652. # intended variant (Basic GFA00 / Matte GFA01 / Silk GFA06), so keep
  3653. # it here too, letting the matcher pin the correct variant slot on a
  3654. # printer holding two same-colour spools of different variants (#2650).
  3655. # If that variant isn't loaded the matcher falls back to type+colour,
  3656. # so an eligible printer never fails to map.
  3657. req["tray_info_idx"] = (
  3658. override.get("tray_info_idx", "") if override.get("force_color_match") else ""
  3659. )
  3660. logger.debug(
  3661. "Queue item %s: Override slot %d -> %s %s",
  3662. item.id,
  3663. req["slot_id"],
  3664. override["type"],
  3665. override["color"],
  3666. )
  3667. except (json.JSONDecodeError, KeyError, TypeError) as e:
  3668. logger.warning("Failed to apply filament overrides for queue item %s: %s", item.id, e)
  3669. def _build_override_direct_mapping(self, force_overrides: list[dict], status) -> list[int] | None:
  3670. """Build an AMS mapping directly from force-color overrides without a 3MF.
  3671. Used when ``_get_filament_requirements`` returns nothing (e.g. the 3MF's
  3672. slice_info is missing or unreadable) but ``force_color_match`` overrides
  3673. are present. Each override's ``slot_id``, ``type``, and ``color`` are
  3674. treated as the filament requirement for that slot and matched against the
  3675. current AMS state of the printer.
  3676. Returns the same format as ``_match_filaments_to_slots``, or None when
  3677. the AMS has no loaded filaments.
  3678. """
  3679. loaded = self._build_loaded_filaments(status)
  3680. if not loaded:
  3681. return None
  3682. reqs = [
  3683. {
  3684. "slot_id": o["slot_id"],
  3685. "type": o.get("type", ""),
  3686. "color": o.get("color", ""),
  3687. # These are all force_color_match overrides, so the idx (when the
  3688. # 3MF carried one) is the intended variant, not a stale swap —
  3689. # keep it so the matcher pins the right variant slot, falling back
  3690. # to type+colour when it isn't loaded (#2650).
  3691. "tray_info_idx": o.get("tray_info_idx", ""),
  3692. }
  3693. for o in force_overrides
  3694. ]
  3695. return self._match_filaments_to_slots(reqs, loaded)
  3696. async def _get_filament_requirements(self, db: AsyncSession, item: PrintQueueItem) -> list[dict] | None:
  3697. """Resolve the queue item's source 3MF and parse the per-slot
  3698. filament requirements out of it. Thin DB-resolver wrapper around
  3699. ``filament_requirements.extract_filament_requirements`` so the VP
  3700. queue-mode write path (#1188) can reuse the same parser at upload
  3701. time.
  3702. """
  3703. from backend.app.services.filament_requirements import extract_filament_requirements
  3704. # Callers rewrite these dicts in place via `_apply_filament_overrides`,
  3705. # so every caller gets its own copy — a shared list would leak one
  3706. # item's overrides into the next item that happens to print the same
  3707. # plate.
  3708. memo_key = (item.archive_id, item.library_file_id, item.plate_id)
  3709. if memo_key in self._filament_req_memo:
  3710. cached = self._filament_req_memo[memo_key]
  3711. return [dict(r) for r in cached] if cached else None
  3712. file_path: Path | None = None
  3713. if item.archive_id:
  3714. result = await db.execute(select(PrintArchive).where(PrintArchive.id == item.archive_id))
  3715. archive = result.scalar_one_or_none()
  3716. if archive:
  3717. file_path = settings.base_dir / archive.file_path
  3718. elif item.library_file_id:
  3719. result = await db.execute(LibraryFile.active().where(LibraryFile.id == item.library_file_id))
  3720. library_file = result.scalar_one_or_none()
  3721. if library_file:
  3722. lib_path = Path(library_file.file_path)
  3723. file_path = lib_path if lib_path.is_absolute() else settings.base_dir / library_file.file_path
  3724. if not file_path or not file_path.exists():
  3725. self._filament_req_memo[memo_key] = None
  3726. return None
  3727. filaments = extract_filament_requirements(file_path, plate_id=item.plate_id)
  3728. self._filament_req_memo[memo_key] = filaments or None
  3729. return [dict(r) for r in filaments] if filaments else None
  3730. def _build_loaded_filaments(self, status) -> list[dict]:
  3731. """Build list of loaded filaments from printer status.
  3732. Args:
  3733. status: PrinterState from printer_manager
  3734. Returns:
  3735. List of loaded filament dicts with type, color, ams_id, tray_id, global_tray_id
  3736. """
  3737. filaments = []
  3738. # Get ams_extruder_map for dual-nozzle printers (H2D, H2D Pro)
  3739. ams_extruder_map = status.raw_data.get("ams_extruder_map", {})
  3740. # Dual-nozzle detection, used below to route external spools to an
  3741. # extruder (#2771). Mirrors `buildLoadedFilaments` in the frontend,
  3742. # which was corrected for #1257 while this copy kept the old signal.
  3743. #
  3744. # `ams_extruder_map` is derived from AMS info bits, so a dual-nozzle
  3745. # printer with zero AMS units reports an empty map — and every external
  3746. # spool then got `extruder_id=None`, which the nozzle-aware filter in
  3747. # `_match_filaments_to_slots` rejects outright because `None` equals
  3748. # neither 0 nor 1. On an X2D feeding from external spools only that left
  3749. # nothing to match, the mapping came back all -1, and the print went out
  3750. # with `use_ams: true` and no mapping table at all — firmware 0700_8012,
  3751. # "Failed to get AMS mapping table".
  3752. #
  3753. # `nozzles` is always a two-entry list (the state seeds it with two empty
  3754. # NozzleInfo stubs), so its length proves nothing; only a populated
  3755. # diameter on the second entry means real hardware. The other two signals
  3756. # are fallbacks for firmware revisions that surface one but not the
  3757. # other: a populated `ams_extruder_map` is dual-nozzle by construction,
  3758. # and so is more than one `vt_tray` entry, since single-nozzle printers
  3759. # expose exactly one external feed.
  3760. nozzles = getattr(status, "nozzles", None) or []
  3761. vt_trays = status.raw_data.get("vt_tray") or []
  3762. is_dual_nozzle = bool(
  3763. (len(nozzles) > 1 and getattr(nozzles[1], "nozzle_diameter", ""))
  3764. or ams_extruder_map
  3765. # isinstance, because a dict here would count its ~30 keys as trays.
  3766. # bambu_mqtt normalises vt_tray to a list before it reaches raw_data,
  3767. # so this is unreachable — but the loop below would raise on a dict
  3768. # and that is the pre-existing behaviour to keep, not to paper over.
  3769. or (isinstance(vt_trays, list) and len(vt_trays) > 1)
  3770. )
  3771. # Parse AMS units from raw_data
  3772. ams_data = status.raw_data.get("ams", [])
  3773. for ams_unit in ams_data:
  3774. ams_id = int(ams_unit.get("id", 0))
  3775. trays = ams_unit.get("tray", [])
  3776. is_ht = len(trays) == 1 # AMS-HT has single tray
  3777. for tray in trays:
  3778. tray_type = tray.get("tray_type")
  3779. if tray_type:
  3780. tray_id = int(tray.get("id", 0))
  3781. tray_color = tray.get("tray_color", "")
  3782. # tray_info_idx identifies the specific spool (e.g., "GFA00", "P4d64437")
  3783. tray_info_idx = tray.get("tray_info_idx", "")
  3784. # Normalize color: remove alpha, add hash
  3785. color = self._normalize_color(tray_color)
  3786. # Calculate global tray ID
  3787. # AMS-HT units have IDs starting at 128 with a single tray
  3788. global_tray_id = ams_id if ams_id >= 128 else ams_id * 4 + tray_id
  3789. filaments.append(
  3790. {
  3791. "type": tray_type,
  3792. "color": color,
  3793. "tray_info_idx": tray_info_idx,
  3794. "ams_id": ams_id,
  3795. "tray_id": tray_id,
  3796. "is_ht": is_ht,
  3797. "is_external": False,
  3798. "global_tray_id": global_tray_id,
  3799. "extruder_id": ams_extruder_map.get(str(ams_id)),
  3800. "remain": tray.get("remain", -1),
  3801. }
  3802. )
  3803. # Check external spool(s) (vt_tray is a list)
  3804. for idx, vt in enumerate(vt_trays):
  3805. if vt.get("tray_type"):
  3806. color = self._normalize_color(vt.get("tray_color", ""))
  3807. tray_id = int(vt.get("id", 254))
  3808. filaments.append(
  3809. {
  3810. "type": vt["tray_type"],
  3811. "color": color,
  3812. "tray_info_idx": vt.get("tray_info_idx", ""),
  3813. "ams_id": -1,
  3814. "tray_id": idx,
  3815. "is_ht": False,
  3816. "is_external": True,
  3817. "global_tray_id": tray_id,
  3818. # 254 = VIRTUAL_TRAY_DEPUTY_ID feeds extruder 1 (left),
  3819. # 255 = VIRTUAL_TRAY_MAIN_ID feeds extruder 0 (right).
  3820. "extruder_id": (255 - tray_id) if is_dual_nozzle else None,
  3821. "remain": vt.get("remain", -1),
  3822. }
  3823. )
  3824. return filaments
  3825. def _normalize_color(self, color: str | None) -> str:
  3826. """Normalize color to #RRGGBB format."""
  3827. if not color:
  3828. return "#808080"
  3829. hex_color = color.replace("#", "")[:6]
  3830. return f"#{hex_color}"
  3831. def _normalize_color_for_compare(self, color: str | None) -> str:
  3832. """Normalize color for comparison (lowercase, no hash)."""
  3833. if not color:
  3834. return ""
  3835. return color.replace("#", "").lower()[:6]
  3836. def _color_distance(self, color1: str | None, color2: str | None) -> float | None:
  3837. """Perceptual (CIEDE2000) distance, or None when either colour is unusable.
  3838. Ranks the candidates ``_colors_are_similar`` admits (#2804). Eligibility
  3839. stays the per-channel box that shipped — this only decides which of
  3840. several eligible spools is closest, so nothing becomes usable or
  3841. unusable because of it.
  3842. It ranks by how far apart the colours *look*, not how far apart their
  3843. numbers are. RGB distance overweights blue badly enough to invert the
  3844. answer: against a required ``#1E4821`` green, a purple ``#38202F`` is
  3845. the nearer of two eligible spools by RGB and the further by a factor of
  3846. four once measured perceptually.
  3847. Alpha is ignored, deliberately: the alpha a slicer writes for a
  3848. transparent filament is not a colour the user chose, and counting it
  3849. would stop a transparent filament matching itself.
  3850. """
  3851. return perceptual_color_distance(color1, color2)
  3852. def _colors_are_similar(self, color1: str | None, color2: str | None, threshold: int = 40) -> bool:
  3853. """Check if two colors are visually similar within a threshold."""
  3854. hex1 = self._normalize_color_for_compare(color1)
  3855. hex2 = self._normalize_color_for_compare(color2)
  3856. if not hex1 or not hex2 or len(hex1) < 6 or len(hex2) < 6:
  3857. return False
  3858. try:
  3859. r1 = int(hex1[0:2], 16)
  3860. g1 = int(hex1[2:4], 16)
  3861. b1 = int(hex1[4:6], 16)
  3862. r2 = int(hex2[0:2], 16)
  3863. g2 = int(hex2[2:4], 16)
  3864. b2 = int(hex2[4:6], 16)
  3865. return abs(r1 - r2) <= threshold and abs(g1 - g2) <= threshold and abs(b1 - b2) <= threshold
  3866. except ValueError:
  3867. return False
  3868. async def _build_inventory_remain_overrides(
  3869. self, db: AsyncSession, printer_id: int, loaded: list[dict]
  3870. ) -> dict[int, float]:
  3871. """Return ``{global_tray_id: remaining_grams}`` for AMS slots the user
  3872. has bound to an inventory spool — Bambuddy-side or Spoolman-side.
  3873. The MQTT ``remain`` field on a tray is the printer firmware's
  3874. RFID-decremented value, which has two limitations the "Prefer Lowest
  3875. Remaining Filament" feature has been ignoring (#1508):
  3876. - it's only meaningful for Bambu RFID spools; everything else reports
  3877. ``-1`` (then clamped to a sentinel), so multiple non-RFID trays
  3878. compare equal and the sort collapses to AMS-slot order — the user
  3879. who's curating inventory weights gets the lower-slot pick instead
  3880. of the lower-remaining pick;
  3881. - even when set, it's the *printer's* counter, not Bambuddy's
  3882. ``label_weight - weight_used`` (internal mode) or Spoolman's
  3883. ``remaining_weight`` (Spoolman mode) — the two diverge any time the
  3884. user re-spools, swaps cardboard, or runs a print outside Bambuddy.
  3885. When the user has bound a spool to a slot, their own inventory
  3886. tracking is authoritative; this helper surfaces that value so the
  3887. sort can prefer it. Slots without a binding are absent from the
  3888. returned map — the caller then falls back to MQTT ``remain`` for
  3889. those, preserving the pre-#1508 behaviour for un-tracked spools.
  3890. Returns an empty map on any failure (no inventory bindings, DB
  3891. error, Spoolman unreachable). A best-effort lookup; "Prefer Lowest"
  3892. is a preference, not a guarantee.
  3893. """
  3894. if not loaded:
  3895. return {}
  3896. # External / virtual-tray slots are tracked separately from AMS — skip
  3897. # them so a VT-loaded spool doesn't accidentally inherit a tracked
  3898. # AMS binding (the tables use ams_id 254/255 for VT, but the cross
  3899. # match is fiddly and out of scope for this fix).
  3900. tracked_slots = [(f["ams_id"], f["tray_id"], f["global_tray_id"]) for f in loaded if not f.get("is_external")]
  3901. if not tracked_slots:
  3902. return {}
  3903. is_spoolman = await self._is_spoolman_mode(db)
  3904. overrides: dict[int, float] = {}
  3905. if is_spoolman:
  3906. result = await db.execute(
  3907. select(SpoolmanSlotAssignment).where(SpoolmanSlotAssignment.printer_id == printer_id)
  3908. )
  3909. assignments = list(result.scalars().all())
  3910. by_slot = {(a.ams_id, a.tray_id): a.spoolman_spool_id for a in assignments}
  3911. from backend.app.services.filament_deficit import _spoolman_remaining_grams
  3912. for ams_id, tray_id, gtid in tracked_slots:
  3913. spoolman_id = by_slot.get((ams_id, tray_id))
  3914. if spoolman_id is None:
  3915. continue
  3916. grams = await _spoolman_remaining_grams(spoolman_id)
  3917. if grams is not None:
  3918. overrides[gtid] = grams
  3919. return overrides
  3920. # Internal inventory mode (default). selectinload matches the pattern
  3921. # used elsewhere (inventory.py, spoolman.py routes) — a single query
  3922. # plus an eager-loaded relationship rather than an explicit join, so
  3923. # the row-attribute shape is exactly what those routes already rely on.
  3924. result = await db.execute(
  3925. select(SpoolAssignment)
  3926. .options(selectinload(SpoolAssignment.spool))
  3927. .where(SpoolAssignment.printer_id == printer_id)
  3928. )
  3929. assignments = list(result.scalars().all())
  3930. by_slot = {(a.ams_id, a.tray_id): a.spool for a in assignments}
  3931. for ams_id, tray_id, gtid in tracked_slots:
  3932. spool = by_slot.get((ams_id, tray_id))
  3933. if spool is None:
  3934. continue
  3935. label = float(spool.label_weight or 0)
  3936. used = float(spool.weight_used or 0)
  3937. overrides[gtid] = max(0.0, label - used)
  3938. return overrides
  3939. @staticmethod
  3940. async def _is_spoolman_mode(db: AsyncSession) -> bool:
  3941. """Mirror of ``filament_deficit._is_spoolman_mode`` — kept private
  3942. here to avoid making this module import-dependent on that private
  3943. helper's signature."""
  3944. try:
  3945. from backend.app.api.routes.settings import get_setting
  3946. v = await get_setting(db, "spoolman_enabled")
  3947. return bool(v) and v.lower() == "true"
  3948. except Exception:
  3949. return False
  3950. @staticmethod
  3951. def _slot_priority(ams_id: int | None, tray_id: int | None) -> int:
  3952. """Deterministic slot-position tie-breaker for the prefer-lowest sort.
  3953. Three bands, matched to the emission order in ``_build_loaded_filaments``
  3954. so a tied sort produces the same physical-position order the pre-#1508
  3955. stable sort did (preserves the regression-free baseline):
  3956. - Regular AMS (``ams_id`` 0..7): ``ams_id * 4 + tray_id`` → 0..31
  3957. - AMS-HT (``ams_id`` >= 128, single tray): ``1000 + (ams_id - 128) * 4``
  3958. - External / VT (``ams_id`` < 0, or ``None``): ``10_000``
  3959. Banding ensures regular AMS < AMS-HT < external on ties, regardless of
  3960. what the raw ``ams_id`` happens to be (in particular, ``ams_id = -1``
  3961. for VT must NOT sort to a negative number or it would beat AMS slot 0).
  3962. """
  3963. if ams_id is None or ams_id < 0:
  3964. return 10_000
  3965. if ams_id >= 128:
  3966. return 1_000 + (ams_id - 128) * 4 + (tray_id or 0)
  3967. return ams_id * 4 + (tray_id or 0)
  3968. @staticmethod
  3969. def _prefer_lowest_sort_key(f: dict, overrides: dict[int, float] | None) -> tuple[int, float, int]:
  3970. """Sort key for the "Prefer Lowest Remaining Filament" preference.
  3971. Two-tier ordering: inventory-tracked spools always sort BEFORE
  3972. non-tracked spools (the user has told us they care about these
  3973. specifically), then ascending by remaining within each tier, then
  3974. ascending by AMS slot position as the deterministic tie-breaker.
  3975. Tiers are flagged by the first tuple element (0 = inventory-tracked,
  3976. 1 = MQTT-only / unknown). Cross-tier value comparisons never run
  3977. because the tier flag dominates — which is what lets us mix grams
  3978. (inventory) and percent (MQTT) without a unit conversion.
  3979. Within the MQTT tier ``remain = -1`` (unknown) is mapped to 101 so
  3980. spools the printer DOES know something about sort ahead of those
  3981. it knows nothing about — preserves pre-#1508 behaviour for the
  3982. no-inventory-binding case.
  3983. Slot tie-breaker via ``_slot_priority`` so regular AMS < AMS-HT <
  3984. external on ties, matching the legacy emission-order stable sort.
  3985. """
  3986. gtid = f.get("global_tray_id")
  3987. slot_order = PrintScheduler._slot_priority(f.get("ams_id"), f.get("tray_id"))
  3988. if overrides and gtid in overrides:
  3989. return (0, overrides[gtid], slot_order)
  3990. remain = f.get("remain", -1)
  3991. return (1, float(remain) if remain is not None and remain >= 0 else 101.0, slot_order)
  3992. def _match_filaments_to_slots(
  3993. self,
  3994. required: list[dict],
  3995. loaded: list[dict],
  3996. prefer_lowest: bool = False,
  3997. inventory_remain_overrides: dict[int, float] | None = None,
  3998. fts_installed: bool = False,
  3999. ) -> list[int] | None:
  4000. """Match required filaments to loaded filaments and build AMS mapping.
  4001. Priority: unique tray_info_idx match > exact color match > similar color match > type-only match
  4002. The tray_info_idx is a filament type identifier stored in the 3MF file when the user
  4003. slices (e.g., "GFA00" for generic PLA, "P4d64437" for custom presets). If the same
  4004. tray_info_idx appears in only ONE available tray, we use that tray. If multiple trays
  4005. have the same tray_info_idx (e.g., two spools of generic PLA), we fall back to color
  4006. matching among those trays.
  4007. Args:
  4008. required: List of required filaments with slot_id, type, color, tray_info_idx
  4009. loaded: List of loaded filaments with type, color, tray_info_idx, global_tray_id
  4010. Returns:
  4011. AMS mapping array (position = slot_id - 1, value = global_tray_id or -1)
  4012. """
  4013. if not required:
  4014. return None
  4015. # Track used trays to avoid duplicate assignment
  4016. used_tray_ids: set[int] = set()
  4017. comparisons = []
  4018. for req in required:
  4019. req_type = (req.get("type") or "").upper()
  4020. req_color = req.get("color", "")
  4021. req_tray_info_idx = req.get("tray_info_idx", "")
  4022. # Find best match: unique tray_info_idx > exact color > similar color > type-only
  4023. idx_match = None
  4024. exact_match = None
  4025. similar_match = None
  4026. similar_distance = float("inf")
  4027. type_only_match = None
  4028. # Get available trays (not already used)
  4029. available = [f for f in loaded if f["global_tray_id"] not in used_tray_ids]
  4030. # Nozzle-aware filtering: restrict to trays on the correct nozzle.
  4031. # Hard filter — cross-nozzle assignment causes print failures
  4032. # ("position of left hotend is abnormal"), so never fall back.
  4033. # Skipped when an FTS is installed: it routes any AMS slot to either
  4034. # extruder, so restricting to one nozzle would wrongly exclude the
  4035. # correct spool sitting in the other nozzle's AMS (#2186).
  4036. req_nozzle_id = req.get("nozzle_id")
  4037. if req_nozzle_id is not None and not fts_installed:
  4038. available = [f for f in available if f.get("extruder_id") == req_nozzle_id]
  4039. # Sort by remaining filament (ascending) so lowest-remain spool wins .find().
  4040. # Inventory-tracked spools sort before MQTT-only ones (#1508); see
  4041. # _prefer_lowest_sort_key for the full rationale.
  4042. if prefer_lowest:
  4043. available.sort(key=lambda f: self._prefer_lowest_sort_key(f, inventory_remain_overrides))
  4044. # INFO-level decision trace for "Prefer Lowest Filament" #1766.
  4045. # One line per filament req so a bug report can be diagnosed
  4046. # without enabling debug logging: shows what the matcher saw
  4047. # (req shape + sorted candidate trays with their remain values
  4048. # and any inventory override that was applied). Mirrored by
  4049. # the picked-match log at the bottom of the loop.
  4050. logger.info(
  4051. "[prefer-lowest] req slot=%s type=%r color=%r tii=%r nozzle=%s; available (sorted lowest-first): %s",
  4052. req.get("slot_id"),
  4053. req_type,
  4054. req_color,
  4055. req_tray_info_idx,
  4056. req_nozzle_id,
  4057. [
  4058. {
  4059. "gtid": f.get("global_tray_id"),
  4060. "type": f.get("type"),
  4061. "color": f.get("color"),
  4062. "tii": f.get("tray_info_idx"),
  4063. "remain": f.get("remain"),
  4064. "inv_g": (
  4065. inventory_remain_overrides.get(f.get("global_tray_id"))
  4066. if inventory_remain_overrides
  4067. else None
  4068. ),
  4069. }
  4070. for f in available
  4071. ],
  4072. )
  4073. # Check if tray_info_idx is unique among available trays
  4074. if req_tray_info_idx:
  4075. idx_matches = [f for f in available if f.get("tray_info_idx") == req_tray_info_idx]
  4076. if len(idx_matches) == 1:
  4077. # Unique tray_info_idx - use it as definitive match
  4078. idx_match = idx_matches[0]
  4079. logger.debug(
  4080. f"Matched filament slot {req.get('slot_id')} by unique tray_info_idx={req_tray_info_idx} "
  4081. f"-> tray {idx_match['global_tray_id']}"
  4082. )
  4083. elif len(idx_matches) > 1:
  4084. # Multiple trays with same tray_info_idx - use color matching among them
  4085. logger.debug(
  4086. f"Non-unique tray_info_idx={req_tray_info_idx} found in {len(idx_matches)} trays, "
  4087. f"using color matching among trays: {[f['global_tray_id'] for f in idx_matches]}"
  4088. )
  4089. if prefer_lowest:
  4090. idx_matches.sort(key=lambda f: self._prefer_lowest_sort_key(f, inventory_remain_overrides))
  4091. # Use color matching within this subset
  4092. for f in idx_matches:
  4093. f_color = f.get("color", "")
  4094. if self._normalize_color_for_compare(f_color) == self._normalize_color_for_compare(req_color):
  4095. if not exact_match:
  4096. exact_match = f
  4097. elif self._colors_are_similar(f_color, req_color):
  4098. distance = self._color_distance(f_color, req_color)
  4099. if distance is not None and distance < similar_distance:
  4100. similar_match = f
  4101. similar_distance = distance
  4102. elif not type_only_match:
  4103. type_only_match = f
  4104. # If no idx_match yet, do standard type/color matching on all available trays
  4105. if not idx_match and not exact_match and not similar_match and not type_only_match:
  4106. for f in available:
  4107. f_type = (f.get("type") or "").upper()
  4108. if canonical_filament_type(f_type) != canonical_filament_type(req_type):
  4109. continue
  4110. # Type matches - check color
  4111. f_color = f.get("color", "")
  4112. if self._normalize_color_for_compare(f_color) == self._normalize_color_for_compare(req_color):
  4113. if not exact_match:
  4114. exact_match = f
  4115. elif self._colors_are_similar(f_color, req_color):
  4116. # Nearest wins, not first-in-tray-order. `available` is
  4117. # already in the caller's order (slot order, or the
  4118. # prefer-lowest sort), and `<` keeps the earliest of
  4119. # equally close spools — so that order survives as the
  4120. # tie-break (#2804).
  4121. distance = self._color_distance(f_color, req_color)
  4122. if distance is not None and distance < similar_distance:
  4123. similar_match = f
  4124. similar_distance = distance
  4125. elif not type_only_match:
  4126. type_only_match = f
  4127. match = idx_match or exact_match or similar_match or type_only_match
  4128. if match:
  4129. used_tray_ids.add(match["global_tray_id"])
  4130. comparisons.append({"slot_id": req.get("slot_id", 0), "global_tray_id": match["global_tray_id"]})
  4131. else:
  4132. comparisons.append({"slot_id": req.get("slot_id", 0), "global_tray_id": -1})
  4133. # Which bucket won, always — not only under Prefer Lowest (#2804).
  4134. # "Why did it pick that spool" is the question every wrong-filament
  4135. # report starts with, and a `similar_color` win is now a ranked
  4136. # choice among several eligible spools rather than whichever tray
  4137. # came first, so it is worth being able to see after the fact.
  4138. # Pairs with the "available (sorted)" log above when Prefer Lowest
  4139. # is on (#1766).
  4140. if match:
  4141. bucket = (
  4142. "idx"
  4143. if idx_match is not None
  4144. else "exact_color"
  4145. if exact_match is not None
  4146. else "similar_color"
  4147. if similar_match is not None
  4148. else "type_only"
  4149. )
  4150. logger.info(
  4151. "[ams-match] picked gtid=%s via %s for req slot=%s%s",
  4152. match["global_tray_id"],
  4153. bucket,
  4154. req.get("slot_id"),
  4155. f" (deltaE {similar_distance:.2f})" if bucket == "similar_color" else "",
  4156. )
  4157. else:
  4158. logger.info(
  4159. "[ams-match] NO MATCH for req slot=%s (type=%r color=%r tii=%r)",
  4160. req.get("slot_id"),
  4161. req_type,
  4162. req_color,
  4163. req_tray_info_idx,
  4164. )
  4165. # Build mapping array
  4166. if not comparisons:
  4167. return None
  4168. max_slot_id = max(c["slot_id"] for c in comparisons)
  4169. if max_slot_id <= 0:
  4170. return None
  4171. mapping = [-1] * max_slot_id
  4172. for c in comparisons:
  4173. slot_id = c["slot_id"]
  4174. if slot_id and slot_id > 0:
  4175. mapping[slot_id - 1] = c["global_tray_id"]
  4176. return mapping
  4177. def _mark_printer_dispatched(
  4178. self,
  4179. printer_id: int,
  4180. pre_state: str | None,
  4181. pre_subtask_id: str | None,
  4182. ) -> None:
  4183. """Record that a print command was just sent to ``printer_id``.
  4184. Held until either the watchdog observes a state/subtask transition
  4185. (success path) or the hard timeout expires. See ``_dispatch_holds``.
  4186. """
  4187. if not pre_state:
  4188. # No pre_state means we can't detect a transition — fall back to a
  4189. # pure time-based hold using empty string as a sentinel that won't
  4190. # match any real printer state.
  4191. pre_state = ""
  4192. self._dispatch_holds[printer_id] = (time.monotonic(), pre_state, pre_subtask_id)
  4193. def _release_dispatch_hold(self, printer_id: int) -> None:
  4194. """Drop the dispatch hold for ``printer_id`` (called by the watchdog)."""
  4195. self._dispatch_holds.pop(printer_id, None)
  4196. def _printer_in_dispatch_hold(self, printer_id: int) -> bool:
  4197. """True if ``printer_id`` is still inside its post-dispatch hold window.
  4198. Returns False (and clears the hold) once any of these are true:
  4199. - hard timeout (``_dispatch_max_hold``) has elapsed
  4200. - the printer has transitioned out of pre_state and we're past the
  4201. minimum cooldown
  4202. - the printer's subtask_id has advanced past pre_subtask_id and we're
  4203. past the minimum cooldown
  4204. Otherwise the printer is held — caller should treat it as busy.
  4205. """
  4206. entry = self._dispatch_holds.get(printer_id)
  4207. if not entry:
  4208. return False
  4209. started_at, pre_state, pre_subtask_id = entry
  4210. elapsed = time.monotonic() - started_at
  4211. if elapsed >= self._dispatch_max_hold:
  4212. self._dispatch_holds.pop(printer_id, None)
  4213. return False
  4214. # Without a pre_state we can't detect a transition — fall back to the
  4215. # min cooldown alone, then drop the hold.
  4216. if not pre_state:
  4217. if elapsed >= self._dispatch_min_cooldown:
  4218. self._dispatch_holds.pop(printer_id, None)
  4219. return False
  4220. return True
  4221. status = printer_manager.get_status(printer_id)
  4222. current_state = getattr(status, "state", None) if status else None
  4223. current_subtask_id = getattr(status, "subtask_id", None) if status else None
  4224. transitioned = (current_state is not None and current_state != pre_state) or (
  4225. pre_subtask_id is not None and current_subtask_id is not None and current_subtask_id != pre_subtask_id
  4226. )
  4227. if transitioned and elapsed >= self._dispatch_min_cooldown:
  4228. self._dispatch_holds.pop(printer_id, None)
  4229. return False
  4230. return True
  4231. def _is_printer_idle(self, printer_id: int, require_plate_clear: bool = True) -> bool:
  4232. """Check if a printer is connected and idle."""
  4233. if not printer_manager.is_connected(printer_id):
  4234. logger.debug("Printer %d: not connected", printer_id)
  4235. return False
  4236. state = printer_manager.get_status(printer_id)
  4237. if not state:
  4238. logger.debug("Printer %d: no status available", printer_id)
  4239. return False
  4240. # Plate-clear gate: if the printer finished/failed a previous print and the user
  4241. # hasn't acknowledged the plate was cleared, the queue must not dispatch the next
  4242. # job — even if the printer currently reports IDLE. After Auto Off cycles the
  4243. # printer, it boots back into IDLE with no memory of the previous finish; without
  4244. # the persisted awaiting flag we'd bypass the confirmation prompt (#961).
  4245. if require_plate_clear and printer_manager.is_awaiting_plate_clear(printer_id):
  4246. logger.debug(
  4247. "Printer %d: not idle — awaiting plate-clear acknowledgment (state=%s)",
  4248. printer_id,
  4249. state.state,
  4250. )
  4251. return False
  4252. idle = state.state in ("IDLE", "FINISH", "FAILED")
  4253. if not idle:
  4254. logger.debug("Printer %d: not idle — state=%s", printer_id, state.state)
  4255. return idle
  4256. @staticmethod
  4257. def _pinned_hold_reason(printer_id: int, printer_name: str, require_plate_clear: bool) -> str:
  4258. """Why a connected, non-idle printer cannot take this job, for the queue row (#3074).
  4259. Only ever asked about a printer :meth:`_is_printer_idle` has just refused
  4260. and that the fixed-printer branch has already found connected, so the two
  4261. offline cases answer at their own exits and never arrive here.
  4262. The default is the model-based branch's ``Busy:`` wording, which
  4263. :meth:`_is_busy_only` reads as "resolves itself, stay quiet". That is also
  4264. the right answer for the connected-but-no-telemetry second or two after a
  4265. reconnect: the model-based branch has always reported it that way, and it
  4266. is not something to wake anybody up for.
  4267. A plate nobody has confirmed is the one case here that does not resolve
  4268. itself — somebody has to walk over to the printer — so it is worded as
  4269. itself and allowed to notify.
  4270. """
  4271. if require_plate_clear and printer_manager.is_awaiting_plate_clear(printer_id):
  4272. return f"Waiting for plate confirmation: {printer_name}"
  4273. return f"Busy: {printer_name}"
  4274. async def _get_setting(self, db: AsyncSession, key: str) -> str | None:
  4275. """Read a setting value from the database."""
  4276. result = await db.execute(select(Settings).where(Settings.key == key))
  4277. setting = result.scalar_one_or_none()
  4278. return setting.value if setting else None
  4279. async def _get_bool_setting(self, db: AsyncSession, key: str, default: bool = False) -> bool:
  4280. """Read a boolean setting from the database."""
  4281. result = await db.execute(select(Settings).where(Settings.key == key))
  4282. setting = result.scalar_one_or_none()
  4283. if setting:
  4284. return setting.value.lower() == "true"
  4285. return default
  4286. async def _get_int_setting(self, db: AsyncSession, key: str, default: int) -> int:
  4287. """Read an int setting; falls back to default on missing/unparseable rows."""
  4288. result = await db.execute(select(Settings).where(Settings.key == key))
  4289. setting = result.scalar_one_or_none()
  4290. if setting and setting.value:
  4291. try:
  4292. return int(setting.value)
  4293. except ValueError:
  4294. pass
  4295. return default
  4296. async def _get_drying_presets(self, db: AsyncSession) -> dict[str, dict[str, int]]:
  4297. """Get drying presets (user-configured or built-in defaults)."""
  4298. result = await db.execute(select(Settings).where(Settings.key == "drying_presets"))
  4299. setting = result.scalar_one_or_none()
  4300. if setting and setting.value:
  4301. try:
  4302. presets = json.loads(setting.value)
  4303. if isinstance(presets, dict) and presets:
  4304. return presets
  4305. except json.JSONDecodeError:
  4306. pass
  4307. return self.DEFAULT_DRYING_PRESETS
  4308. async def _get_humidity_thresholds(self, db: AsyncSession) -> dict[str, int]:
  4309. """Per-filament humidity thresholds (#1605).
  4310. Returns the user-configured overrides map keyed by normalized filament
  4311. type (uppercase base, e.g. ``PLA``, ``ASA``) plus a ``default`` key for
  4312. unknown / unmapped types. Empty / unset → empty dict, in which case
  4313. callers fall back to ``ams_humidity_fair``.
  4314. """
  4315. result = await db.execute(select(Settings).where(Settings.key == "ams_humidity_thresholds"))
  4316. setting = result.scalar_one_or_none()
  4317. if not setting or not setting.value:
  4318. return {}
  4319. try:
  4320. data = json.loads(setting.value)
  4321. except json.JSONDecodeError:
  4322. return {}
  4323. if not isinstance(data, dict):
  4324. return {}
  4325. out: dict[str, int] = {}
  4326. for key, value in data.items():
  4327. try:
  4328. out[str(key).upper() if key != "default" else "default"] = int(value)
  4329. except (TypeError, ValueError):
  4330. continue
  4331. return out
  4332. # Materials whose AMS spelling differs from the key the tables above use.
  4333. # Bambu labels nylon "PA" while its own composites spell the family out, so
  4334. # PA6, PA11, PA12 and PAHT would otherwise miss a table with a perfectly
  4335. # good PA row (#3067).
  4336. #
  4337. # Mirrors DRYING_MATERIAL_ALIASES in frontend/src/utils/dryingPresets.ts.
  4338. # The drying popover has resolved these correctly since #2774 and the
  4339. # scheduler never did, which is exactly why #3067's reporter could dry a
  4340. # PA6-CF spool by hand while auto-drying skipped it every pass.
  4341. #
  4342. # PPA is here too. Polyphthalamide is a distinct polymer rather than a grade
  4343. # of nylon, so it is the one entry that is a judgement rather than a
  4344. # spelling -- but it is an aromatic polyamide, it absorbs moisture the same
  4345. # way, and PA's row is the hottest the table has. Drying it there is closer
  4346. # to right than not drying it at all, which is what it got before.
  4347. FILAMENT_KEY_ALIASES: dict[str, str] = {
  4348. "NYLON": "PA",
  4349. "PA6": "PA",
  4350. "PA11": "PA",
  4351. "PA12": "PA",
  4352. "PAHT": "PA",
  4353. "PPA": "PA",
  4354. }
  4355. @classmethod
  4356. def _resolve_filament_key(cls, tray_type: str | None, table: Mapping[str, object]) -> str | None:
  4357. """The key in *table* that answers for this tray's material, or None.
  4358. The printer reports the material in ``tray_type``, and it spells filled
  4359. and foamed variants out: PLA-CF, PETG-CF, ABS-GF, PLA-AERO, PA6-CF. The
  4360. tables here are keyed by base material, so matching the raw string alone
  4361. found a row for 8 of the 41 types a printer can report and skipped the
  4362. rest -- silently, because every caller reads "no row" as "nothing to do
  4363. for this tray". Auto-drying therefore ignored every composite spool on
  4364. the install (#3067).
  4365. Exact match first, so a table the user has extended with a row of its
  4366. own -- ``PA6-CF`` at a temperature they picked -- still wins over the
  4367. base material's. Then the suffix is dropped, then the alias map above
  4368. answers for the polyamide spellings.
  4369. Returns None rather than a default: what to do with an unrecognised
  4370. material differs per caller, and only the caller knows whether "no row"
  4371. means skip the tray or fall back to a catch-all. Nothing here invents a
  4372. temperature for a material the table does not list.
  4373. """
  4374. raw = cls._normalize_filament_type(tray_type or "")
  4375. if not raw:
  4376. return None
  4377. for candidate in (raw, raw.split("-")[0]):
  4378. key = cls.FILAMENT_KEY_ALIASES.get(candidate, candidate)
  4379. if key in table:
  4380. return key
  4381. return None
  4382. @staticmethod
  4383. def resolve_humidity_threshold(trays: list[dict], thresholds: dict[str, int], fallback: int) -> int:
  4384. """Resolve the effective humidity threshold for an AMS unit (#1605).
  4385. For mixed filament types loaded into one AMS, returns the most
  4386. restrictive (lowest) threshold across all loaded tray types — matches
  4387. the conservative-params strategy already used for drying temp/hours.
  4388. Empty / unloaded trays contribute no constraint. Unknown types use the
  4389. ``default`` key, falling through to ``fallback`` (= ``ams_humidity_fair``)
  4390. when no per-type map is configured at all.
  4391. """
  4392. default = thresholds.get("default", fallback)
  4393. if not thresholds:
  4394. return fallback
  4395. candidates: list[int] = []
  4396. for tray in trays:
  4397. tray_type = str(tray.get("tray_type") or "").strip()
  4398. if not tray_type:
  4399. continue
  4400. # A composite carries its base material's threshold when it has no
  4401. # row of its own, the same way it takes its drying preset (#3067).
  4402. key = PrintScheduler._resolve_filament_key(tray_type, thresholds)
  4403. candidates.append(thresholds[key] if key is not None else default)
  4404. if not candidates:
  4405. return default
  4406. return min(candidates)
  4407. def _get_conservative_drying_params(
  4408. self, trays: list[dict], module_type: str, presets: dict[str, dict[str, int]]
  4409. ) -> tuple[int, int, str] | None:
  4410. """Get the most conservative drying params for mixed filament types in an AMS unit.
  4411. Returns (temp, duration_hours, filament_type) or None if no drying-eligible filaments.
  4412. """
  4413. temp_key = module_type if module_type in ("n3f", "n3s") else "n3f"
  4414. hours_key = f"{temp_key}_hours"
  4415. min_temp = None
  4416. max_hours = None
  4417. filament_type = ""
  4418. for tray in trays:
  4419. tray_type = tray.get("tray_type", "")
  4420. if not tray_type:
  4421. continue
  4422. # "PLA Basic" -> PLA, and "PA6-CF" -> PA rather than nothing at all,
  4423. # which is what stopped auto-drying on every composite spool (#3067).
  4424. base_type = self._resolve_filament_key(tray_type, presets)
  4425. if base_type is None:
  4426. continue
  4427. # The table is user-editable JSON with no per-row validation, so a
  4428. # row can be present and empty. That has always meant "skip this
  4429. # material", and it has to keep meaning it: the reads below fall
  4430. # back to 55C/12h per missing field, which is a temperature nobody
  4431. # chose and would deform a PLA spool.
  4432. preset = presets[base_type]
  4433. if not preset:
  4434. continue
  4435. temp = preset.get(temp_key, 55)
  4436. hours = preset.get(hours_key, 12)
  4437. # Conservative: lowest temp, longest duration
  4438. if min_temp is None or temp < min_temp:
  4439. min_temp = temp
  4440. if max_hours is None or hours > max_hours:
  4441. max_hours = hours
  4442. if not filament_type:
  4443. filament_type = base_type
  4444. if min_temp is None:
  4445. return None
  4446. return (min_temp, max_hours or 12, filament_type)
  4447. async def _check_filament_low(self, db: AsyncSession) -> None:
  4448. """Alert on AMS slots whose assigned spool has crossed its low-stock threshold (#2913).
  4449. ``on_filament_low`` has had a column, a schema field, a route, a template
  4450. and a UI toggle since the notification system was built, and no caller --
  4451. so the toggle could be switched on and could never fire. This is the
  4452. producer.
  4453. No new setting. ``low_stock_threshold`` (default 20%) with the per-spool
  4454. ``low_stock_threshold_pct`` override is already exactly this decision:
  4455. already configurable, already surfaced, already driving the Inventory
  4456. page's Low Stock count. The AMS ``remain`` percentage is deliberately not
  4457. consulted -- it has been measured up to 56 points out against a scale,
  4458. which is not a number to page someone on.
  4459. Only slots with an assigned spool produce an event. A slot Bambuddy
  4460. cannot resolve to a spool has no remaining weight it can stand behind,
  4461. and guessing one is how the remain percentage would have got in.
  4462. """
  4463. from backend.app.models.spool import Spool
  4464. # Time-gated rather than run on every pass. run() sleeps
  4465. # _fast_check_interval -- 3 seconds -- on any productive pass, and the
  4466. # early-return path counts as productive while an upload is in flight
  4467. # (#2602), so an unthrottled check would re-read the whole spool
  4468. # collection every 3 seconds for the length of a batch drain. In
  4469. # Spoolman mode that is a request storm against a third-party service,
  4470. # which is the thing this producer's design note argues against; and
  4471. # when Spoolman is down, _get_with_retry burns ~16 s inside
  4472. # check_queue holding the scheduler's session, so the outage would
  4473. # throttle the print queue itself. A low spool is not a 3-second
  4474. # concern -- the idle interval is the natural resolution.
  4475. now = time.monotonic()
  4476. if now < self._filament_low_next_check:
  4477. return
  4478. self._filament_low_next_check = now + _FILAMENT_LOW_MIN_INTERVAL
  4479. try:
  4480. # on_filament_low defaults to off on every provider, so on most
  4481. # installs nobody wants this alert. Without this the check would
  4482. # still read every assigned spool each interval -- the whole
  4483. # collection over HTTP in Spoolman mode -- for an event that is
  4484. # switched off. One query against the provider table settles it
  4485. # before any spool work, the same guard the bed-cooled waiter uses.
  4486. # No printer_id: this asks whether any provider wants the event at
  4487. # all; per-printer scoping is applied when the event is sent.
  4488. #
  4489. # While nobody wants it, no pass sees a slot go back above its
  4490. # threshold, so nothing can re-arm; a spool refilled under the same
  4491. # id in that time would stay silenced until a restart. Forgetting
  4492. # what was sent is right anyway: whoever switches the event back on
  4493. # is told about the spools that are low now.
  4494. if not await notification_service._get_providers_for_event(db, "on_filament_low"):
  4495. self._notified_filament_low.clear()
  4496. return
  4497. global_threshold = await self._get_low_stock_threshold(db)
  4498. spoolman_on = await self._get_bool_setting(db, "spoolman_enabled")
  4499. # (printer_id, ams_id, tray_id, spool_id) -> (remaining_pct, threshold, colour name)
  4500. slots: dict[tuple[int, int, int, int], tuple[float, float, str | None]] = {}
  4501. if spoolman_on:
  4502. slots = await self._filament_low_slots_spoolman(db, global_threshold)
  4503. else:
  4504. rows = (
  4505. await db.execute(
  4506. select(SpoolAssignment, Spool)
  4507. .join(Spool, SpoolAssignment.spool_id == Spool.id)
  4508. # An archived spool is not stock. The Inventory page's
  4509. # Low Stock count skips them (InventoryPage.tsx:1089)
  4510. # and get_all_spools without allow_archived already
  4511. # excludes them in Spoolman mode, so without this the
  4512. # internal path is the only one that alerts on them.
  4513. .where(Spool.archived_at.is_(None))
  4514. )
  4515. ).all()
  4516. for assignment, spool in rows:
  4517. pct = _remaining_percent(spool.label_weight, spool.weight_used)
  4518. if pct is None:
  4519. continue
  4520. threshold = float(spool.low_stock_threshold_pct or global_threshold)
  4521. slots[(assignment.printer_id, assignment.ams_id, assignment.tray_id, spool.id)] = (
  4522. pct,
  4523. threshold,
  4524. spool.color_name,
  4525. )
  4526. if not slots:
  4527. # Nothing resolvable this pass. Deliberately not clearing the
  4528. # notified set: a Spoolman that is briefly unreachable would
  4529. # otherwise re-alert on every spool as soon as it came back.
  4530. return
  4531. await self._emit_filament_low(db, slots)
  4532. except Exception as e:
  4533. logger.warning("Low-filament check failed: %s", e, exc_info=True)
  4534. async def _get_low_stock_threshold(self, db: AsyncSession) -> float:
  4535. """The global low-stock percentage, defaulting to the schema's 20.0."""
  4536. raw = (await db.execute(select(Settings).where(Settings.key == "low_stock_threshold"))).scalar_one_or_none()
  4537. if raw is None or raw.value is None:
  4538. return 20.0
  4539. try:
  4540. value = float(raw.value)
  4541. except (TypeError, ValueError):
  4542. return 20.0
  4543. return value if 0 < value <= 100 else 20.0
  4544. async def _filament_low_slots_spoolman(
  4545. self, db: AsyncSession, global_threshold: float
  4546. ) -> dict[tuple[int, int, int, int], tuple[float, float, str | None]]:
  4547. """Resolve Spoolman-mode slots to (remaining %, threshold, colour name).
  4548. Spoolman spools carry no per-spool override -- ``low_stock_threshold_pct``
  4549. is a column on Bambuddy's own spool table and has no Spoolman equivalent,
  4550. so the global threshold is the only one that applies here. That matches
  4551. what the Inventory page already does in this mode.
  4552. One ``get_all_spools`` call covers every slot rather than a request per
  4553. spool. Archived spools do not appear -- ``get_all_spools`` excludes them
  4554. without ``allow_archived`` -- which is the behaviour the internal path
  4555. has to ask for explicitly.
  4556. An unreachable Spoolman returns no slots rather than raising. It is an
  4557. ordinary state for a third-party service, not an error in this pass: the
  4558. caller's "nothing resolvable" branch already leaves the notified set
  4559. alone, which is exactly right here, whereas letting it reach the broad
  4560. ``except`` would write a full traceback every time the check runs.
  4561. """
  4562. from backend.app.api.routes._spoolman_helpers import _map_spoolman_spool
  4563. from backend.app.services.spoolman import SpoolmanUnavailableError, get_spoolman_client
  4564. assignments = (await db.execute(select(SpoolmanSlotAssignment))).scalars().all()
  4565. if not assignments:
  4566. return {}
  4567. client = await get_spoolman_client()
  4568. if client is None:
  4569. return {}
  4570. try:
  4571. all_spools = await client.get_all_spools()
  4572. except SpoolmanUnavailableError as e:
  4573. logger.debug("Low-filament check skipped, Spoolman unreachable: %s", e)
  4574. return {}
  4575. by_id: dict[int, dict] = {}
  4576. for raw in all_spools:
  4577. raw_id = raw.get("id")
  4578. if isinstance(raw_id, int):
  4579. by_id[raw_id] = raw
  4580. slots: dict[tuple[int, int, int, int], tuple[float, float, str | None]] = {}
  4581. for assignment in assignments:
  4582. raw = by_id.get(assignment.spoolman_spool_id)
  4583. if raw is None:
  4584. continue
  4585. try:
  4586. mapped = _map_spoolman_spool(raw)
  4587. except ValueError:
  4588. continue
  4589. pct = _remaining_percent(mapped.get("label_weight"), mapped.get("weight_used"))
  4590. if pct is None:
  4591. continue
  4592. key = (assignment.printer_id, assignment.ams_id, assignment.tray_id, assignment.spoolman_spool_id)
  4593. slots[key] = (pct, global_threshold, mapped.get("color_name"))
  4594. return slots
  4595. async def _emit_filament_low(
  4596. self, db: AsyncSession, slots: dict[tuple[int, int, int, int], tuple[float, float, str | None]]
  4597. ) -> None:
  4598. """Send one notification per slot that has newly crossed its threshold."""
  4599. printer_names: dict[int, str] = {}
  4600. for key, (pct, threshold, color) in slots.items():
  4601. printer_id, ams_id, tray_id, _spool_id = key
  4602. if pct >= threshold:
  4603. # Back above the line: re-arm rather than expire on a timer, so a
  4604. # refilled slot can alert again and a spool sitting just under the
  4605. # threshold stays quiet.
  4606. self._notified_filament_low.discard(key)
  4607. continue
  4608. if key in self._notified_filament_low:
  4609. continue
  4610. # Marked before sending, which is a choice rather than the only
  4611. # option, and it is the opposite failure mode from the one the
  4612. # debounce argues against: one transient provider failure loses this
  4613. # alert until the spool goes back above the threshold and crosses it
  4614. # again. Marking after a successful send would trade that for
  4615. # re-alerting on every pass while a provider is down -- which is the
  4616. # repetition the whole debounce exists to prevent, and the louder of
  4617. # the two failures. A spool that is low stays low, so the next real
  4618. # signal is not far away; a provider stuck retrying is a signal that
  4619. # never stops.
  4620. self._notified_filament_low.add(key)
  4621. if printer_id not in printer_names:
  4622. printer = (await db.execute(select(Printer).where(Printer.id == printer_id))).scalar_one_or_none()
  4623. if printer is None:
  4624. continue
  4625. printer_names[printer_id] = printer.name
  4626. try:
  4627. await notification_service.on_filament_low(
  4628. printer_id,
  4629. printer_names[printer_id],
  4630. _ams_slot_label(ams_id, tray_id),
  4631. int(pct),
  4632. db,
  4633. color=color,
  4634. )
  4635. except Exception as e:
  4636. logger.warning("Low-filament notification failed for slot %s: %s", key, e)
  4637. async def _check_stock_forecast(self, db: AsyncSession) -> None:
  4638. """Alert when a filament SKU reaches its reorder point or is about to break (#2955).
  4639. ``on_stock_reorder_alert`` and ``on_stock_break_alert`` have had a column,
  4640. a schema field, a template and a UI toggle, and nothing that computes the
  4641. condition -- the Forecast panel does it in the browser, so with no page
  4642. open nothing could ever alert. This runs the same arithmetic
  4643. (``stock_forecast``) from the scheduler loop.
  4644. An alert is sent when a SKU *moves into* a condition, once, and again
  4645. after the condition has cleared. A SKU that worsens from reorder to break
  4646. alerts again as a break, because the panel treats the two as exclusive; one
  4647. that eases from break back to reorder does not alert again.
  4648. A SKU with alerts snoozed is treated as not alerting, so un-snoozing it
  4649. while it is still low tells you.
  4650. Time-gated to hourly for the reasons ``_check_filament_low`` is gated: the
  4651. loop runs every 3 s while an upload is in flight, and the whole spool
  4652. collection is read over HTTP in Spoolman mode. Both events default to off
  4653. on every provider, so nothing is read unless a provider wants one.
  4654. """
  4655. now = time.monotonic()
  4656. if now < self._stock_forecast_next_check:
  4657. return
  4658. self._stock_forecast_next_check = now + _STOCK_FORECAST_MIN_INTERVAL
  4659. try:
  4660. if self._stock_alerts_persisted is None:
  4661. await self._load_stock_alerts(db)
  4662. wanted = {
  4663. "reorder": bool(await notification_service._get_providers_for_event(db, "on_stock_reorder_alert")),
  4664. "break": bool(await notification_service._get_providers_for_event(db, "on_stock_break_alert")),
  4665. }
  4666. if not any(wanted.values()):
  4667. # Nobody wants either event. Forget what was sent, so whoever
  4668. # switches one on is told about the SKUs that are low now.
  4669. self._notified_stock_alerts.clear()
  4670. await self._save_stock_alerts(db)
  4671. return
  4672. forecasts = await self._stock_forecasts(db)
  4673. if forecasts is None:
  4674. # Spoolman unreachable. Deliberately not clearing the notified
  4675. # set: a brief outage would otherwise re-alert every SKU when it
  4676. # came back.
  4677. return
  4678. await self._emit_stock_alerts(db, forecasts, wanted)
  4679. await self._save_stock_alerts(db)
  4680. except Exception as e:
  4681. logger.warning("Stock forecast check failed: %s", e, exc_info=True)
  4682. def _serialize_stock_alerts(self) -> str:
  4683. """The notified set as JSON, in a fixed order so an unchanged set compares equal."""
  4684. rows = sorted([list(key), kind, sorted(events)] for key, (kind, events) in self._notified_stock_alerts.items())
  4685. return json.dumps(rows)
  4686. async def _load_stock_alerts(self, db: AsyncSession) -> None:
  4687. """Read the notified set saved by the last run, once per process.
  4688. A missing or unreadable row is an empty set: the worst that does is one
  4689. repeat of what a restart used to cause every time.
  4690. """
  4691. raw = (await db.execute(select(Settings).where(Settings.key == _STOCK_ALERTS_SETTING_KEY))).scalar_one_or_none()
  4692. loaded: dict[stock_forecast.SkuKey, tuple[str, frozenset[str]]] = {}
  4693. if raw and raw.value:
  4694. try:
  4695. for key, kind, events in json.loads(raw.value):
  4696. if len(key) == 4 and kind in ("reorder", "break"):
  4697. loaded[tuple(str(part) for part in key)] = (kind, frozenset(str(e) for e in events))
  4698. except (ValueError, TypeError):
  4699. logger.warning("Ignoring unreadable %s setting", _STOCK_ALERTS_SETTING_KEY)
  4700. loaded = {}
  4701. self._notified_stock_alerts = loaded
  4702. self._stock_alerts_persisted = self._serialize_stock_alerts()
  4703. async def _save_stock_alerts(self, db: AsyncSession) -> None:
  4704. """Write the notified set to its settings row if it changed since it was last read or written."""
  4705. serialized = self._serialize_stock_alerts()
  4706. if serialized == self._stock_alerts_persisted:
  4707. return
  4708. from backend.app.core.db_dialect import upsert_setting
  4709. await upsert_setting(db, Settings, _STOCK_ALERTS_SETTING_KEY, serialized)
  4710. await db.commit()
  4711. self._stock_alerts_persisted = serialized
  4712. async def _stock_forecasts(self, db: AsyncSession) -> stock_forecast.ForecastMap | None:
  4713. """Forecast every active SKU in whichever inventory mode is on. None if it cannot be read."""
  4714. from backend.app.models.filament_sku_settings import FilamentSkuSettings
  4715. global_lead_time = max(0, await self._get_int_setting(db, "forecast_global_lead_time_days", default=0))
  4716. sku_rows = (await db.execute(select(FilamentSkuSettings))).scalars().all()
  4717. sku_settings = {
  4718. stock_forecast.sku_key(row.material, row.subtype, row.brand, row.color_name): stock_forecast.SkuSettings(
  4719. lead_time_days=row.lead_time_days,
  4720. safety_margin_value=row.safety_margin_value,
  4721. safety_margin_unit=row.safety_margin_unit,
  4722. alerts_snoozed=bool(row.alerts_snoozed),
  4723. )
  4724. for row in sku_rows
  4725. }
  4726. if await self._get_bool_setting(db, "spoolman_enabled"):
  4727. spools = await self._stock_spools_spoolman()
  4728. if spools is None:
  4729. return None
  4730. # Spoolman owns the usage in this mode and Bambuddy's own history
  4731. # table holds nothing for those spools, so the rate is always the
  4732. # delta rate. Bambuddy's table is deliberately not consulted: its ids
  4733. # are local spool ids, and a Spoolman id that happens to match one
  4734. # (left over from before Spoolman was switched on) would borrow an
  4735. # unrelated spool's history.
  4736. history: dict[int, list[stock_forecast.UsageRecord]] = {}
  4737. else:
  4738. spools = await self._stock_spools_internal(db)
  4739. history = await self._stock_usage_history(db)
  4740. return stock_forecast.forecast_all(spools, history, sku_settings, global_lead_time, utcnow_naive())
  4741. async def _stock_spools_internal(self, db: AsyncSession) -> list[stock_forecast.StockSpool]:
  4742. from backend.app.models.spool import Spool
  4743. rows = (await db.execute(select(Spool).where(Spool.archived_at.is_(None)))).scalars().all()
  4744. return [
  4745. stock_forecast.StockSpool(
  4746. id=spool.id,
  4747. material=spool.material,
  4748. subtype=spool.subtype,
  4749. brand=spool.brand,
  4750. color_name=spool.color_name,
  4751. label_weight=float(spool.label_weight or 0),
  4752. weight_used=float(spool.weight_used or 0),
  4753. weight_used_baseline=float(spool.weight_used_baseline or 0),
  4754. created_at=spool.created_at,
  4755. )
  4756. for spool in rows
  4757. ]
  4758. async def _stock_usage_history(self, db: AsyncSession) -> dict[int, list[stock_forecast.UsageRecord]]:
  4759. from backend.app.models.spool_usage_history import SpoolUsageHistory
  4760. rows = (
  4761. await db.execute(
  4762. select(SpoolUsageHistory.spool_id, SpoolUsageHistory.created_at, SpoolUsageHistory.weight_used)
  4763. .order_by(SpoolUsageHistory.created_at.desc())
  4764. .limit(_STOCK_FORECAST_HISTORY_LIMIT)
  4765. )
  4766. ).all()
  4767. history: dict[int, list[stock_forecast.UsageRecord]] = {}
  4768. for spool_id, created_at, weight_used in rows:
  4769. if created_at is None:
  4770. continue
  4771. history.setdefault(spool_id, []).append(stock_forecast.UsageRecord(created_at, float(weight_used or 0)))
  4772. return history
  4773. async def _stock_spools_spoolman(self) -> list[stock_forecast.StockSpool] | None:
  4774. """Spoolman's spools in the forecast's shape, or None when Spoolman cannot be reached.
  4775. ``get_all_spools`` leaves archived spools out unless asked, which is the
  4776. exclusion the internal query applies explicitly. ``_map_spoolman_spool``
  4777. is the same mapping the Inventory page (and so the panel) reads.
  4778. """
  4779. from backend.app.api.routes._spoolman_helpers import _map_spoolman_spool
  4780. from backend.app.services.spoolman import SpoolmanUnavailableError, get_spoolman_client
  4781. client = await get_spoolman_client()
  4782. if client is None:
  4783. logger.debug("Stock forecast skipped, no Spoolman client (spoolman_enabled without a URL?)")
  4784. return None
  4785. try:
  4786. raw_spools = await client.get_all_spools()
  4787. except SpoolmanUnavailableError as e:
  4788. logger.debug("Stock forecast skipped, Spoolman unreachable: %s", e)
  4789. return None
  4790. spools: list[stock_forecast.StockSpool] = []
  4791. for raw in raw_spools:
  4792. try:
  4793. mapped = _map_spoolman_spool(raw)
  4794. except ValueError:
  4795. continue
  4796. created_at = None
  4797. if mapped.get("created_at"):
  4798. try:
  4799. created_at = datetime.fromisoformat(str(mapped["created_at"]).replace("Z", "+00:00"))
  4800. except ValueError:
  4801. created_at = None
  4802. spools.append(
  4803. stock_forecast.StockSpool(
  4804. id=mapped["id"],
  4805. material=mapped["material"],
  4806. subtype=mapped.get("subtype"),
  4807. brand=mapped.get("brand"),
  4808. color_name=mapped.get("color_name"),
  4809. label_weight=float(mapped.get("label_weight") or 0),
  4810. weight_used=float(mapped.get("weight_used") or 0),
  4811. weight_used_baseline=float(mapped.get("weight_used_baseline") or 0),
  4812. created_at=created_at,
  4813. color_name_is_synthesized=bool(mapped.get("color_name_is_synthesized")),
  4814. )
  4815. )
  4816. return spools
  4817. async def _emit_stock_alerts(
  4818. self,
  4819. db: AsyncSession,
  4820. forecasts: stock_forecast.ForecastMap,
  4821. wanted: dict[str, bool],
  4822. ) -> None:
  4823. """Send the notifications for each SKU that has moved into a stock alert condition.
  4824. A SKU in break has also reached its reorder point, so both events apply to
  4825. it: the break event goes to providers that have it on, and the reorder event
  4826. to the providers that have only that one on. A provider with both on gets the
  4827. break and no second message.
  4828. What is remembered per SKU is the condition and which events had a
  4829. subscriber when it was sent. An event with no subscriber is forgotten, so
  4830. switching it on later reports the SKUs already in that condition.
  4831. """
  4832. active_events = {event for event, on in wanted.items() if on}
  4833. for key, (forecast, spool) in forecasts.items():
  4834. kind: str | None = None
  4835. if not forecast.snoozed:
  4836. if forecast.stock_break_alert:
  4837. kind = "break"
  4838. elif forecast.reorder_alert:
  4839. kind = "reorder"
  4840. applicable = {"break": {"break", "reorder"}, "reorder": {"reorder"}}.get(kind or "", set())
  4841. events = applicable & active_events
  4842. if not events:
  4843. self._notified_stock_alerts.pop(key, None)
  4844. continue
  4845. previous = self._notified_stock_alerts.get(key)
  4846. told = (previous[1] & active_events) if previous else frozenset()
  4847. # Trimmed to the events this condition can send. That is what makes easing from
  4848. # break back to reorder silent (the reorder event was told with the break) and
  4849. # worsening again news again (the break event is no longer remembered).
  4850. self._notified_stock_alerts[key] = (kind, frozenset((told | events) & applicable))
  4851. # Recorded before sending, for the reason _emit_filament_low gives: a
  4852. # provider that is down should lose this alert, not repeat it every hour.
  4853. to_send = events - told
  4854. # Both flags imply a positive rate and so a day count.
  4855. days_left = forecast.days_remaining if forecast.days_remaining is not None else 0
  4856. rate = forecast.daily_rate_g if forecast.daily_rate_g is not None else 0.0
  4857. color = None if spool.color_name_is_synthesized else spool.color_name
  4858. try:
  4859. if "break" in to_send:
  4860. await notification_service.on_stock_break_alert(
  4861. spool.material,
  4862. spool.brand,
  4863. forecast.remaining_g,
  4864. rate,
  4865. days_left,
  4866. forecast.effective_lead_time_days,
  4867. db,
  4868. subtype=spool.subtype,
  4869. color=color,
  4870. )
  4871. if "reorder" in to_send:
  4872. await notification_service.on_stock_reorder_alert(
  4873. spool.material,
  4874. spool.brand,
  4875. forecast.remaining_g,
  4876. rate,
  4877. days_left,
  4878. db,
  4879. subtype=spool.subtype,
  4880. color=color,
  4881. skip_break_subscribers=kind == "break",
  4882. )
  4883. except Exception as e:
  4884. logger.warning("Stock %s notification failed for %s: %s", kind, key, e)
  4885. # A SKU with no spools left is no longer alerting.
  4886. for key in [k for k in self._notified_stock_alerts if k not in forecasts]:
  4887. del self._notified_stock_alerts[key]
  4888. async def _check_auto_drying(
  4889. self,
  4890. db: AsyncSession,
  4891. queue_items: list[PrintQueueItem],
  4892. dispatching_printers: set[int],
  4893. ):
  4894. """Start drying on idle printers based on humidity.
  4895. Three modes (can all be enabled independently):
  4896. - queue_drying_enabled: Dry between scheduled queue prints
  4897. - ambient_drying_enabled: Dry any idle printer when humidity is high, regardless of queue
  4898. - print_drying_enabled: Also evaluate printers that are currently printing,
  4899. when model+firmware supports "Print While Drying" (gated by
  4900. supports_drying_while_printing). Drying temperature is capped at
  4901. max(40, preset_temp - 5) to protect spools mid-print.
  4902. """
  4903. queue_drying_enabled = await self._get_bool_setting(db, "queue_drying_enabled")
  4904. ambient_drying_enabled = await self._get_bool_setting(db, "ambient_drying_enabled")
  4905. print_drying_enabled = await self._get_bool_setting(db, "print_drying_enabled")
  4906. sustained_minutes = await self._get_int_setting(db, "ambient_drying_sustained_minutes", default=0)
  4907. # The wait belongs to ambient drying: the settings UI only shows it while
  4908. # ambient drying is on, so a value left behind when ambient is turned off
  4909. # must not keep delaying the one path that still reaches the wait gate
  4910. # without it (a mid-print start under print_drying).
  4911. sustained_wait_active = sustained_minutes > 0 and ambient_drying_enabled
  4912. # Clear every streak as soon as the wait is inactive. An early return (or
  4913. # an already-drying unit) may otherwise skip the per-unit cleanup and let
  4914. # a quick toggle-on inherit an old streak.
  4915. if not sustained_wait_active:
  4916. self._auto_dry_above.clear()
  4917. if not queue_drying_enabled and not ambient_drying_enabled:
  4918. # Stop active drying on all printers if both features disabled
  4919. if self._drying_in_progress:
  4920. for pid in list(self._drying_in_progress):
  4921. if pid in self._scheduled_drying_printer_ids:
  4922. continue
  4923. logger.info("Auto-drying: printer %d — stopping, auto-drying disabled", pid)
  4924. await self._stop_drying(pid)
  4925. return
  4926. # Update drying state from printer status (handles backend restart)
  4927. self._sync_drying_state()
  4928. # Find printers with scheduled items (for queue drying mode)
  4929. printers_with_scheduled: set[int] = set()
  4930. printers_with_items: set[int] = set()
  4931. for item in queue_items:
  4932. if item.printer_id:
  4933. printers_with_items.add(item.printer_id)
  4934. if item.scheduled_time and not item.manual_start:
  4935. printers_with_scheduled.add(item.printer_id)
  4936. # If only queue mode is on and no printers have scheduled items, stop drying
  4937. # (but skip this short-circuit when print_drying_enabled is on — busy printers
  4938. # may still be eligible for mid-print drying regardless of queue state).
  4939. if not ambient_drying_enabled and not printers_with_scheduled and not print_drying_enabled:
  4940. for pid in list(self._drying_in_progress):
  4941. if pid in self._scheduled_drying_printer_ids:
  4942. continue
  4943. logger.info("Auto-drying: printer %d — stopping, no scheduled prints in queue", pid)
  4944. await self._stop_drying(pid)
  4945. return
  4946. # Get humidity threshold (global fallback)
  4947. result = await db.execute(select(Settings).where(Settings.key == "ams_humidity_fair"))
  4948. setting = result.scalar_one_or_none()
  4949. global_humidity_threshold = int(setting.value) if setting else 60
  4950. # Per-filament humidity threshold overrides (#1605). Empty → fall back
  4951. # to the global threshold for every AMS unit.
  4952. per_type_thresholds = await self._get_humidity_thresholds(db)
  4953. # Get drying presets
  4954. presets = await self._get_drying_presets(db)
  4955. # Determine if drying should be skipped for printers with pending items
  4956. block_for_drying = await self._get_bool_setting(db, "queue_drying_block")
  4957. # Get all active printers
  4958. all_printers = await db.execute(select(Printer).where(Printer.is_active.is_(True)))
  4959. for printer in all_printers.scalars():
  4960. pid = printer.id
  4961. # Resolve model+firmware up front — needed to decide whether this printer
  4962. # qualifies for mid-print drying (busy printer on capable hardware).
  4963. state = printer_manager.get_status(pid)
  4964. if not state:
  4965. logger.debug("Auto-drying: printer %d skipped — no state", pid)
  4966. continue
  4967. model = printer_manager.get_model(pid)
  4968. firmware = state.firmware_version
  4969. # "Mid-print" has to mean the printer is actually printing (#2801).
  4970. # It used to be inferred from the dispatch set, which also holds
  4971. # printers that merely could not be dispatched to -- so a printer
  4972. # sitting in FINISH behind an unacknowledged plate was treated as
  4973. # printing, had its drying temperature capped by the mid-print
  4974. # spool protection, and was logged as (mid-print) while idle.
  4975. is_printing = state.state in _ACTIVE_PRINT_STATES
  4976. mid_print = is_printing and print_drying_enabled and supports_drying_while_printing(model, firmware)
  4977. # A printer whose print is running or imminent is left alone unless
  4978. # it can dry through it. `dispatching_printers` is deliberately the
  4979. # narrow set: running, held post-dispatch, or mid-upload.
  4980. if (is_printing or pid in dispatching_printers) and not mid_print:
  4981. logger.debug("Auto-drying: printer %d skipped — printing or about to", pid)
  4982. continue
  4983. if not mid_print:
  4984. # In queue-only mode, only dry printers that have scheduled prints
  4985. if not ambient_drying_enabled and pid not in printers_with_scheduled:
  4986. if self._drying_in_progress.get(pid) and pid not in self._scheduled_drying_printer_ids:
  4987. logger.info("Auto-drying: printer %d — stopping, no scheduled prints for this printer", pid)
  4988. await self._stop_drying(pid)
  4989. logger.debug("Auto-drying: printer %d skipped — no scheduled prints", pid)
  4990. continue
  4991. # When block mode is on, don't START new drying on printers with pending items.
  4992. # But allow already-drying printers through so humidity auto-stop logic still runs.
  4993. if block_for_drying and pid in printers_with_items and not self._drying_in_progress.get(pid):
  4994. logger.debug("Auto-drying: printer %d skipped — has pending items (block mode)", pid)
  4995. continue
  4996. if not printer_manager.is_connected(pid):
  4997. logger.debug("Auto-drying: printer %d skipped — not connected", pid)
  4998. continue
  4999. # Plate-clear is deliberately ignored here (#2801). It answers
  5000. # "is the bed ready for the next job", which says nothing about
  5001. # whether the AMS may heat -- and the gap between a finished print
  5002. # and the acknowledgment is exactly when drying is most useful,
  5003. # because the printer is free and nobody is waiting on it. Leaving
  5004. # the plate unacknowledged is also how people hold the queue by
  5005. # hand, and that hold should not cost them their drying.
  5006. if not mid_print and not self._is_printer_idle(pid, require_plate_clear=False):
  5007. logger.debug("Auto-drying: printer %d skipped — not idle", pid)
  5008. continue
  5009. # Check drying capability. For mid-print path, supports_drying_while_printing
  5010. # was already verified when computing mid_print above.
  5011. if not mid_print and not supports_drying(model, firmware):
  5012. logger.debug("Auto-drying: printer %d skipped — model %s does not support drying", pid, model)
  5013. continue
  5014. # Check each AMS unit from raw_data
  5015. ams_list = state.raw_data.get("ams", [])
  5016. logger.debug("Auto-drying: printer %d — checking %d AMS units", pid, len(ams_list))
  5017. for ams_data in ams_list:
  5018. module_type = str(ams_data.get("module_type") or "")
  5019. ams_id = int(ams_data.get("id", 0))
  5020. # Only n3f/n3s support drying
  5021. if module_type not in ("n3f", "n3s"):
  5022. logger.debug("Auto-drying: printer %d AMS %d skipped — module_type=%s", pid, ams_id, module_type)
  5023. continue
  5024. # Resolve per-filament humidity threshold for this AMS unit (#1605).
  5025. # Most-restrictive of all loaded tray types; falls back to the
  5026. # global threshold when no overrides are configured.
  5027. trays = ams_data.get("tray", []) or []
  5028. humidity_threshold = self.resolve_humidity_threshold(
  5029. trays, per_type_thresholds, global_humidity_threshold
  5030. )
  5031. dry_time = int(ams_data.get("dry_time") or 0)
  5032. # Read humidity as a percentage. The 1-5 index is never
  5033. # substituted: it is inverted, and being unable to exceed any
  5034. # threshold it would read as "dry" forever (#3140). ``None``
  5035. # already means "skip this unit" everywhere below.
  5036. humidity_pct = ams_humidity_percent(ams_data)
  5037. humidity = int(round(humidity_pct)) if humidity_pct is not None else None
  5038. unit_key = (pid, ams_id)
  5039. unit_state = self._auto_dry_units.get(unit_key)
  5040. # Already drying — let it run to its configured duration (#1892).
  5041. #
  5042. # We deliberately do NOT stop drying from a humidity re-check here.
  5043. # Relative humidity drops steeply in heated air, so the AMS sensor
  5044. # reads ~15-20% within minutes of the dryer starting even while the
  5045. # filament is still saturated. A humidity-based early-stop therefore
  5046. # always fires at the minimum-time floor, truncating both user-started
  5047. # manual cycles and Bambuddy's own preset-duration dries to ~30 min.
  5048. # The firmware stops when the configured duration elapses; scheduling
  5049. # stops (print takes priority, queue no longer needs drying) are
  5050. # handled separately via _stop_drying().
  5051. if dry_time > 0:
  5052. if pid not in self._drying_in_progress:
  5053. # Drying we didn't start (manual or from before restart) —
  5054. # track it so scheduling stops still apply; never auto-stop it.
  5055. self._drying_in_progress[pid] = time.monotonic()
  5056. if unit_state is not None:
  5057. unit_state["running"] = True
  5058. logger.debug(
  5059. "Auto-drying: printer %d AMS %d — drying (%dm left, humidity %s%%), letting it run",
  5060. pid,
  5061. ams_id,
  5062. dry_time,
  5063. humidity,
  5064. )
  5065. continue
  5066. # Nothing is drying. Close out a cycle we armed ourselves and
  5067. # judge whether it achieved anything (#2770).
  5068. #
  5069. # "Achieved anything" is measured against the LOWEST reading any
  5070. # cycle on this unit has ended at, not against the threshold and
  5071. # not against the previous cycle. A spool that is genuinely wet
  5072. # in a humid room comes down slowly — 40, 37, 35 — and must be
  5073. # allowed to keep going for as long as it is still coming down,
  5074. # however far it still is from the threshold. What must not
  5075. # continue is a cycle that ends exactly where the last one did,
  5076. # which is the reporter's signature: 15, 15, 16, 15, forever.
  5077. # Comparing against the running minimum rather than the previous
  5078. # end is what stops a sensor oscillating by one point between two
  5079. # values from reading as progress every other cycle.
  5080. if unit_state is not None and unit_state.pop("running", False):
  5081. unit_state["ended_at"] = time.monotonic()
  5082. # The streak that armed this cycle is spent; the next one
  5083. # starts fresh (and accumulates through the re-arm cooldown
  5084. # below, so the wait overlaps the cooldown, never stacks on
  5085. # top of it).
  5086. self._auto_dry_above.pop(unit_key, None)
  5087. if humidity is not None and humidity > humidity_threshold:
  5088. best = unit_state.get("best_end_humidity")
  5089. if isinstance(best, int) and humidity < best:
  5090. # Still coming down. Keep going, however far the
  5091. # threshold still is.
  5092. unproductive = 0
  5093. else:
  5094. unproductive = int(unit_state.get("unproductive", 0)) + 1
  5095. if not isinstance(best, int) or humidity < best:
  5096. unit_state["best_end_humidity"] = humidity
  5097. unit_state["unproductive"] = unproductive
  5098. logger.info(
  5099. "Auto-drying: printer %d AMS %d — cycle ended with humidity still %d%% > "
  5100. "threshold %d%% (best so far %s%%, %d unproductive in a row)",
  5101. pid,
  5102. ams_id,
  5103. humidity,
  5104. humidity_threshold,
  5105. unit_state.get("best_end_humidity"),
  5106. unproductive,
  5107. )
  5108. # A cycle that ended at or below the threshold needs no
  5109. # counter reset here: the branch below drops the whole entry.
  5110. # Humidity below threshold — no need to start drying. This is
  5111. # also the only thing that lifts a suspension: the reading we
  5112. # gave up on has come down, so auto-drying works again and the
  5113. # unit goes back to having no history at all.
  5114. if humidity is None or humidity <= humidity_threshold:
  5115. if unit_state is not None and unit_state.get("suspended"):
  5116. logger.info(
  5117. "Auto-drying: printer %d AMS %d — humidity %s%% is back at or below the %d%% "
  5118. "threshold, resuming automatic drying",
  5119. pid,
  5120. ams_id,
  5121. humidity,
  5122. humidity_threshold,
  5123. )
  5124. # Clear the judgement, keep the clock (#2801). Dropping the
  5125. # whole entry also dropped `ended_at`, and with it the
  5126. # 30-minute cooldown -- so a reading that dips to the
  5127. # threshold as the AMS cools and comes back above it once
  5128. # warm wiped its own history and re-armed immediately. That
  5129. # oscillation is the very thing #2770's cooldown exists to
  5130. # ride out, and it is worst at exactly the margin that makes
  5131. # a unit dry repeatedly: a point or two above the threshold.
  5132. if unit_state is not None:
  5133. unit_state.pop("suspended", None)
  5134. unit_state.pop("unproductive", None)
  5135. unit_state.pop("best_end_humidity", None)
  5136. # A real below-threshold reading ends any sustained-wait
  5137. # streak (#2518) -- "continuously above" means exactly that.
  5138. # An absent reading (humidity is None) is no-information and
  5139. # leaves the streak alone; the observation-gap guard handles
  5140. # a prolonged sensor silence.
  5141. if humidity is not None:
  5142. _above = self._auto_dry_above.pop(unit_key, None)
  5143. if _above is not None and sustained_wait_active:
  5144. logger.info(
  5145. "Auto-drying: printer %d AMS %d — humidity fell back to %s%% after "
  5146. "%.0fs of the required %dm above the %d%% threshold; not drying",
  5147. pid,
  5148. ams_id,
  5149. humidity,
  5150. time.monotonic() - _above["since"],
  5151. sustained_minutes,
  5152. humidity_threshold,
  5153. )
  5154. logger.debug(
  5155. "Auto-drying: printer %d AMS %d skipped — humidity %s <= threshold %d",
  5156. pid,
  5157. ams_id,
  5158. humidity,
  5159. humidity_threshold,
  5160. )
  5161. continue
  5162. # Sustained-humidity streak (#2518): updated on every pass that
  5163. # observes the reading above the threshold, BEFORE the
  5164. # suspension/cooldown gates below -- a suspended or cooling-down
  5165. # unit still accumulates streak time, so the wait overlaps those
  5166. # gates instead of stacking after them. Inert when the feature
  5167. # is off: no entries are written, and an entry left over from a
  5168. # toggle-off is dropped so it cannot seed a stale streak later.
  5169. if sustained_wait_active:
  5170. _now = time.monotonic()
  5171. # Four missed scheduler passes, floored: a single slow pass
  5172. # must not void a streak, but the ceiling has to scale with
  5173. # the cadence or a slow loop silently restarts every streak.
  5174. _gap_ceiling = max(4 * self._check_interval, AUTO_DRY_SUSTAINED_GAP_FLOOR_SECONDS)
  5175. _above = self._auto_dry_above.get(unit_key)
  5176. if _above is None:
  5177. self._auto_dry_above[unit_key] = {"since": _now, "last": _now}
  5178. elif _now - _above["last"] > _gap_ceiling:
  5179. # Restart, and say so at the same level as the dip
  5180. # reset: a silent restart voids the streak invisibly,
  5181. # and a user who set a long wait and never gets a dry
  5182. # has no way to see why.
  5183. logger.info(
  5184. "Auto-drying: printer %d AMS %d — sustained-humidity streak restarted after a "
  5185. "%.0fs observation gap (ceiling %ds); the %dm wait starts over",
  5186. pid,
  5187. ams_id,
  5188. _now - _above["last"],
  5189. _gap_ceiling,
  5190. sustained_minutes,
  5191. )
  5192. self._auto_dry_above[unit_key] = {"since": _now, "last": _now}
  5193. else:
  5194. _above["last"] = _now
  5195. else:
  5196. self._auto_dry_above.pop(unit_key, None)
  5197. if unit_state is not None:
  5198. if unit_state.get("suspended"):
  5199. logger.debug(
  5200. "Auto-drying: printer %d AMS %d skipped — suspended, %d cycles left humidity "
  5201. "above the %d%% threshold",
  5202. pid,
  5203. ams_id,
  5204. int(unit_state.get("unproductive", 0)),
  5205. humidity_threshold,
  5206. )
  5207. continue
  5208. if int(unit_state.get("unproductive", 0)) >= AUTO_DRY_MAX_UNPRODUCTIVE_CYCLES:
  5209. unit_state["suspended"] = True
  5210. logger.warning(
  5211. "Auto-drying: printer %d AMS %d — suspending automatic drying. %d cycles in a "
  5212. "row ended with humidity at %d%%, still above the %d%% threshold. The AMS reads "
  5213. "higher while it is warm, so a threshold in that range can never be reached and "
  5214. "re-arming would loop. Raise the threshold or dry the spools off the printer.",
  5215. pid,
  5216. ams_id,
  5217. int(unit_state.get("unproductive", 0)),
  5218. humidity,
  5219. humidity_threshold,
  5220. )
  5221. await self._notify_auto_drying_suspended(
  5222. db, printer, ams_id, humidity, humidity_threshold, int(unit_state.get("unproductive", 0))
  5223. )
  5224. continue
  5225. ended_at = unit_state.get("ended_at")
  5226. if isinstance(ended_at, float) and time.monotonic() - ended_at < AUTO_DRY_REARM_COOLDOWN_SECONDS:
  5227. logger.debug(
  5228. "Auto-drying: printer %d AMS %d skipped — cooling off for %ds after the last "
  5229. "cycle before the humidity reading is worth acting on",
  5230. pid,
  5231. ams_id,
  5232. AUTO_DRY_REARM_COOLDOWN_SECONDS,
  5233. )
  5234. continue
  5235. # Check cannot-dry reasons (power constraints etc.). Sits
  5236. # ahead of the sustained wait so a unit the firmware refuses
  5237. # to dry never logs a wait it was never going to cash in.
  5238. sf_reasons = ams_data.get("dry_sf_reason", [])
  5239. if sf_reasons:
  5240. logger.debug(
  5241. "Auto-drying: printer %d AMS %d skipped — cannot dry reasons: %s",
  5242. pid,
  5243. ams_id,
  5244. sf_reasons,
  5245. )
  5246. continue
  5247. # Sustained-humidity wait (#2518): ambient-triggered starts
  5248. # wait; only a printer with a scheduled queue item pending
  5249. # keeps the instant behavior, because that drying has a real
  5250. # deadline. Mid-print is deliberately NOT exempt while ambient
  5251. # drying is on: a humidity start on a printer that happens to be
  5252. # printing is as vulnerable to a lid-open spike as one on an idle
  5253. # printer (proven live: a 2-point threshold crossing mid-print
  5254. # bought a parked 12h command). Inactive when ambient drying is
  5255. # off: print_drying then still starts mid-print cycles on its own
  5256. # (#1816, "regardless of queue state"), and those stay instant.
  5257. if sustained_wait_active and pid not in printers_with_scheduled:
  5258. _above = self._auto_dry_above.get(unit_key)
  5259. _waited = time.monotonic() - _above["since"] if _above else 0.0
  5260. if _waited < sustained_minutes * 60:
  5261. logger.debug(
  5262. "Auto-drying: printer %d AMS %d waiting — humidity %s%% above the %d%% "
  5263. "threshold for %.0fs of the required %dm",
  5264. pid,
  5265. ams_id,
  5266. humidity,
  5267. humidity_threshold,
  5268. _waited,
  5269. sustained_minutes,
  5270. )
  5271. continue
  5272. # Get conservative drying params for mixed filaments
  5273. params = self._get_conservative_drying_params(trays, module_type, presets)
  5274. if not params:
  5275. logger.debug(
  5276. "Auto-drying: printer %d AMS %d skipped — no drying-eligible filaments in trays", pid, ams_id
  5277. )
  5278. continue
  5279. temp, duration_hours, filament_type = params
  5280. # Mid-print drying: cap drying temperature to protect spools (Bambu warns
  5281. # "drying temperature must not exceed the filament's softening temperature"
  5282. # for Print While Drying). Floor at 40 degC — below that the dryer is
  5283. # ineffective and firmware will reject anyway.
  5284. if mid_print:
  5285. temp = max(40, temp - 5)
  5286. # Start drying
  5287. logger.info(
  5288. "Auto-drying: printer %d AMS %d — humidity %d%% > threshold %d%%, "
  5289. "starting %s drying at %d°C for %dh%s",
  5290. pid,
  5291. ams_id,
  5292. humidity,
  5293. humidity_threshold,
  5294. filament_type,
  5295. temp,
  5296. duration_hours,
  5297. " (mid-print)" if mid_print else "",
  5298. )
  5299. success = printer_manager.send_drying_command(
  5300. pid, ams_id, temp, duration_hours, mode=1, filament=filament_type
  5301. )
  5302. if success:
  5303. self._drying_in_progress[pid] = time.monotonic()
  5304. armed = self._auto_dry_units.setdefault(
  5305. unit_key, {"unproductive": 0, "suspended": False, "ended_at": None}
  5306. )
  5307. armed["running"] = True
  5308. async def _notify_auto_drying_suspended(
  5309. self,
  5310. db: AsyncSession,
  5311. printer: Printer,
  5312. ams_id: int,
  5313. humidity: int,
  5314. threshold: int,
  5315. cycles: int,
  5316. ) -> None:
  5317. """Tell the user auto-drying has given up on one AMS unit (#2770).
  5318. Fires once per suspension — the caller sets ``suspended`` before calling
  5319. and every later pass short-circuits on it — because the whole point is
  5320. that Bambuddy has stopped acting. Somebody whose printer sits in another
  5321. building needs that to reach them, and the hourly humidity alarm they
  5322. are already getting says the opposite of what happened here.
  5323. Never raises: a notification provider being down must not stop the
  5324. suspension itself from taking effect.
  5325. """
  5326. ams_label = f"HT-{chr(65 + (ams_id - 128))}" if ams_id >= 128 else f"AMS-{chr(65 + ams_id)}"
  5327. try:
  5328. await notification_service.on_ams_drying_suspended(
  5329. printer.id,
  5330. printer.name,
  5331. ams_label,
  5332. float(humidity),
  5333. float(threshold),
  5334. cycles,
  5335. db,
  5336. )
  5337. except Exception as e:
  5338. logger.warning("Failed to send auto-drying suspended notification: %s", e)
  5339. def forget_auto_dry_cycle(self, printer_id: int, ams_id: int) -> None:
  5340. """Stop judging the drying cycle currently on this AMS unit (#2770).
  5341. The unproductive-cycle counter exists to notice that *drying* is not
  5342. moving the humidity reading. A cycle that ended because somebody sent a
  5343. stop says nothing about that — it was cut short before it had a chance —
  5344. so counting it would suspend auto-drying for a reason that has nothing to
  5345. do with the loop the counter is there to break.
  5346. Two callers, both of them a stop Bambuddy is responsible for: the
  5347. print-takes-priority stop below, and the manual Stop button. Left
  5348. uncalled, an install that dries between queue jobs would suspend its own
  5349. auto-drying after two prints interrupted a dry — exactly the install
  5350. queue-drying exists for.
  5351. The rest of the unit's history is kept: the cooldown before re-arming
  5352. still applies, and an earlier count still stands.
  5353. """
  5354. state = self._auto_dry_units.get((printer_id, ams_id))
  5355. if state is not None:
  5356. state.pop("running", None)
  5357. state["ended_at"] = time.monotonic()
  5358. # A stopped cycle spends the streak that armed it, same as a completed
  5359. # one (#2518).
  5360. self._auto_dry_above.pop((printer_id, ams_id), None)
  5361. def _sync_drying_state(self):
  5362. """Drop printers from ``_drying_in_progress`` that are no longer drying.
  5363. One direction only: it prunes, it never adds. A printer drying without an
  5364. entry here — because the user started the cycle from Studio, the printer's
  5365. screen or Bambuddy's own manual Dry button, or because Bambuddy restarted
  5366. mid-cycle — stays unknown to the scheduler, so the "print takes priority"
  5367. stop at ``check_queue`` only ever applies to cycles Bambuddy itself began.
  5368. That is deliberate for now rather than an oversight: populating this from
  5369. telemetry would hand the scheduler authority to stop drying a user started
  5370. by hand. It also means the backend-restart case this used to claim to
  5371. handle is not handled.
  5372. """
  5373. to_remove = []
  5374. for pid in self._drying_in_progress:
  5375. state = printer_manager.get_status(pid)
  5376. if not state:
  5377. to_remove.append(pid)
  5378. continue
  5379. # Check if any AMS unit is still drying
  5380. ams_list = state.raw_data.get("ams", [])
  5381. any_drying = any(int(a.get("dry_time") or 0) > 0 for a in ams_list)
  5382. if not any_drying:
  5383. to_remove.append(pid)
  5384. for pid in to_remove:
  5385. self._drying_in_progress.pop(pid, None)
  5386. # A printer that has gone away entirely takes its per-AMS auto-drying
  5387. # history with it (#2770), so a printer deleted and re-added does not
  5388. # inherit a suspension it never earned.
  5389. for key in [k for k in self._auto_dry_units if printer_manager.get_status(k[0]) is None]:
  5390. self._auto_dry_units.pop(key, None)
  5391. # Same for sustained-wait streaks (#2518): a deleted-and-re-added
  5392. # printer starts a fresh wait, and vanished printers do not leak
  5393. # entries.
  5394. for key in [k for k in self._auto_dry_above if printer_manager.get_status(k[0]) is None]:
  5395. self._auto_dry_above.pop(key, None)
  5396. @staticmethod
  5397. def _drying_is_only_parked(printer_id: int) -> bool:
  5398. """True when every AMS unit with a drying timer on this printer is parked.
  5399. A parked timer (#2896: the command was taken but the countdown never
  5400. runs) does not end on its own, so nothing may wait on it. False when no
  5401. unit reports a timer yet -- a command just sent that the firmware has not
  5402. reported back is real drying about to begin.
  5403. """
  5404. state = printer_manager.get_status(printer_id)
  5405. units = [a for a in ((state.raw_data or {}).get("ams") or [] if state else []) if isinstance(a, dict)]
  5406. timed = []
  5407. for unit in units:
  5408. try:
  5409. if int(unit.get("dry_time") or 0) > 0:
  5410. timed.append(unit)
  5411. except (TypeError, ValueError):
  5412. continue
  5413. return bool(timed) and all(is_countdown_parked(unit) for unit in timed)
  5414. async def _drying_may_continue_through_print(self, db: AsyncSession, printer_id: int) -> bool:
  5415. """True when a running cycle can be left alone while the next print runs.
  5416. Some hardware dries happily through a print and #2758 settled that we
  5417. should not tear those cycles down: the X2D there was refusing to
  5418. *start* a job, which is a different problem, and stopping drying before
  5419. every dispatch would throw away cycles the printer was content to run.
  5420. Where the model cannot do it, or the user has not enabled it, the print
  5421. takes priority and the cycle stops -- which is what the queue_drying_block
  5422. setting has always promised in its off position.
  5423. """
  5424. if not await self._get_bool_setting(db, "print_drying_enabled"):
  5425. return False
  5426. status = printer_manager.get_status(printer_id)
  5427. return supports_drying_while_printing(
  5428. printer_manager.get_model(printer_id),
  5429. status.firmware_version if status else None,
  5430. )
  5431. async def _stop_drying(self, printer_id: int):
  5432. """Stop drying cycles Bambuddy armed on a printer (print takes priority).
  5433. Scoped to units in ``_auto_dry_units``. It used to send a stop to every
  5434. AMS reporting ``dry_time > 0``, which meant one auto-dried unit was
  5435. enough to kill a cycle the user had started by hand on a *different*
  5436. unit of the same printer (#2801). That contradicted the contract
  5437. ``_sync_drying_state`` already documents -- the entry gate deliberately
  5438. only knows about cycles Bambuddy began, so the action must not reach
  5439. past them either.
  5440. """
  5441. state = printer_manager.get_status(printer_id)
  5442. if not state:
  5443. self._drying_in_progress.pop(printer_id, None)
  5444. return
  5445. ams_list = state.raw_data.get("ams", [])
  5446. for ams_data in ams_list:
  5447. dry_time = int(ams_data.get("dry_time") or 0)
  5448. if dry_time > 0:
  5449. ams_id = int(ams_data.get("id", 0))
  5450. if (printer_id, ams_id) not in self._auto_dry_units:
  5451. logger.debug(
  5452. "Auto-drying: leaving printer %d AMS %d alone — not a cycle Bambuddy started",
  5453. printer_id,
  5454. ams_id,
  5455. )
  5456. continue
  5457. logger.info(
  5458. "Auto-drying: stopping drying on printer %d AMS %d — print takes priority",
  5459. printer_id,
  5460. ams_id,
  5461. )
  5462. printer_manager.send_drying_command(printer_id, ams_id, 0, 0, mode=0)
  5463. self.forget_auto_dry_cycle(printer_id, ams_id)
  5464. self._drying_in_progress.pop(printer_id, None)
  5465. # Scheduled manual drying (#2638) -----------------------------------
  5466. SCHEDULED_DRYING_GRACE_SECONDS = 120 # firmware needs time to report dry_time
  5467. SCHEDULED_DRYING_COMPLETE_FRACTION = 0.9 # dry_time==0 earlier than this = interrupted
  5468. async def _check_scheduled_dryings(self, db: AsyncSession):
  5469. """Dispatch due scheduled drying runs and track running ones."""
  5470. now = utcnow_naive()
  5471. # Hourly, not every pass: see SCHEDULED_DRYING_PRUNE_INTERVAL_SECONDS.
  5472. # Monotonic, so a clock adjustment cannot park the prune for hours.
  5473. since_prune = time.monotonic()
  5474. if (
  5475. self._last_scheduled_drying_prune is None
  5476. or since_prune - self._last_scheduled_drying_prune >= SCHEDULED_DRYING_PRUNE_INTERVAL_SECONDS
  5477. ):
  5478. self._last_scheduled_drying_prune = since_prune
  5479. await db.execute(
  5480. delete(ScheduledDrying).where(
  5481. ScheduledDrying.status.in_(("completed", "cancelled", "failed")),
  5482. ScheduledDrying.completed_at.is_not(None),
  5483. ScheduledDrying.completed_at < now - timedelta(days=SCHEDULED_DRYING_RETENTION_DAYS),
  5484. )
  5485. )
  5486. # Same order as the list route: with two rows due on one printer the
  5487. # earliest scheduled wins rather than whatever the DB hands back first.
  5488. result = await db.execute(
  5489. select(ScheduledDrying)
  5490. .where(ScheduledDrying.status.in_(("pending", "running")))
  5491. .order_by(ScheduledDrying.start_after.asc().nullsfirst(), ScheduledDrying.id.asc())
  5492. )
  5493. rows = list(result.scalars().all())
  5494. # Rebuild from the DB every tick so route-side cancels and completions
  5495. # show up. Auto-drying's stop-all branches check this set before
  5496. # stopping anything (#2638).
  5497. # Kept from the previous pass so a run that ended between passes — a
  5498. # cancel through the route, say — can still be released below.
  5499. previously_running = self._scheduled_drying_printer_ids
  5500. self._scheduled_drying_printer_ids = {row.printer_id for row in rows if row.status == "running"}
  5501. running_printer_ids = set(self._scheduled_drying_printer_ids)
  5502. # Model and firmware come from the printer row, not the live state.
  5503. printer_ids = {row.printer_id for row in rows}
  5504. printers_by_id: dict[int, Printer] = {}
  5505. if printer_ids:
  5506. printer_rows = await db.execute(select(Printer).where(Printer.id.in_(printer_ids)))
  5507. printers_by_id = {p.id: p for p in printer_rows.scalars()}
  5508. for row in rows:
  5509. if row.status == "running":
  5510. self._update_running_scheduled_drying(row, now)
  5511. continue
  5512. if row.start_after is not None and row.start_after > now:
  5513. continue
  5514. state = printer_manager.get_status(row.printer_id)
  5515. if not state:
  5516. row.waiting_reason = "printer_offline"
  5517. continue
  5518. # Same preflight the immediate endpoint runs. Without it the publish
  5519. # succeeds, the row goes to running, the printer ignores the command
  5520. # and the run silently cancels itself after the grace window.
  5521. printer = printers_by_id.get(row.printer_id)
  5522. unsupported = drying_preflight.check_drying_supported(
  5523. printer.model if printer else None, state.firmware_version
  5524. )
  5525. if unsupported:
  5526. row.status = "failed"
  5527. row.error_message = unsupported
  5528. row.error_code = drying_preflight.DETAIL_CODES.get(unsupported)
  5529. row.completed_at = now
  5530. logger.warning("Scheduled drying %d: %s", row.id, unsupported)
  5531. continue
  5532. if (
  5533. self._drying_in_progress.get(row.printer_id) and not self._drying_is_only_parked(row.printer_id)
  5534. ) or row.printer_id in running_printer_ids:
  5535. row.waiting_reason = "already_drying"
  5536. continue
  5537. if not self._is_printer_idle(row.printer_id, require_plate_clear=False):
  5538. row.waiting_reason = "printer_busy"
  5539. continue
  5540. target = drying_preflight.find_ams_unit(state, row.ams_id)
  5541. if target is None:
  5542. row.waiting_reason = "ams_not_found"
  5543. continue
  5544. blocking = drying_preflight.blocking_reason_codes(target)
  5545. if blocking:
  5546. # Keep the power case distinct; it needs the user to act, so the
  5547. # card can say so instead of waiting silently.
  5548. row.waiting_reason = drying_preflight.waiting_reason_for_codes(blocking)
  5549. continue
  5550. filament = drying_preflight.resolve_filament(target, row.filament)
  5551. logger.info(
  5552. "Scheduled drying %d: starting on printer %d AMS %d at %d°C for %dh",
  5553. row.id,
  5554. row.printer_id,
  5555. row.ams_id,
  5556. row.temp,
  5557. row.duration_hours,
  5558. )
  5559. success = printer_manager.send_drying_command(
  5560. row.printer_id,
  5561. row.ams_id,
  5562. row.temp,
  5563. row.duration_hours,
  5564. mode=1,
  5565. filament=filament,
  5566. rotate_tray=row.rotate_tray,
  5567. )
  5568. if success:
  5569. row.status = "running"
  5570. row.started_at = now
  5571. row.waiting_reason = None
  5572. row.filament = filament
  5573. self._drying_in_progress[row.printer_id] = time.monotonic()
  5574. self._scheduled_drying_printer_ids.add(row.printer_id)
  5575. running_printer_ids.add(row.printer_id)
  5576. else:
  5577. row.waiting_reason = "printer_offline"
  5578. # Release the printers whose run has ended. `_drying_in_progress` is
  5579. # shared with auto-drying, which prunes it in `_sync_drying_state()` —
  5580. # but that call sits behind the auto-drying enabled check, and this
  5581. # method is the one writer that runs whether auto-drying is on or not.
  5582. # With it off, nothing would ever drop the entry short of a print being
  5583. # dispatched to the same printer, so the next scheduled run would wait
  5584. # on "already_drying" forever and `queue_drying_block` would hold the
  5585. # printer's prints too. Covers a run that ended during this pass and one
  5586. # cancelled through the route between passes.
  5587. self._scheduled_drying_printer_ids = {row.printer_id for row in rows if row.status == "running"}
  5588. for printer_id in (previously_running | running_printer_ids) - self._scheduled_drying_printer_ids:
  5589. self._drying_in_progress.pop(printer_id, None)
  5590. await db.commit()
  5591. def _update_running_scheduled_drying(self, row: ScheduledDrying, now: datetime):
  5592. """Detect completion or interruption of a running scheduled drying.
  5593. The firmware reports remaining minutes in ams.dry_time; 0 means not
  5594. drying. Within the grace window after start we ignore dry_time==0
  5595. (the status lags the command). After that, dry_time==0 near the end
  5596. of the configured duration means completed. Much earlier means the
  5597. run was stopped: re-queue it if a print preempted the dryer, but a
  5598. stop while the printer is idle was deliberate, so cancel the row
  5599. rather than restart drying the user just stopped.
  5600. """
  5601. if row.started_at is None:
  5602. row.started_at = now
  5603. return
  5604. elapsed = (now - row.started_at).total_seconds()
  5605. if elapsed < self.SCHEDULED_DRYING_GRACE_SECONDS:
  5606. return
  5607. state = printer_manager.get_status(row.printer_id)
  5608. if not state:
  5609. return # offline mid-dry; resolve when it reconnects
  5610. # find_ams_unit, not a local lookup: this runs inside check_queue, so a
  5611. # throw on a malformed id would cost the whole pass including print
  5612. # dispatch, every tick.
  5613. target = drying_preflight.find_ams_unit(state, row.ams_id)
  5614. try:
  5615. dry_time = int(target.get("dry_time") or 0) if target else 0
  5616. except (TypeError, ValueError):
  5617. dry_time = 0
  5618. if dry_time > 0 and is_countdown_parked(target):
  5619. # The printer took the command but the countdown is not running
  5620. # (#2896) and will never reach 0, so the run would stay "running"
  5621. # forever. A print in progress is the likely cause (the power budget
  5622. # is spent), so re-queue it like any interruption; the next start
  5623. # waits for the printer to be idle. Parked on an idle printer is a
  5624. # refusal, not something a retry fixes. The timer itself is left on
  5625. # the printer: it may yet start once power frees up.
  5626. if not self._is_printer_idle(row.printer_id, require_plate_clear=False):
  5627. logger.info(
  5628. "Scheduled drying %d: AMS %d countdown is not running during a print; re-queued",
  5629. row.id,
  5630. row.ams_id,
  5631. )
  5632. row.status = "pending"
  5633. row.started_at = None
  5634. row.waiting_reason = "interrupted"
  5635. else:
  5636. logger.warning(
  5637. "Scheduled drying %d: printer accepted the command but AMS %d never started drying",
  5638. row.id,
  5639. row.ams_id,
  5640. )
  5641. row.status = "failed"
  5642. row.error_message = drying_preflight.DID_NOT_START_DETAIL
  5643. row.error_code = drying_preflight.DETAIL_CODES[drying_preflight.DID_NOT_START_DETAIL]
  5644. row.completed_at = now
  5645. return
  5646. if dry_time > 0:
  5647. return
  5648. if elapsed >= row.duration_hours * 3600 * self.SCHEDULED_DRYING_COMPLETE_FRACTION:
  5649. row.status = "completed"
  5650. row.completed_at = now
  5651. elif not self._is_printer_idle(row.printer_id, require_plate_clear=False):
  5652. row.status = "pending"
  5653. row.started_at = None
  5654. row.waiting_reason = "interrupted"
  5655. else:
  5656. row.status = "cancelled"
  5657. row.completed_at = now
  5658. async def _get_smart_plugs(self, db: AsyncSession, printer_id: int) -> list[SmartPlug]:
  5659. """Get all smart plugs associated with a printer."""
  5660. result = await db.execute(select(SmartPlug).where(SmartPlug.printer_id == printer_id))
  5661. return list(result.scalars().all())
  5662. @staticmethod
  5663. def _pick_power_plug(auto_on_plugs: list[SmartPlug]) -> SmartPlug:
  5664. """Pick the plug to power-cycle a printer back online with (#2629).
  5665. Only a plug flagged ``controls_printer_power`` can actually bring the
  5666. printer back; waiting for a boot on an accessory (filter fan, lights)
  5667. just burns the power-on timeout and fails the dispatch. Falls back to
  5668. the first plug when none is flagged, which is the pre-#2629 behaviour.
  5669. Callers must pass a non-empty list.
  5670. """
  5671. for plug in auto_on_plugs:
  5672. if plug.controls_printer_power:
  5673. return plug
  5674. return auto_on_plugs[0]
  5675. # Bundled defaults for preheat_filament_targets (#1468). Values are the
  5676. # chamber-temperature recommendations BambuStudio ships for the matching
  5677. # filament profile; users can override via Settings → Workflow → Preheat
  5678. # card. "default" applies when a loaded tray's normalised type isn't in
  5679. # the map (rare — Bambu RFID-tagged spools always carry a known type).
  5680. DEFAULT_PREHEAT_FILAMENT_TARGETS: dict[str, int] = {
  5681. "PLA": 0,
  5682. "PETG": 0,
  5683. "PETG-CF": 40,
  5684. "ABS": 45,
  5685. "ASA": 45,
  5686. "PA": 50,
  5687. "PA-CF": 55,
  5688. "PC": 50,
  5689. "PC-FR": 50,
  5690. "TPU": 0,
  5691. "PVA": 0,
  5692. "default": 0,
  5693. }
  5694. @classmethod
  5695. def _bundled_preheat_targets(cls) -> dict[str, int]:
  5696. """The bundled map under the same key casing a parsed one gets.
  5697. The constant is declared with a lowercase ``default`` because that is
  5698. the key the Settings editor writes and displays. Every read of the map
  5699. happens after ``str(key).upper()``, so handing the constant back as
  5700. declared broke the contract the parser documents: an install that had
  5701. never touched the setting returned a dict with no ``DEFAULT`` in it,
  5702. and the resolution loop's fallback silently found nothing. It read the
  5703. right number only because the bundled default happens to be 0 -- change
  5704. that constant and every unconfigured install would keep preheating to
  5705. zero with no way to tell why.
  5706. """
  5707. return {key.upper(): value for key, value in cls.DEFAULT_PREHEAT_FILAMENT_TARGETS.items()}
  5708. async def _get_preheat_filament_targets(self, db: AsyncSession) -> dict[str, int]:
  5709. """Parse the user-configured filament→chamber-target map, falling back
  5710. to DEFAULT_PREHEAT_FILAMENT_TARGETS on missing / malformed JSON. Keys
  5711. are uppercased and the 'default' fallback is always present in the
  5712. returned dict so the resolution loop can index it unconditionally."""
  5713. raw = await self._get_setting(db, "preheat_filament_targets")
  5714. if not raw:
  5715. return self._bundled_preheat_targets()
  5716. try:
  5717. parsed = json.loads(raw)
  5718. if not isinstance(parsed, dict):
  5719. raise ValueError("not an object")
  5720. except (json.JSONDecodeError, ValueError) as exc:
  5721. logger.warning("preheat_filament_targets unparseable, using defaults: %s", exc)
  5722. return self._bundled_preheat_targets()
  5723. # Coerce values to int; drop unparseable rows so a stray string
  5724. # doesn't crash the loop.
  5725. out: dict[str, int] = {}
  5726. for key, value in parsed.items():
  5727. try:
  5728. out[str(key).upper()] = int(value)
  5729. except (TypeError, ValueError):
  5730. continue
  5731. if "DEFAULT" not in out:
  5732. out["DEFAULT"] = self.DEFAULT_PREHEAT_FILAMENT_TARGETS["default"]
  5733. return out
  5734. @staticmethod
  5735. def _normalize_filament_type(tray_type: str) -> str:
  5736. """Reduce the printer's tray_type to a preset-lookup key. Mirrors the
  5737. existing drying-preset normalisation (split-at-space, upper-case) so
  5738. the two maps share vocabulary — "PLA Basic" → "PLA", "PA-CF" stays
  5739. "PA-CF" (no space to split on).
  5740. This is the first stage of ``_resolve_filament_key``, which goes on to
  5741. drop the suffix and consult the alias map; on its own it only decides
  5742. what the tray is called, not which row answers for it.
  5743. Indexing the split rather than testing the input: a tray_type of spaces
  5744. is truthy and splits to nothing, so the old ``if tray_type`` guard let
  5745. it through to an IndexError.
  5746. """
  5747. words = (tray_type or "").split()
  5748. return words[0].upper() if words else ""
  5749. def _target_for_tray_type(self, tray_type: str | None, targets: dict[str, int]) -> int:
  5750. """Per-filament chamber target for one tray's reported type, or 0 when
  5751. the tray is empty / RFID-less and reports no type at all.
  5752. A filled or foamed variant wants its base material's chamber when the
  5753. map has no row of its own: ASA-GF is ASA and needs ASA's 45 degrees,
  5754. not the 0 an unknown type falls to. The specific type is still tried
  5755. first, so PETG-CF and PA-CF keep the hotter rows they are listed with
  5756. (#2902).
  5757. That lookup is now the shared one, which adds the polyamide aliases on
  5758. top of the suffix it already dropped -- so PA6-CF reaches PA's row here
  5759. too, rather than the catch-all it was landing on (#3067).
  5760. """
  5761. if not self._normalize_filament_type(tray_type or ""):
  5762. return 0
  5763. key = self._resolve_filament_key(tray_type, targets)
  5764. if key is not None:
  5765. return targets[key]
  5766. # A type the map does not list at all, which is not the same as a tray
  5767. # with nothing in it -- that already returned 0 above.
  5768. return targets.get("DEFAULT", 0)
  5769. def _derive_chamber_target(
  5770. self,
  5771. printer: Printer,
  5772. targets: dict[str, int],
  5773. item: PrintQueueItem | None = None,
  5774. ) -> int:
  5775. """Chamber target for the trays this print actually loads: the max of
  5776. their per-filament targets. Returns 0 when there is nothing to read (no
  5777. status, no AMS telemetry — e.g. external-spool prints) or when every
  5778. tray considered maps to 0, and the chamber phase then short-circuits in
  5779. the main loop.
  5780. ``item`` narrows the scan to the trays named in its ``ams_mapping``.
  5781. Scanning the whole unit instead meant one ASA spool parked in the AMS
  5782. forced a 45°C chamber onto every PLA job sharing it — the full max-wait
  5783. plus soak burned ahead of each upload, on a printer whose chamber never
  5784. reaches the target anyway (#2886). An item with no usable mapping falls
  5785. back to scanning every loaded tray: that is the only signal left, and
  5786. narrowing to nothing would skip preheat on prints that genuinely need
  5787. it.
  5788. Reads from `printer_manager.get_status(...).raw_data['ams']`, which is
  5789. the same source the dispatcher uses for AMS slot mapping. Empty / RFID-
  5790. less slots have empty `tray_type` and contribute nothing. The external
  5791. spool is consulted only when the mapping names it (>= 254); it stays
  5792. out of the unnarrowed scan, so an item without a mapping derives from
  5793. the AMS alone exactly as before.
  5794. """
  5795. state = printer_manager.get_status(printer.id)
  5796. if state is None:
  5797. return 0
  5798. raw_data = state.raw_data or {}
  5799. used = _used_global_tray_ids(item)
  5800. ams_list = raw_data.get("ams")
  5801. # Older Bambu firmware nests AMS as {"ams": {"ams": [...]}} — try both.
  5802. if isinstance(ams_list, dict):
  5803. ams_list = ams_list.get("ams") or []
  5804. if not isinstance(ams_list, list):
  5805. ams_list = []
  5806. best = 0
  5807. for ams in ams_list:
  5808. if not isinstance(ams, dict):
  5809. continue
  5810. ams_id = _int_or(ams.get("id"), 0)
  5811. for tray in ams.get("tray") or []:
  5812. # A non-dict entry has never been seen from real firmware, but
  5813. # `.get` on one raises, and nothing between here and
  5814. # `_dispatch_one`'s try/finally catches it — the item would be
  5815. # left holding its dispatch claim. Preheat is best-effort by
  5816. # contract, so step over it instead.
  5817. if not isinstance(tray, dict):
  5818. continue
  5819. if used is not None and _global_tray_id(ams_id, _int_or(tray.get("id"), 0)) not in used:
  5820. continue
  5821. best = max(best, self._target_for_tray_type(tray.get("tray_type"), targets))
  5822. if used is not None and any(t >= _EXTERNAL_TRAY_ID_MIN for t in used):
  5823. for vt in raw_data.get("vt_tray") or []:
  5824. if not isinstance(vt, dict):
  5825. continue
  5826. # `_build_loaded_filaments` addresses external feeds by the id
  5827. # the firmware reports — 255 main, 254 deputy — defaulting to
  5828. # 254 when the field is absent. Same expression here so the two
  5829. # agree on which entry a mapping's 254/255 refers to.
  5830. if _int_or(vt.get("id"), _EXTERNAL_TRAY_ID_MIN) not in used:
  5831. continue
  5832. best = max(best, self._target_for_tray_type(vt.get("tray_type"), targets))
  5833. return best
  5834. def _release_keep_warm(self, pid: int) -> None:
  5835. """Release keep-warm on a printer that left the candidate set.
  5836. Publishes ``set_bed_temperature(0)`` once — but only if firmware still
  5837. reports the target we set (``entry.held_target``), so a user or
  5838. subsequent print that changed the bed target since is not clobbered.
  5839. Best-effort, never raises.
  5840. The entry is kept, not dropped, when the printer cannot be reached
  5841. right now: a printer that is briefly offline still has a hot bed, and
  5842. holding the entry is what keeps the max-duration timeout applying and
  5843. lets a later tick retry the release. Only a printer that has left the
  5844. manager entirely gives up on that, in ``_sample_chamber_temps``.
  5845. """
  5846. entry = self._keep_warm.get(pid)
  5847. if entry is None:
  5848. return
  5849. state = printer_manager.get_status(pid)
  5850. client = printer_manager.get_client(pid)
  5851. if state is None or client is None:
  5852. logger.debug(
  5853. "Queue: keep-warm release for printer %d deferred — printer unreachable, entry kept",
  5854. pid,
  5855. )
  5856. return
  5857. cur_bed_target = float((state.temperatures or {}).get("bed_target", 0) or 0)
  5858. if int(cur_bed_target) != entry.held_target:
  5859. # Someone else owns the bed now, so there is nothing of ours to
  5860. # undo and nothing left to track.
  5861. logger.info(
  5862. "Queue: keep-warm release for printer %d skipped bed-off (firmware target %d != held %d)",
  5863. pid,
  5864. int(cur_bed_target),
  5865. entry.held_target,
  5866. )
  5867. self._keep_warm.pop(pid, None)
  5868. return
  5869. try:
  5870. client.set_bed_temperature(0)
  5871. logger.info("Queue: keep-warm released for printer %d (bed → 0)", pid)
  5872. self._keep_warm.pop(pid, None)
  5873. except Exception as exc:
  5874. # Keep the entry so the next tick tries again rather than leaving
  5875. # the bed hot with nothing tracking it.
  5876. logger.warning("Queue: keep-warm release for printer %d failed: %s", pid, exc)
  5877. def _sweep_keep_warm(self, active_candidates: set[int], dispatched: set[int]) -> None:
  5878. """Release printers that dropped out of the keep-warm candidate set.
  5879. Called from ``_apply_keep_warm`` on every tick (with the current
  5880. candidate set), and from ``check_queue``'s no-pending-items early
  5881. return (with an empty candidate set) so orphaned holds still get
  5882. released when the queue empties. Also called with an empty candidate
  5883. set when any of the three gate settings toggles off, so a printer
  5884. whose feature was disabled mid-hold gets its bed released.
  5885. Printers being dispatched this tick are excluded from the bed-off
  5886. publish: ``_preheat_and_soak`` owns the bed from that tick on, so a
  5887. transient 0 in between would just churn against preheat. Ownership of
  5888. the hot bed transfers to the preheat rollback pin instead — if the
  5889. dispatch aborts before the print starts (failed upload, cancelled
  5890. item), `_rollback_preheat_pin` turns the bed off; if preheat itself
  5891. skips (e.g. the item has no bed_temperature metadata) the pin entry
  5892. is the ONLY thing standing between an aborted dispatch and a bed
  5893. left hot with no owner. A successful print start clears the pin and
  5894. the print's own gcode takes over, as usual.
  5895. """
  5896. for _pid in list(self._keep_warm):
  5897. if _pid in active_candidates:
  5898. continue
  5899. if _pid in dispatched:
  5900. handed_over = self._keep_warm.pop(_pid, None)
  5901. self._preheat_pin.setdefault(_pid, set()).add("bed")
  5902. if handed_over is not None:
  5903. self._preheat_pin_bed[_pid] = handed_over.held_target
  5904. continue
  5905. self._release_keep_warm(_pid)
  5906. async def _apply_keep_warm(
  5907. self,
  5908. db: AsyncSession,
  5909. items: list[PrintQueueItem],
  5910. dispatch_ids: list[int] | set[int],
  5911. busy_printers: set[int],
  5912. require_plate_clear: bool,
  5913. ) -> None:
  5914. """Hold the bed warm on FINISH printers whose next queued item needs chamber heat.
  5915. When a printer just finished a job (FINISH state) and the next queued
  5916. item needs chamber heating, hold the bed hot so the chamber stays warm
  5917. during the bed-clearing window. The bed is the chamber's heating
  5918. element here, not a print surface — nothing is printing during the
  5919. hold and the dispatched print's own preheat/gcode re-targets the bed —
  5920. so the hold temperature is ``queue_keep_warm_bed_temp`` (default 90°C,
  5921. chosen to sustain chamber warmth and to satisfy bed-threshold-linked
  5922. aftermarket chamber heaters), raised to the item's own parsed
  5923. bed_temperature when that is higher. Items whose archive metadata has
  5924. no bed temperature (e.g. OrcaSlicer gcode.3mf exports) therefore still
  5925. get a hold — chamber need is what gates the feature, not metadata.
  5926. Skips entirely for filaments that map to a 0°C chamber target
  5927. (PLA, PETG, etc.) — read off the trays the next item's ``ams_mapping``
  5928. names, so a hot-chamber spool it never touches does not hold the bed of
  5929. a PLA job (#2886). An item still awaiting its mapping is judged on the
  5930. whole unit, as every item was before. Printers being dispatched this
  5931. cycle are excluded:
  5932. ``_preheat_and_soak`` already handles their bed temperature.
  5933. Bounded by ``queue_keep_warm_max_minutes`` — on timeout the bed is
  5934. released to 0 and the entry is latched ``expired=True`` so
  5935. subsequent ticks neither re-engage nor re-seed the clock. Idempotent
  5936. MQTT: publish is skipped when firmware already has the target.
  5937. The release sweep runs BEFORE the engagement gate so a printer that
  5938. was owned by keep-warm still gets its bed released when any of the
  5939. three gate settings is toggled off mid-hold. The
  5940. ``check_queue`` early-return-when-no-items path also calls
  5941. ``_sweep_keep_warm`` directly to release orphaned holds.
  5942. """
  5943. dispatch_set = set(dispatch_ids)
  5944. dispatched_printers = {it.printer_id for it in items if it.id in dispatch_set and it.printer_id}
  5945. pending_printer_ids = {it.printer_id for it in items if it.printer_id}
  5946. warm_candidates = (pending_printer_ids & busy_printers) - dispatched_printers
  5947. keep_warm_enabled = await self._get_bool_setting(db, "queue_keep_bed_warm", default=False)
  5948. preheat_on = await self._get_bool_setting(db, "preheat_enabled", default=False)
  5949. gate_open = keep_warm_enabled and require_plate_clear and preheat_on
  5950. # Release sweep first — must run even when gate_open is False so a
  5951. # printer owned by keep-warm when a gate toggles off gets released.
  5952. self._sweep_keep_warm(
  5953. active_candidates=warm_candidates if gate_open else set(),
  5954. dispatched=dispatched_printers,
  5955. )
  5956. if not gate_open:
  5957. return
  5958. hold_temp = await self._get_int_setting(db, "queue_keep_warm_bed_temp", default=90)
  5959. max_hold_seconds = (
  5960. await self._get_int_setting(db, "queue_keep_warm_max_minutes", default=_KEEP_WARM_MAX_MINUTES_DEFAULT) * 60
  5961. )
  5962. now_mono = time.monotonic()
  5963. filament_targets: dict[str, int] | None = None
  5964. for pid in warm_candidates:
  5965. entry = self._keep_warm.get(pid)
  5966. # Latched-expired: max-duration timeout already fired for this
  5967. # printer. Skip until the release sweep drops the entry (i.e.
  5968. # until the printer leaves the candidate set).
  5969. if entry is not None and entry.expired:
  5970. continue
  5971. # These two guards sit ahead of the max-duration check below, so an
  5972. # engaged hold only ages out while its printer is still reachable
  5973. # and still in FINISH. That is deliberate rather than a hole: with
  5974. # no status or no client there is no M140 to send anyway, and the
  5975. # elapsed check runs off `entry.started` so it fires on the first
  5976. # tick after the printer comes back. Leaving FINISH means the plate
  5977. # was cleared, which drops the printer out of `warm_candidates` and
  5978. # hands it to `_release_keep_warm` instead. The invariant worth
  5979. # preserving if this is ever reordered: every path out of an
  5980. # engaged hold ends in a bed-off, whether by timeout or release.
  5981. state = printer_manager.get_status(pid)
  5982. if state is None or state.state != "FINISH":
  5983. continue
  5984. client = printer_manager.get_client(pid)
  5985. if client is None:
  5986. continue
  5987. next_item = next((it for it in items if it.printer_id == pid), None)
  5988. if next_item is None:
  5989. continue
  5990. # Hold temperature: the configured keep-warm temp, raised to the
  5991. # item's own bed temp when the metadata reports a higher one. A
  5992. # missing bed_temperature (Orca gcode.3mf exports parse without
  5993. # one) does NOT skip the hold — chamber need gates the feature.
  5994. archive = next_item.archive
  5995. item_bed = int(archive.bed_temperature) if archive and archive.bed_temperature else 0
  5996. bed_target = max(item_bed, hold_temp)
  5997. if bed_target <= 0:
  5998. continue
  5999. explicit = getattr(next_item, "preheat_chamber_target_override", None)
  6000. if explicit is not None:
  6001. chamber_needed = int(explicit) > 0
  6002. else:
  6003. if filament_targets is None:
  6004. filament_targets = await self._get_preheat_filament_targets(db)
  6005. printer_obj = await self._get_printer(db, pid)
  6006. chamber_needed = (
  6007. printer_obj is not None
  6008. and self._derive_chamber_target(printer_obj, filament_targets, next_item) > 0
  6009. )
  6010. if not chamber_needed:
  6011. continue
  6012. # Seed the timer on first engagement; keep it on subsequent ticks
  6013. # (never re-seed — that would defeat the max-duration cap).
  6014. if entry is None:
  6015. entry = _KeepWarmEntry(started=now_mono, held_target=bed_target)
  6016. self._keep_warm[pid] = entry
  6017. elapsed = now_mono - entry.started
  6018. if elapsed > max_hold_seconds:
  6019. # Timeout: publish bed → 0 once (if firmware still holds our
  6020. # target) and latch expired. The entry stays until the release
  6021. # sweep drops it, preventing the next tick from re-seeding.
  6022. logger.warning(
  6023. "Queue: keep-warm timeout for printer %d (held for %.0fs) — publishing bed → 0",
  6024. pid,
  6025. elapsed,
  6026. )
  6027. cur_bed_target = float((state.temperatures or {}).get("bed_target", 0) or 0)
  6028. if int(cur_bed_target) == entry.held_target:
  6029. try:
  6030. client.set_bed_temperature(0)
  6031. except Exception as exc:
  6032. logger.warning(
  6033. "Queue: keep-warm timeout bed-off failed for printer %d: %s",
  6034. pid,
  6035. exc,
  6036. )
  6037. else:
  6038. logger.info(
  6039. "Queue: keep-warm timeout for printer %d skipped bed-off (firmware target %d != held %d)",
  6040. pid,
  6041. int(cur_bed_target),
  6042. entry.held_target,
  6043. )
  6044. entry.expired = True
  6045. continue
  6046. # Idempotence: skip publish when firmware already has our target.
  6047. cur_bed_target = float((state.temperatures or {}).get("bed_target", 0) or 0)
  6048. if int(cur_bed_target) == bed_target:
  6049. entry.held_target = bed_target
  6050. continue
  6051. try:
  6052. client.set_bed_temperature(bed_target)
  6053. entry.held_target = bed_target
  6054. logger.info(
  6055. "Queue: keeping bed warm at %d°C for printer %d (FINISH, next item needs chamber heat)",
  6056. bed_target,
  6057. pid,
  6058. )
  6059. except Exception as exc:
  6060. logger.warning("Queue: keep-warm bed command failed for printer %d: %s", pid, exc)
  6061. def _sample_chamber_temps(self) -> None:
  6062. """Record a chamber temperature sample for every connected printer.
  6063. Called once per scheduler tick (every 3–30 s). Entries older than
  6064. _chamber_history_ttl are pruned on each write so the deques stay bounded.
  6065. Also evicts per-printer state whose printer_id is no longer registered
  6066. (e.g. deleted from the DB), so nothing accumulates for gone printers.
  6067. """
  6068. now = time.monotonic()
  6069. cutoff = now - self._chamber_history_ttl
  6070. statuses = printer_manager.get_all_statuses()
  6071. known_pids = set(statuses.keys())
  6072. for pid, status in statuses.items():
  6073. if status is None or not status.connected:
  6074. continue
  6075. temps = status.temperatures or {}
  6076. chamber = temps.get("chamber")
  6077. if chamber is None:
  6078. continue
  6079. hist = self._chamber_history.setdefault(pid, deque())
  6080. hist.append((now, float(chamber)))
  6081. while hist and hist[0][0] < cutoff:
  6082. hist.popleft()
  6083. # Evict state for printers that are no longer registered with the manager.
  6084. # This is the one place a keep-warm entry is dropped without releasing
  6085. # the bed: the printer is gone from the manager, so there is no client
  6086. # left to send M140 to. `_release_keep_warm` deliberately keeps entries
  6087. # for printers that are merely unreachable, which is what makes this
  6088. # the terminal case rather than a silent leak.
  6089. for pid in list(self._chamber_history):
  6090. if pid not in known_pids:
  6091. self._chamber_history.pop(pid, None)
  6092. for pid in list(self._keep_warm):
  6093. if pid not in known_pids:
  6094. logger.info(
  6095. "Queue: dropping keep-warm state for printer %d — no longer registered",
  6096. pid,
  6097. )
  6098. self._keep_warm.pop(pid, None)
  6099. for pid in list(self._preheat_pin):
  6100. if pid not in known_pids:
  6101. self._preheat_pin.pop(pid, None)
  6102. self._preheat_pin_bed.pop(pid, None)
  6103. def _chamber_soak_remaining(
  6104. self,
  6105. printer_id: int,
  6106. chamber_target: float,
  6107. soak_seconds: int,
  6108. tolerance: float = 2.0,
  6109. ) -> int:
  6110. """Return how many seconds of soak time are still needed.
  6111. Credits the time the chamber has already spent at temperature against
  6112. the configured soak. The credit may not start earlier than any of:
  6113. * **The newest sample.** Nothing recent means the printer stopped
  6114. reporting mid-observation and the chamber may have cooled unseen, so
  6115. the full soak is required. (A 2 h history whose last reading is half
  6116. an hour old is not evidence of anything — the measured cooling rate
  6117. is fast enough to cross the threshold in that time.)
  6118. * **The most recent contiguous run of samples.** A gap wider than
  6119. ``_CHAMBER_SAMPLE_MAX_GAP_SECONDS`` is a disconnect, and time on its
  6120. far side is not evidence of temperature.
  6121. * **The end of the most recent real dip below the threshold.**
  6122. A dip only counts as real once it lasts ``_CHAMBER_DIP_GRACE_SECONDS``
  6123. — see that constant for the thermal reasoning. A stray low reading is
  6124. an artifact, and treating it as cooling would discard a soak that
  6125. actually happened.
  6126. Returns ``soak_seconds`` when nothing can be credited (no history,
  6127. stale history, or the chamber is below the threshold right now) and 0
  6128. once the credited time covers the whole soak.
  6129. """
  6130. hist = self._chamber_history.get(printer_id)
  6131. if not hist:
  6132. return soak_seconds
  6133. now = time.monotonic()
  6134. newest_ts, newest_temp = hist[-1]
  6135. if now - newest_ts > _CHAMBER_SAMPLE_MAX_GAP_SECONDS:
  6136. return soak_seconds # stale — no fresh evidence to credit
  6137. threshold = chamber_target - tolerance
  6138. if newest_temp < threshold:
  6139. return soak_seconds # below target right now; nothing is soaked
  6140. samples = list(hist)
  6141. # Earliest point we have unbroken observations for.
  6142. credit_from = samples[-1][0]
  6143. for i in range(len(samples) - 1, 0, -1):
  6144. if samples[i][0] - samples[i - 1][0] > _CHAMBER_SAMPLE_MAX_GAP_SECONDS:
  6145. break
  6146. credit_from = samples[i - 1][0]
  6147. # Pull the credit forward to the end of the last significant dip. Each
  6148. # excursion is measured between the in-range readings that bracket it,
  6149. # so a lone stray sample is charged one sampling interval rather than
  6150. # zero, and the comparison errs towards calling a dip real.
  6151. i = 0
  6152. while i < len(samples):
  6153. if samples[i][1] >= threshold:
  6154. i += 1
  6155. continue
  6156. j = i
  6157. while j < len(samples) and samples[j][1] < threshold:
  6158. j += 1
  6159. # `newest_temp >= threshold` was checked above, so j is in range.
  6160. opened_at = samples[i - 1][0] if i > 0 else samples[i][0]
  6161. if samples[j][0] - opened_at >= _CHAMBER_DIP_GRACE_SECONDS:
  6162. # Credit resumes at the last below-threshold sample rather than
  6163. # the first good one after it, so a recovered dip over-credits
  6164. # by up to one sampling interval — the opposite lean to the
  6165. # bracketing above. Both are bounded by the sample cadence and
  6166. # dwarfed by the grace period, so neither is worth the extra
  6167. # arithmetic to remove.
  6168. credit_from = max(credit_from, samples[j - 1][0])
  6169. i = j
  6170. return max(0, soak_seconds - int(now - credit_from))
  6171. def notify_dispatch_cancelled(self, item_id: int) -> None:
  6172. """Tell an in-flight dispatch that its item no longer wants to print.
  6173. Called by the queue's cancel and delete routes. Those only write to the
  6174. database, which a dispatch coroutine parked in ``asyncio.sleep`` cannot
  6175. observe — so preheat would keep heating for the rest of max_wait + soak
  6176. (45 minutes at the defaults) and keep the printer in ``busy_printers``,
  6177. blocking every other queued item behind a print that is not happening.
  6178. Signalling in memory rather than re-reading the row keeps this off the
  6179. database entirely: no second session, no transaction held across a long
  6180. sleep, and no snapshot staleness deciding whether a print goes ahead.
  6181. Bambuddy serves from a single uvicorn process with one scheduler task,
  6182. so the route and the dispatch always share this object. The flag is
  6183. advisory — dropping it (e.g. after a restart) only costs a wasted
  6184. preheat, never a wrongly-abandoned print.
  6185. Only ids with a dispatch actually in flight are recorded, so the set
  6186. stays bounded by the upload pool rather than growing once per cancelled
  6187. item for the life of the process. Skipping the rest loses nothing: an
  6188. item that is not in flight cannot start heating later either, because
  6189. ``_claim_for_dispatch`` only claims rows that are still ``pending`` and
  6190. the caller has already committed a terminal status (or deleted the row)
  6191. before calling this.
  6192. """
  6193. if item_id in self._inflight:
  6194. self._cancelled_dispatches.add(item_id)
  6195. async def _preheat_sleep(self, item_id: int, seconds: float) -> bool:
  6196. """Sleep in slices, returning False as soon as the item stops wanting preheat.
  6197. A single long ``asyncio.sleep`` cannot notice a cancellation that lands
  6198. while it is parked, so the wait is chopped into
  6199. ``_PREHEAT_CANCEL_CHECK_SECONDS`` slices with a check after each.
  6200. """
  6201. remaining = float(seconds)
  6202. while remaining > 0:
  6203. slice_secs = min(_PREHEAT_CANCEL_CHECK_SECONDS, remaining)
  6204. await asyncio.sleep(slice_secs)
  6205. remaining -= slice_secs
  6206. if item_id in self._cancelled_dispatches:
  6207. return False
  6208. return True
  6209. def _preheat_flap_to_cooling(self, item_id: int, printer: Printer) -> None:
  6210. """Put the airduct flap back to cooling for a print that wants no chamber heat.
  6211. The full preheat stage does this as part of its own dispatch: an H2D
  6212. left in heating mode by the ABS job before it would otherwise cook the
  6213. PLA that follows. The skip path never reaches that code, so it calls
  6214. this instead -- one idempotent MQTT command, no waiting, and nothing to
  6215. add to the rollback pin, because a flap set to cooling for a print that
  6216. needs no heat is where it should have been either way.
  6217. Best-effort like everything else in the stage: a refused command logs
  6218. and the dispatch carries on.
  6219. """
  6220. model = printer.model or ""
  6221. if not supports_airduct(model):
  6222. return
  6223. state = printer_manager.get_status(printer.id)
  6224. current = getattr(state, "airduct_mode", None) if state else None
  6225. if current == _AIRDUCT_MODE_COOLING:
  6226. return
  6227. client = printer_manager.get_client(printer.id)
  6228. if client is None:
  6229. return
  6230. try:
  6231. client.set_airduct_mode("cooling")
  6232. except Exception as exc:
  6233. logger.warning("Queue item %s: preheat-skip airduct cooling failed: %s", item_id, exc)
  6234. async def _preheat_and_soak(
  6235. self,
  6236. db: AsyncSession,
  6237. item: PrintQueueItem,
  6238. printer: Printer,
  6239. archive: PrintArchive | None,
  6240. ) -> bool:
  6241. """Run the per-printer preheat + heat-soak stage before FTP upload (#1468).
  6242. Returns True when the dispatch should carry on to the upload — including
  6243. every case where preheat is skipped, since a skipped preheat is not a
  6244. reason to abandon the print. Returns False only when the item stopped
  6245. wanting to be printed while the stage was waiting (cancelled or
  6246. deleted); the caller must then abandon the dispatch, and
  6247. ``_dispatch_one``'s rollback shuts the heaters off on the way out.
  6248. Resolution order:
  6249. 1. `item.preheat_override` — 'off' skips entirely; 'inherit' falls back
  6250. to the global `preheat_enabled` setting; 'on' forces the stage on
  6251. even if the global is off.
  6252. 2. Chamber target — `item.preheat_chamber_target_override` if non-null;
  6253. else max of `preheat_filament_targets[normalize(t.tray_type)]`
  6254. across the trays `item.ams_mapping` names (every loaded slot when
  6255. it names none).
  6256. 3. A target of 0 off the filament map skips the whole stage: the
  6257. materials this print loads want no chamber, so there is nothing to
  6258. soak for and the bed phase would only delay the upload (#3041).
  6259. An explicit 0 typed into the per-item override, or a per-item
  6260. 'on', still runs the bed phase and the soak — both are the user
  6261. asking for a warm bed in so many words.
  6262. 4. Three hardware tiers branch the wait loop:
  6263. - Chamber heater (H2C/H2D/H2DPro/H2S/X2D/X1E via supports_chamber_heater):
  6264. send M141 to the resolved target, then wait for the chamber sensor
  6265. to reach it (or the max-wait timeout to elapse).
  6266. - Chamber sensor only (X1C/P2S via supports_chamber_temp ∧ ¬supports_chamber_heater):
  6267. no M141; the bed is the only heat source, so we wait for the chamber
  6268. sensor to rise via bed radiation OR fall through on timeout.
  6269. - No chamber sensor (P1S/P1P/A1/A1 Mini): no way to verify chamber
  6270. temperature; the function just heats the bed and holds for the
  6271. configured soak duration.
  6272. The bed target comes from the archive's parsed metadata
  6273. (`bed_temperature`); if missing the preheat stage logs and returns
  6274. without dispatching anything, rather than guessing at a default that
  6275. might wreck filament setup.
  6276. Failures are logged but never re-raised — preheat is best-effort. A
  6277. printer that goes offline mid-soak, a refused gcode command, or a
  6278. missing temperature reading must not turn into a failed queue item; the
  6279. normal upload + start path runs immediately after this method returns.
  6280. """
  6281. override = (getattr(item, "preheat_override", None) or "inherit").lower()
  6282. if override == "off":
  6283. return True
  6284. if override == "inherit":
  6285. enabled = await self._get_bool_setting(db, "preheat_enabled", default=False)
  6286. if not enabled:
  6287. return True
  6288. # override == "on" forces the stage on regardless of the global setting.
  6289. max_wait = await self._get_int_setting(db, "preheat_max_wait_seconds", default=900)
  6290. soak_seconds = await self._get_int_setting(db, "preheat_soak_seconds", default=300)
  6291. # Chamber target resolution:
  6292. # 1. Explicit per-item override beats everything (user knows best).
  6293. # 2. Otherwise derive from the filament types this print loads, via
  6294. # the per-filament target map. A PLA-only print derives 0 and the
  6295. # block below skips the stage without the user touching anything,
  6296. # even when an ASA spool is sitting in another slot of the same
  6297. # AMS (#2886).
  6298. explicit_target = getattr(item, "preheat_chamber_target_override", None)
  6299. if explicit_target is not None and explicit_target > 0:
  6300. chamber_target = int(explicit_target)
  6301. chamber_source = "item-override"
  6302. elif explicit_target == 0:
  6303. chamber_target = 0 # explicit 0 means "no chamber, even if filament wants it"
  6304. chamber_source = "item-override-zero"
  6305. else:
  6306. targets = await self._get_preheat_filament_targets(db)
  6307. chamber_target = self._derive_chamber_target(printer, targets, item)
  6308. chamber_source = "filament-map"
  6309. # Nothing to preheat *for*. A zero that came out of the filament map is
  6310. # the map saying this print's materials want no chamber conditioning --
  6311. # PLA, PETG, TPU and PVA all sit at 0 by default. Running the stage
  6312. # anyway heated the bed and then held it for the full soak, which
  6313. # delayed every PLA dispatch by minutes and bought nothing: the print's
  6314. # own G-code sets the bed the moment it starts, so preheating it here
  6315. # only moves that heating ahead of the upload instead of overlapping
  6316. # with it, and the soak has no chamber to condition (#3041).
  6317. #
  6318. # An explicit statement from the user still runs the stage. Forcing the
  6319. # per-item override to 'on', or typing a chamber target of exactly 0,
  6320. # both mean "preheat the bed for this print" -- the second is
  6321. # documented as doing precisely that. Only the automatic path, the
  6322. # global toggle plus the filament map, short-circuits here.
  6323. if chamber_target <= 0 and chamber_source == "filament-map" and override != "on":
  6324. logger.info(
  6325. "Queue item %s: preheat skipped -- the loaded filaments derive no chamber "
  6326. "target, so there is nothing to soak for (override=%s model=%s)",
  6327. item.id,
  6328. override,
  6329. printer.model or "",
  6330. )
  6331. self._preheat_flap_to_cooling(item.id, printer)
  6332. return True
  6333. bed_target = int(archive.bed_temperature) if archive and archive.bed_temperature else 0
  6334. if bed_target <= 0:
  6335. # No bed temperature in the slicer metadata. When the print needs a
  6336. # hot chamber the bed is simply how we heat it, so fall back to the
  6337. # configured chamber-heating bed temperature rather than skipping
  6338. # the whole stage — otherwise the print starts with a cold chamber,
  6339. # which is exactly what preheat exists to prevent. Without a chamber
  6340. # requirement there is nothing to preheat *for*, so skip as before
  6341. # rather than guess a bed temperature for the print itself.
  6342. if chamber_target <= 0:
  6343. logger.info(
  6344. "Queue item %s: preheat skipped — archive has no bed_temperature metadata and no chamber target",
  6345. item.id,
  6346. )
  6347. return True
  6348. bed_target = await self._get_int_setting(db, "queue_keep_warm_bed_temp", default=90)
  6349. logger.info(
  6350. "Queue item %s: archive has no bed_temperature metadata — heating the bed to "
  6351. "%d°C to drive the chamber to %d°C",
  6352. item.id,
  6353. bed_target,
  6354. chamber_target,
  6355. )
  6356. client = printer_manager.get_client(printer.id)
  6357. if client is None:
  6358. logger.warning("Queue item %s: preheat skipped — printer client unavailable", item.id)
  6359. return True
  6360. model = printer.model or ""
  6361. has_heater = supports_chamber_heater(model)
  6362. has_sensor = supports_chamber_temp(model)
  6363. do_chamber = chamber_target > 0 and (has_heater or has_sensor)
  6364. # Fast path: if the chamber has been continuously above target for at
  6365. # least soak_seconds and the bed is already at temperature, skip the
  6366. # entire preheat stage. Typical case: keep-warm held the bed between
  6367. # consecutive same-material prints and the chamber never dropped.
  6368. if do_chamber and has_sensor and soak_seconds > 0:
  6369. remaining = self._chamber_soak_remaining(printer.id, float(chamber_target), soak_seconds)
  6370. if remaining == 0:
  6371. cur = printer_manager.get_status(printer.id)
  6372. if cur:
  6373. cur_temps = cur.temperatures or {}
  6374. if (
  6375. float(cur_temps.get("bed", 0) or 0) >= bed_target - 2.0
  6376. and float(cur_temps.get("chamber", 0) or 0) >= chamber_target - 2.0
  6377. ):
  6378. logger.info(
  6379. "Queue item %s: preheat skipped — chamber has been above %d°C for ≥%ds "
  6380. "and bed is already at temperature (chamber history fast-path)",
  6381. item.id,
  6382. chamber_target,
  6383. soak_seconds,
  6384. )
  6385. # Still set targets to prevent cooling during the 3MF upload window.
  6386. # Register each successful set in the preheat pin so `_dispatch_one`
  6387. # unwinds them on any non-success exit.
  6388. pin = self._preheat_pin.setdefault(printer.id, set())
  6389. try:
  6390. client.set_bed_temperature(bed_target)
  6391. pin.add("bed")
  6392. self._preheat_pin_bed[printer.id] = bed_target
  6393. except Exception as exc:
  6394. logger.warning("Queue item %s: fast-path bed M140 failed: %s", item.id, exc)
  6395. if supports_airduct(model):
  6396. cur_airduct = getattr(cur, "airduct_mode", None)
  6397. if cur_airduct != _AIRDUCT_MODE_HEATING:
  6398. try:
  6399. client.set_airduct_mode("heating")
  6400. # Only undo what we can see we replaced. `None`
  6401. # means no mode has been observed yet, and
  6402. # rolling that back to cooling would assert a
  6403. # state the printer never reported.
  6404. if cur_airduct == _AIRDUCT_MODE_COOLING:
  6405. pin.add("airduct")
  6406. except Exception as exc:
  6407. logger.warning("Queue item %s: fast-path airduct failed: %s", item.id, exc)
  6408. if has_heater:
  6409. try:
  6410. client.set_chamber_temperature(chamber_target)
  6411. pin.add("chamber")
  6412. except Exception as exc:
  6413. logger.warning("Queue item %s: fast-path chamber M141 failed: %s", item.id, exc)
  6414. return True
  6415. logger.info(
  6416. "Queue item %s: preheat starting — bed=%d°C chamber_target=%d°C (source=%s override=%s "
  6417. "model=%s has_heater=%s has_sensor=%s) max_wait=%ds soak=%ds",
  6418. item.id,
  6419. bed_target,
  6420. chamber_target if do_chamber else 0,
  6421. chamber_source,
  6422. override,
  6423. model,
  6424. has_heater,
  6425. has_sensor,
  6426. max_wait,
  6427. soak_seconds,
  6428. )
  6429. # Preheat rollback registry: everything we set below is recorded here so
  6430. # `_dispatch_one`'s finally clause can unwind the whole heating regime
  6431. # (bed off, chamber off, airduct back to cooling) on any non-success
  6432. # exit. Populated as each command succeeds; consumed and cleared by
  6433. # `_dispatch_one`.
  6434. pin = self._preheat_pin.setdefault(printer.id, set())
  6435. # Dispatch heaters. set_bed_temperature / set_chamber_temperature already
  6436. # cache the target locally so the polling reads below see consistent
  6437. # state (firmware MQTT echoes lag by ~1s).
  6438. try:
  6439. client.set_bed_temperature(bed_target)
  6440. pin.add("bed")
  6441. self._preheat_pin_bed[printer.id] = bed_target
  6442. except Exception as exc:
  6443. logger.warning("Queue item %s: preheat bed M140 failed: %s", item.id, exc)
  6444. return True
  6445. # Airduct mode (#1468 follow-up). Models with the cooling/heating flap
  6446. # (H2C/H2D/H2D Pro/H2S/X2D/P2S) keep the flap whatever the user last
  6447. # left it on, regardless of M141. Default cooling actively vents the
  6448. # chamber, so a `chamber_target > 0` print with the flap stuck in
  6449. # cooling never converges — the heater fights the open exhaust. We
  6450. # flip the flap BEFORE M141 to "heating" when the preheat wants
  6451. # chamber heat, and back to "cooling" when it doesn't (PLA-only print
  6452. # on an H2D that was previously running ABS would otherwise stay in
  6453. # heating mode and overheat PLA). The current-state read keeps the
  6454. # command idempotent — no MQTT chatter when the flap is already where
  6455. # we want it.
  6456. if supports_airduct(model):
  6457. desired_airduct = "heating" if chamber_target > 0 else "cooling"
  6458. desired_id = _AIRDUCT_MODE_HEATING if desired_airduct == "heating" else _AIRDUCT_MODE_COOLING
  6459. current_state = printer_manager.get_status(printer.id)
  6460. current_airduct = getattr(current_state, "airduct_mode", None) if current_state else None
  6461. if current_airduct != desired_id:
  6462. try:
  6463. client.set_airduct_mode(desired_airduct)
  6464. # As in the fast path: only pin a rollback for a flap we
  6465. # saw in cooling. `current_airduct` of None means no mode
  6466. # has been observed, so there is nothing to restore to.
  6467. if desired_airduct == "heating" and current_airduct == _AIRDUCT_MODE_COOLING:
  6468. pin.add("airduct")
  6469. except Exception as exc:
  6470. logger.warning(
  6471. "Queue item %s: preheat airduct %s mode failed: %s",
  6472. item.id,
  6473. desired_airduct,
  6474. exc,
  6475. )
  6476. if do_chamber and has_heater:
  6477. try:
  6478. client.set_chamber_temperature(chamber_target)
  6479. pin.add("chamber")
  6480. except Exception as exc:
  6481. logger.warning("Queue item %s: preheat chamber M141 failed: %s", item.id, exc)
  6482. # Release the pooled DB connection before the (potentially many-minute)
  6483. # heat-soak wait below (#2572). Every setting this method needs is read
  6484. # above; the wait/soak loop only polls printer_manager state and sleeps —
  6485. # it never touches the DB. Without this the caller's transaction sat
  6486. # "idle in transaction" for the whole soak, pinning one pooled connection
  6487. # per preheating printer. expire_on_commit=False keeps item/printer
  6488. # readable afterwards; there are no pending writes to lose here.
  6489. await db.commit()
  6490. # Wait for convergence. Bed warm-up is fast (~5 min from cold); chamber
  6491. # via M141 takes a few minutes; chamber via bed radiation can take 20+.
  6492. # Poll every 3s — frequent enough for responsive logging without
  6493. # spamming the MQTT state stream. The "converged" predicate is:
  6494. # bed reached target (within 2°C tolerance for floating-point + heater hysteresis),
  6495. # AND
  6496. # chamber phase satisfied (no chamber phase, no sensor, or sensor reached target).
  6497. BED_TOLERANCE = 2.0
  6498. CHAMBER_TOLERANCE = 2.0
  6499. POLL_INTERVAL = 3.0
  6500. deadline = asyncio.get_event_loop().time() + max_wait
  6501. while True:
  6502. state = printer_manager.get_status(printer.id)
  6503. if state is None:
  6504. logger.warning("Queue item %s: preheat lost state during wait", item.id)
  6505. break
  6506. temps = state.temperatures or {}
  6507. bed_now = float(temps.get("bed", 0) or 0)
  6508. chamber_now = float(temps.get("chamber", 0) or 0)
  6509. bed_ok = bed_now >= bed_target - BED_TOLERANCE
  6510. if not do_chamber:
  6511. chamber_ok = True # phase disabled or model has neither sensor nor heater
  6512. elif not has_sensor:
  6513. chamber_ok = True # P1S etc — can't read, rely on soak timer only
  6514. else:
  6515. chamber_ok = chamber_now >= chamber_target - CHAMBER_TOLERANCE
  6516. if bed_ok and chamber_ok:
  6517. logger.info(
  6518. "Queue item %s: preheat target reached (bed=%.1f chamber=%.1f) — entering soak",
  6519. item.id,
  6520. bed_now,
  6521. chamber_now,
  6522. )
  6523. break
  6524. if asyncio.get_event_loop().time() >= deadline:
  6525. logger.info(
  6526. "Queue item %s: preheat max_wait reached (bed=%.1f/%d chamber=%.1f/%d) — falling through to soak",
  6527. item.id,
  6528. bed_now,
  6529. bed_target,
  6530. chamber_now,
  6531. chamber_target if do_chamber else 0,
  6532. )
  6533. break
  6534. if not await self._preheat_sleep(item.id, POLL_INTERVAL):
  6535. logger.info(
  6536. "Queue item %s: preheat aborted — item cancelled or deleted while waiting for temperature",
  6537. item.id,
  6538. )
  6539. return False
  6540. if soak_seconds > 0:
  6541. if do_chamber and has_sensor:
  6542. remaining = self._chamber_soak_remaining(printer.id, float(chamber_target), soak_seconds)
  6543. else:
  6544. remaining = soak_seconds # no sensor — can't verify history, run full soak
  6545. if remaining > 0:
  6546. logger.info(
  6547. "Queue item %s: preheat soak — holding for %ds (of %ds configured; chamber "
  6548. "has been above target for ~%ds already)",
  6549. item.id,
  6550. remaining,
  6551. soak_seconds,
  6552. soak_seconds - remaining,
  6553. )
  6554. if not await self._preheat_sleep(item.id, remaining):
  6555. logger.info(
  6556. "Queue item %s: preheat aborted — item cancelled or deleted during soak",
  6557. item.id,
  6558. )
  6559. return False
  6560. else:
  6561. logger.info(
  6562. "Queue item %s: preheat soak skipped — chamber has been above %d°C for ≥%ds",
  6563. item.id,
  6564. chamber_target,
  6565. soak_seconds,
  6566. )
  6567. logger.info("Queue item %s: preheat complete — proceeding to upload", item.id)
  6568. return True
  6569. async def _power_on_and_wait(self, plug: SmartPlug, printer_id: int, db: AsyncSession) -> bool:
  6570. """Turn on smart plug and wait for printer to connect.
  6571. Returns True if printer connected successfully within timeout.
  6572. """
  6573. # Get the appropriate service for the plug type (Tasmota or Home Assistant)
  6574. service = await smart_plug_manager.get_service_for_plug(plug, db)
  6575. # Check current plug state
  6576. status = await service.get_status(plug)
  6577. if not status.get("reachable"):
  6578. logger.warning("Smart plug '%s' is not reachable", plug.name)
  6579. return False
  6580. # Turn on if not already on
  6581. if status.get("state") != "ON":
  6582. success = await service.turn_on(plug)
  6583. if not success:
  6584. logger.warning("Failed to turn on smart plug '%s'", plug.name)
  6585. return False
  6586. logger.info("Powered on smart plug '%s' for printer %s", plug.name, printer_id)
  6587. # Get printer from database for connection
  6588. result = await db.execute(select(Printer).where(Printer.id == printer_id))
  6589. printer = result.scalar_one_or_none()
  6590. if not printer:
  6591. logger.error("Printer %s not found in database", printer_id)
  6592. return False
  6593. # Wait for printer to boot (give it some time before trying to connect)
  6594. logger.info("Waiting 30s for printer %s to boot...", printer_id)
  6595. await asyncio.sleep(30)
  6596. # Try to connect to the printer periodically
  6597. elapsed = 30 # Already waited 30s
  6598. while elapsed < self._power_on_wait_time:
  6599. # Try to connect
  6600. logger.info("Attempting to connect to printer %s...", printer_id)
  6601. try:
  6602. connected = await printer_manager.connect_printer(printer)
  6603. if connected:
  6604. logger.info("Printer %s connected after %ss", printer_id, elapsed)
  6605. # Give it a moment to stabilize and get status
  6606. await asyncio.sleep(5)
  6607. return True
  6608. except Exception as e:
  6609. logger.debug("Connection attempt failed: %s", e)
  6610. await asyncio.sleep(self._power_on_check_interval)
  6611. elapsed += self._power_on_check_interval
  6612. logger.debug("Waiting for printer %s to connect... (%ss)", printer_id, elapsed)
  6613. logger.warning("Printer %s did not connect within %ss after power on", printer_id, self._power_on_wait_time)
  6614. return False
  6615. async def _check_previous_success(self, db: AsyncSession, item: PrintQueueItem) -> bool:
  6616. """Check if the previous print on this printer succeeded.
  6617. A user-cancelled predecessor is treated as neutral — `cancelled` is a
  6618. deliberate action, not a failure, so subsequent items should still
  6619. dispatch (#1667). `skipped` is excluded from the lookback entirely:
  6620. a skip isn't an actual print attempt, so it must not gate downstream
  6621. items — counting it as a failed predecessor was the cascade bug that
  6622. let a single cancellation block 18 items over 3 days for the reporter.
  6623. Only `failed` and `aborted` — real print-attempt failures — block.
  6624. Failures with `gate_acknowledged=True` (set by the per-printer Resume
  6625. action — #1818) are also excluded from the lookback so the user can
  6626. clear the gate after fixing the physical issue without having to
  6627. re-queue every downstream job.
  6628. """
  6629. result = await db.execute(
  6630. select(PrintQueueItem)
  6631. .where(PrintQueueItem.printer_id == item.printer_id)
  6632. .where(PrintQueueItem.id != item.id)
  6633. .where(PrintQueueItem.status.in_(["completed", "failed", "cancelled", "aborted"]))
  6634. .where(PrintQueueItem.gate_acknowledged == False) # noqa: E712
  6635. .order_by(PrintQueueItem.completed_at.desc())
  6636. .limit(1)
  6637. )
  6638. prev_item = result.scalar_one_or_none()
  6639. # If no previous item, assume success (first in queue)
  6640. if not prev_item:
  6641. return True
  6642. return prev_item.status in ("completed", "cancelled")
  6643. async def _repoint_siblings_at_archive(
  6644. self,
  6645. db: AsyncSession,
  6646. *,
  6647. consumed_library_file_id: int,
  6648. archive_id: int,
  6649. dispatched_item_id: int,
  6650. ) -> int:
  6651. """Move the other queue items off a library row that is about to be deleted (#2819).
  6652. ``cleanup_library_after_dispatch`` consumes the library row: the printer-card
  6653. upload-and-print flow uploads a transient file, prints it, and deletes it.
  6654. Queue creation happily puts that flag on every copy of a ``quantity > 1``
  6655. request, and ``_clone_queue_item`` copies ``library_file_id`` onto batch
  6656. clones, so the first dispatch could pull the file out from under rows that
  6657. had not run yet. What those rows did next depended on the database, and
  6658. neither answer was right: SQLite ships with ``PRAGMA foreign_keys`` off, so
  6659. the ``ON DELETE CASCADE`` on ``print_queue.library_file_id`` never fired and
  6660. they were left pointing at a row that no longer existed, failing with
  6661. "Library file not found" whenever someone started them -- or sitting
  6662. ``pending`` forever under ``manual_start``. PostgreSQL enforces the same
  6663. constraint, so there the rows were deleted outright and the queued copies
  6664. simply vanished, with no error and no history.
  6665. The archive holds its own copy of the 3MF, so the remaining copies can print
  6666. from it instead. Two things happen here, and both must happen *before* the
  6667. delete -- afterwards there is nothing left to repair on PostgreSQL:
  6668. * every row still naming the file has ``library_file_id`` cleared, which is
  6669. what takes it out of the cascade's reach. That covers rows this cannot
  6670. re-point as well -- a copy already printing from its own archive, and the
  6671. finished ones, which are not spare parts but the record a batch order
  6672. counts its progress from.
  6673. * the rows that still need something to print are pointed at the archive.
  6674. Returns how many items were re-pointed.
  6675. """
  6676. variant_item_ids = (
  6677. (
  6678. await db.execute(
  6679. select(PrintQueueVariant.queue_item_id).where(
  6680. PrintQueueVariant.library_file_id == consumed_library_file_id
  6681. )
  6682. )
  6683. )
  6684. .scalars()
  6685. .all()
  6686. )
  6687. # Candidate rows naming the consumed file have to go rather than be
  6688. # cleared: `library_file_id` is NOT NULL there, so there is no way to keep
  6689. # one out of the cascade. This is what PostgreSQL already does, and
  6690. # _candidates_for skips such a variant on SQLite anyway, so no selection
  6691. # outcome changes -- the two backends simply stop disagreeing about
  6692. # whether the row is still there.
  6693. #
  6694. # It also has to happen before the re-point below: a cross-model item
  6695. # (#671) picks a variant every pass and folds it onto the row, and
  6696. # _resolve_variant clears archive_id as it does so, which would undo the
  6697. # re-point on the very next lap.
  6698. await db.execute(delete(PrintQueueVariant).where(PrintQueueVariant.library_file_id == consumed_library_file_id))
  6699. # An item left with other candidates still has somewhere to go, and those
  6700. # carry their own target model -- pointing it at this archive would print a
  6701. # file the matcher never chose. It is excluded from the re-point and simply
  6702. # re-resolves against what is left.
  6703. surviving_variant_item_ids = set(
  6704. (
  6705. await db.execute(
  6706. select(PrintQueueVariant.queue_item_id).where(
  6707. PrintQueueVariant.queue_item_id.in_(variant_item_ids) if variant_item_ids else false()
  6708. )
  6709. )
  6710. )
  6711. .scalars()
  6712. .all()
  6713. )
  6714. repoint_ids = set(
  6715. (
  6716. await db.execute(
  6717. select(PrintQueueItem.id)
  6718. .where(PrintQueueItem.id != dispatched_item_id)
  6719. .where(PrintQueueItem.archive_id.is_(None))
  6720. # "skipped" belongs here with the two live states: it is not
  6721. # terminal. Clearing a printer's previous-success gate puts
  6722. # every item skipped by it back to "pending"
  6723. # (resume_after_failure), and one restored onto a deleted
  6724. # file is the same orphan by a slower route. "failed",
  6725. # "cancelled", "aborted" and "completed" never return.
  6726. .where(PrintQueueItem.status.in_(("pending", "printing", "skipped")))
  6727. .where(
  6728. PrintQueueItem.id.notin_(surviving_variant_item_ids) if surviving_variant_item_ids else true()
  6729. )
  6730. .where(
  6731. or_(
  6732. PrintQueueItem.library_file_id == consumed_library_file_id,
  6733. PrintQueueItem.id.in_(variant_item_ids) if variant_item_ids else false(),
  6734. )
  6735. )
  6736. )
  6737. )
  6738. .scalars()
  6739. .all()
  6740. )
  6741. if repoint_ids:
  6742. await db.execute(
  6743. update(PrintQueueItem)
  6744. .where(PrintQueueItem.id.in_(repoint_ids))
  6745. .values(
  6746. archive_id=archive_id,
  6747. library_file_id=None,
  6748. # The file this flag named is already consumed. Leaving it set
  6749. # would arm every re-pointed copy to delete whatever library
  6750. # row it is next given.
  6751. cleanup_library_after_dispatch=False,
  6752. )
  6753. )
  6754. logger.info(
  6755. "Queue items %s: re-pointed at archive %s -- library file %s was consumed by item %s",
  6756. sorted(repoint_ids),
  6757. archive_id,
  6758. consumed_library_file_id,
  6759. dispatched_item_id,
  6760. )
  6761. # Everything else that still names the file: taken out of the cascade's
  6762. # reach without touching what it prints. The file is gone either way; what
  6763. # this preserves is the row.
  6764. await db.execute(
  6765. update(PrintQueueItem)
  6766. .where(PrintQueueItem.id != dispatched_item_id)
  6767. .where(PrintQueueItem.library_file_id == consumed_library_file_id)
  6768. .values(library_file_id=None)
  6769. )
  6770. return len(repoint_ids)
  6771. async def _power_off_if_needed(self, db: AsyncSession, item: PrintQueueItem):
  6772. """Schedule power-off if the queue item enabled auto_off_after.
  6773. Delegates to the smart-plug manager so the off honours each plug's
  6774. configured strategy (time delay or temperature threshold), is cancelled
  6775. if the printer starts printing again, and never cuts power on a loaded
  6776. print (#1890). Previously this hardcoded a 50°C / 600s cooldown wait and
  6777. powered off on the timeout regardless of print state.
  6778. """
  6779. if not item.auto_off_after:
  6780. return
  6781. try:
  6782. await smart_plug_manager.schedule_off_after_queue_job(item.printer_id, db)
  6783. except Exception as e:
  6784. logger.warning("Auto-off: Failed to schedule power-off for printer %s: %s", item.printer_id, e)
  6785. async def _get_job_name(self, db: AsyncSession, item: PrintQueueItem) -> str:
  6786. """Get a human-readable name for a queue item."""
  6787. if item.archive_id:
  6788. result = await db.execute(select(PrintArchive).where(PrintArchive.id == item.archive_id))
  6789. archive = result.scalar_one_or_none()
  6790. if archive:
  6791. return archive.filename.replace(".gcode.3mf", "").replace(".3mf", "")
  6792. if item.library_file_id:
  6793. result = await db.execute(LibraryFile.active().where(LibraryFile.id == item.library_file_id))
  6794. library_file = result.scalar_one_or_none()
  6795. if library_file:
  6796. return library_file.filename.replace(".gcode.3mf", "").replace(".3mf", "")
  6797. # A cross-model item (#671) holds no file of its own until a printer is
  6798. # picked, so name it after its first candidate — otherwise every waiting
  6799. # notification for one reads "Job #12". Queried rather than read off
  6800. # item.variants because callers outside the selection loop have not
  6801. # eager-loaded them, and a lazy load raises in async.
  6802. first_variant_name = (
  6803. await db.execute(
  6804. select(LibraryFile.filename)
  6805. .join(PrintQueueVariant, PrintQueueVariant.library_file_id == LibraryFile.id)
  6806. .where(PrintQueueVariant.queue_item_id == item.id)
  6807. .order_by(PrintQueueVariant.position, PrintQueueVariant.id)
  6808. .limit(1)
  6809. )
  6810. ).scalar_one_or_none()
  6811. if first_variant_name:
  6812. return first_variant_name.replace(".gcode.3mf", "").replace(".3mf", "")
  6813. return f"Job #{item.id}"
  6814. async def _get_printer(self, db: AsyncSession, printer_id: int) -> Printer | None:
  6815. """Get printer by ID."""
  6816. result = await db.execute(select(Printer).where(Printer.id == printer_id))
  6817. return result.scalar_one_or_none()
  6818. async def _notify_dispatch_gave_up(
  6819. self,
  6820. queue_item_id: int,
  6821. printer_id: int,
  6822. created_by_id: int | None,
  6823. reason: str = "Printer accepted the file but never started printing",
  6824. ) -> None:
  6825. """Tell the user the queue item was failed after exhausting its dispatch retries.
  6826. Called from the watchdog, which is a background task with no session of
  6827. its own — hence the fresh one here. Best-effort throughout: the row is
  6828. already marked failed and that is the load-bearing part; a notification
  6829. provider being down must not resurrect the retry loop we just stopped.
  6830. ``reason`` defaults to the exhausted-retries wording. The command-rejected
  6831. path passes its own, because "accepted the file but never started" is the
  6832. opposite of what happened there — the printer refused it outright (#2732).
  6833. """
  6834. try:
  6835. async with async_session() as db:
  6836. item = await db.get(PrintQueueItem, queue_item_id)
  6837. if not item:
  6838. return
  6839. job_name = await self._get_job_name(db, item)
  6840. printer = await self._get_printer(db, printer_id)
  6841. await notification_service.on_queue_job_failed(
  6842. job_name=job_name,
  6843. printer_id=printer_id,
  6844. printer_name=printer.name if printer else "Unknown",
  6845. reason=reason,
  6846. db=db,
  6847. )
  6848. except Exception as e:
  6849. logger.warning("Queue item %s: give-up notification failed: %s", queue_item_id, e)
  6850. try:
  6851. await ws_manager.send_queue_item_failed(
  6852. user_id=created_by_id,
  6853. queue_item_id=queue_item_id,
  6854. printer_id=printer_id,
  6855. reason="never_started",
  6856. )
  6857. except Exception:
  6858. pass # toast is best-effort
  6859. async def _block_on_filament_deficit(
  6860. self,
  6861. db: AsyncSession,
  6862. item: PrintQueueItem,
  6863. ) -> bool:
  6864. """Promote the item to manual_start when the assigned spool is short (#1496).
  6865. Returns True when this dispatch attempt was blocked, False when the
  6866. item is clear to start. A previously-flagged item whose spool has
  6867. since been swapped to one with enough material clears the flag here
  6868. so the next scheduler tick dispatches it.
  6869. """
  6870. # User has explicitly acknowledged the deficit ("Print Anyway") —
  6871. # don't re-flag, don't even compute. Without this short-circuit the
  6872. # scheduler bounces between "user said anyway" (route clears
  6873. # manual_start) and "scheduler re-blocked" (this method re-flags it
  6874. # on identical spool state) (#1698-followup).
  6875. if item.skip_filament_check:
  6876. # #1762 diagnostic: surface the short-circuit at INFO so a
  6877. # future "Print Anyway didn't work" report (e.g. issue #1762
  6878. # comment 3) has actionable evidence in the support bundle
  6879. # without needing DEBUG enabled.
  6880. logger.info(
  6881. "Queue item %s honouring user's Print Anyway acknowledgement — skipping deficit check",
  6882. item.id,
  6883. )
  6884. return False
  6885. try:
  6886. deficit = await compute_deficit_for_queue_item(db, item)
  6887. except Exception as e:
  6888. # Never let a flaky deficit check wedge the queue — log and let
  6889. # dispatch proceed. The PrintModal-side check still runs on the
  6890. # manual paths.
  6891. logger.warning("Filament deficit check failed for item %s: %s", item.id, e)
  6892. return False
  6893. if deficit:
  6894. item.filament_short = True
  6895. item.manual_start = True
  6896. await db.commit()
  6897. job_name = await self._get_job_name(db, item)
  6898. printer = await self._get_printer(db, item.printer_id) if item.printer_id else None
  6899. logger.info(
  6900. "Queue item %s blocked on filament deficit (%d slot(s)) — promoted to manual_start",
  6901. item.id,
  6902. len(deficit),
  6903. )
  6904. try:
  6905. await notification_service.on_queue_job_waiting(
  6906. job_name=job_name,
  6907. target_model=(printer.model if printer else "") or "",
  6908. waiting_reason="filament_short",
  6909. db=db,
  6910. )
  6911. except Exception as e:
  6912. logger.debug("filament_short notification failed for item %s: %s", item.id, e)
  6913. return True
  6914. # No deficit — clear any stale flag from a previous tick.
  6915. if item.filament_short:
  6916. item.filament_short = False
  6917. await db.commit()
  6918. return False
  6919. async def _block_on_unmatched_filament(
  6920. self,
  6921. db: AsyncSession,
  6922. item: PrintQueueItem,
  6923. release_assignment: bool = False,
  6924. ) -> bool:
  6925. """Promote to manual_start when a slot the plate prints has no tray (#2799).
  6926. The matcher leaves a requirement it cannot satisfy at ``-1`` and dispatch
  6927. goes ahead, letting the printer choose — which is how a job prints in the
  6928. wrong material without anyone being asked. Holding is the same answer the
  6929. deficit gate already gives for "the spool is too light", so it reuses the
  6930. same promote-and-notify machinery.
  6931. Keyed off unresolved slots in the *computed* mapping intersected with the
  6932. plate's own requirement list. Intersecting is what makes ``-1`` safe to
  6933. read: on its own it also pads slots this plate does not print, so the
  6934. array alone would hold perfectly good jobs. Reading the mapping rather
  6935. than the printer's loaded filament types also inherits the matcher's
  6936. per-nozzle restriction for free — a dual-nozzle printer carrying the
  6937. filament on the other nozzle's AMS has it "loaded" but unusable, and a
  6938. type-only scan would wave that through.
  6939. A mapping that resolved *nothing* is the same finding arriving as an
  6940. absence. ``_ensure_ams_mapping`` clears a rejected mapping its recompute
  6941. could not replace, and dispatch then goes out as ``use_ams`` with no
  6942. table at all — several megabytes uploaded for an 0700_8012 rejection.
  6943. That is held too, but only once live status positively reports loaded
  6944. trays: #2589's mapping is bogus precisely because the AMS was not known
  6945. yet, and its empty loaded list is that ignorance rather than a miss.
  6946. ``release_assignment`` is for the model-based path, which commits its
  6947. printer choice before the gates run. Holding an "any P2S" job would
  6948. otherwise pin it to the one P2S that could not run it.
  6949. Returns True when this dispatch attempt was blocked.
  6950. """
  6951. if item.skip_filament_check or not item.printer_id:
  6952. return False
  6953. if item.ams_mapping:
  6954. try:
  6955. mapping = json.loads(item.ams_mapping)
  6956. except (json.JSONDecodeError, TypeError):
  6957. return False
  6958. if not isinstance(mapping, list):
  6959. return False
  6960. else:
  6961. # Nothing resolved, or nothing survived revalidation. Only a printer
  6962. # that reported loaded trays makes that a finding — see the
  6963. # docstring on why an empty list is not one.
  6964. status = printer_manager.get_status(item.printer_id)
  6965. if status is None or not self._build_loaded_filaments(status):
  6966. return False
  6967. # Every required slot reads unresolved against an empty mapping.
  6968. mapping = []
  6969. required = await self._get_filament_requirements(db, item)
  6970. if not required:
  6971. return False
  6972. self._apply_filament_overrides(item, required)
  6973. unmatched = _unresolved_required(required, mapping)
  6974. if not unmatched:
  6975. # No cleanup needed here: once start is pressed the item is no
  6976. # longer staged, and whichever exit runs next overwrites the reason
  6977. # (`hold_item` on the fixed-printer branch, the assignment on the
  6978. # model-based one), as does dispatch.
  6979. return False
  6980. wanted = ", ".join(_describe_filament(r, "nozzle_id") for r in unmatched)
  6981. held_by = item.printer_id
  6982. item.manual_start = True
  6983. # Human-readable: this renders on the queue row.
  6984. item.waiting_reason = f"{_UNMATCHED_HOLD_PREFIX}{wanted}"
  6985. if release_assignment:
  6986. # "Any P2S" means any, so the job returns to the pool rather than
  6987. # waiting on the printer that could not take it. The mapping goes
  6988. # with the assignment: its tray IDs were resolved against the
  6989. # printer being released and mean nothing on the next one (#2799),
  6990. # and keeping them would hold the job again even on a printer that
  6991. # has the filament, since an unresolved slot is deliberately not a
  6992. # conflict worth recomputing over.
  6993. item.printer_id = None
  6994. item.ams_mapping = None
  6995. await db.commit()
  6996. job_name = await self._get_job_name(db, item)
  6997. printer = await self._get_printer(db, held_by)
  6998. logger.info(
  6999. "Queue item %s blocked — printer %s has nothing loaded for %s; promoted to manual_start",
  7000. item.id,
  7001. held_by,
  7002. wanted,
  7003. )
  7004. try:
  7005. await notification_service.on_queue_job_waiting(
  7006. job_name=job_name,
  7007. target_model=(printer.model if printer else "") or "",
  7008. waiting_reason=f"needs {wanted}",
  7009. db=db,
  7010. )
  7011. except Exception as e:
  7012. logger.debug("filament_missing notification failed for item %s: %s", item.id, e)
  7013. return True
  7014. async def _propagate_owner_to_printer_manager(self, db: AsyncSession, item: PrintQueueItem) -> None:
  7015. """Hand the queue item's owner to printer_manager so the
  7016. print-complete callback can credit the user in PrintLogEntry (#1670).
  7017. No-ops when the item has no `created_by_id` or the referenced user
  7018. row is missing (e.g. user deleted between queue-add and dispatch —
  7019. in that case the print log row falls back to the existing un-credited
  7020. behaviour rather than crashing the dispatch).
  7021. """
  7022. if not item.created_by_id:
  7023. return
  7024. from backend.app.models.user import User
  7025. owner = await db.get(User, item.created_by_id)
  7026. if owner:
  7027. printer_manager.set_current_print_user(item.printer_id, owner.id, owner.username)
  7028. async def _start_print(self, db: AsyncSession, item: PrintQueueItem):
  7029. """Upload file and start print for a queue item.
  7030. Supports two sources:
  7031. - archive_id: Print from an existing archive
  7032. - library_file_id: Print from a library file (file manager)
  7033. """
  7034. logger.info("Starting queue item %s", item.id)
  7035. # Also covers a reservation left active by a process interruption
  7036. # during an earlier attempt. `_dispatch_one` releases this marker on
  7037. # every exit unless start_print() confirms that the command was sent.
  7038. self._unconfirmed_budget_reservations.add(item.id)
  7039. try:
  7040. from backend.app.models.user import User
  7041. queue_user = await db.get(User, item.created_by_id) if item.created_by_id is not None else None
  7042. # Recompute at the final authorization boundary as well as enqueue
  7043. # time. This covers rows created before the server-side estimate
  7044. # migration and prevents any alternate write path from weakening
  7045. # the budget reservation.
  7046. archive = await db.get(PrintArchive, item.archive_id) if item.archive_id is not None else None
  7047. library_file = await db.get(LibraryFile, item.library_file_id) if item.library_file_id is not None else None
  7048. item.estimated_cost = await estimate_queue_source_cost(
  7049. db,
  7050. archive=archive,
  7051. library_file=library_file,
  7052. plate_id=item.plate_id,
  7053. ams_mapping=item.ams_mapping,
  7054. printer_id=item.printer_id,
  7055. )
  7056. await validate_print_budget(
  7057. db,
  7058. cost_center_id=item.cost_center_id,
  7059. estimated_cost=item.estimated_cost,
  7060. current_user=queue_user,
  7061. exclude_queue_item_id=item.id,
  7062. exclude_reservation_source_type="print_queue",
  7063. exclude_reservation_source_id=item.id,
  7064. )
  7065. budget_reservation = await create_budget_reservation(
  7066. db,
  7067. cost_center_id=item.cost_center_id,
  7068. estimated_cost=item.estimated_cost,
  7069. current_user=queue_user,
  7070. source_type="print_queue",
  7071. source_id=item.id,
  7072. print_archive_id=item.archive_id,
  7073. exclude_queue_item_id=item.id,
  7074. )
  7075. await db.commit()
  7076. except HTTPException as exc:
  7077. item.status = "failed"
  7078. item.error_message = str(exc.detail)
  7079. item.completed_at = datetime.now(timezone.utc)
  7080. await db.commit()
  7081. logger.error("Queue item %s: Budget check failed: %s", item.id, item.error_message)
  7082. await self._power_off_if_needed(db, item)
  7083. return
  7084. # Get printer first (needed for both paths)
  7085. result = await db.execute(select(Printer).where(Printer.id == item.printer_id))
  7086. printer = result.scalar_one_or_none()
  7087. if not printer:
  7088. item.status = "failed"
  7089. item.error_message = "Printer not found"
  7090. item.completed_at = datetime.now(timezone.utc)
  7091. await db.commit()
  7092. logger.error("Queue item %s: Printer %s not found", item.id, item.printer_id)
  7093. await self._power_off_if_needed(db, item)
  7094. return
  7095. # Check printer is connected
  7096. if not printer_manager.is_connected(item.printer_id):
  7097. item.status = "failed"
  7098. item.error_message = "Printer not connected"
  7099. item.completed_at = datetime.now(timezone.utc)
  7100. await db.commit()
  7101. logger.error("Queue item %s: Printer %s not connected", item.id, item.printer_id)
  7102. await self._power_off_if_needed(db, item)
  7103. return
  7104. # Cancel-while-dispatching race (#1853): the scheduler's snapshot of
  7105. # `items` was taken at the top of check_queue, but the user can /cancel
  7106. # any pending row in the gap before we reach this point. Re-read the
  7107. # row and bail out cleanly instead of starting an FTP upload for a row
  7108. # that's already cancelled. The atomic CAS at the pending→printing
  7109. # transition (below, before start_print) is the load-bearing guard;
  7110. # this is the early-exit optimisation that avoids wasted FTP I/O.
  7111. await db.refresh(item)
  7112. if item.status != "pending":
  7113. logger.info(
  7114. "Queue item %s no longer pending (status=%s) — aborting dispatch",
  7115. item.id,
  7116. item.status,
  7117. )
  7118. return
  7119. # Busy-printer guard (#2598). check_queue gates dispatch on
  7120. # _is_printer_idle(), but that treats FINISH as idle and a printer can
  7121. # keep reporting FINISH for tens of seconds *after* it accepted a
  7122. # project_file (see the watchdog's phase-B note). A watchdog revert
  7123. # (#2555) also releases the dispatch hold, so a re-selected item can
  7124. # reach here while its printer has actually started printing. Uploading
  7125. # and dispatching then collides with the live job — the firmware answers
  7126. # 0500_4004 and, on an A1 mini, cancels the running print. Re-check the
  7127. # live state right before the expensive FTP upload: if the printer is
  7128. # busy, leave the item pending and let a later tick dispatch it once the
  7129. # printer is genuinely idle. No wasted upload, no collision.
  7130. pre_dispatch_state = getattr(printer_manager.get_status(item.printer_id), "state", None)
  7131. if pre_dispatch_state in _ACTIVE_PRINT_STATES:
  7132. logger.info(
  7133. "Queue item %s: printer %s is busy (state=%s) — deferring dispatch, "
  7134. "leaving item pending for a later tick (#2598)",
  7135. item.id,
  7136. item.printer_id,
  7137. pre_dispatch_state,
  7138. )
  7139. return
  7140. # Determine source: archive or library file
  7141. archive = None
  7142. library_file = None
  7143. file_path = None
  7144. filename = None
  7145. cleanup_disk_paths: list[Path] = []
  7146. # Set when a dispatch consumes its library file, so the photos can be
  7147. # carried over after the commit that removes the row (#3077).
  7148. consumed_library_file_id: int | None = None
  7149. consumed_photos: list[str] = []
  7150. if item.archive_id:
  7151. # Print from archive
  7152. result = await db.execute(select(PrintArchive).where(PrintArchive.id == item.archive_id))
  7153. archive = result.scalar_one_or_none()
  7154. if not archive:
  7155. item.status = "failed"
  7156. item.error_message = "Archive not found"
  7157. item.completed_at = datetime.now(timezone.utc)
  7158. await db.commit()
  7159. logger.error("Queue item %s: Archive %s not found", item.id, item.archive_id)
  7160. await self._power_off_if_needed(db, item)
  7161. return
  7162. # Persist the queue item's selected plate onto the archive so Print
  7163. # History can show the actual plate after cancel/fail/complete (#2603).
  7164. # Only when the archive doesn't already carry one, so a reprint of a
  7165. # plate-specific archive isn't relabelled by a differently-plated
  7166. # queue row.
  7167. if archive.plate_id is None and item.plate_id is not None:
  7168. archive.plate_id = item.plate_id
  7169. # Ask-for-outcome opt-in rides from the queue item to the archive
  7170. # the same way (#1898); never cleared here so a reprint of an
  7171. # archive that already asked keeps asking.
  7172. if item.confirm_outcome:
  7173. archive.confirm_requested = True
  7174. file_path = settings.base_dir / archive.file_path
  7175. filename = archive.filename
  7176. elif item.library_file_id:
  7177. # Print from library file (file manager)
  7178. result = await db.execute(LibraryFile.active().where(LibraryFile.id == item.library_file_id))
  7179. library_file = result.scalar_one_or_none()
  7180. if not library_file:
  7181. # "Not found" covers two different situations and only one of
  7182. # them is recoverable, so say which. A trashed file is still
  7183. # there and restoring it makes a re-queued job work; a file that
  7184. # is really gone needs a different one. Neither is knowable from
  7185. # the queue, which is the whole complaint about this message.
  7186. trashed = (
  7187. await db.execute(select(LibraryFile.filename).where(LibraryFile.id == item.library_file_id))
  7188. ).scalar_one_or_none()
  7189. item.status = "failed"
  7190. item.error_message = (
  7191. f"'{trashed}' is in the library trash — restore it and queue the print again"
  7192. if trashed
  7193. else "Library file not found — it was deleted after this job was queued"
  7194. )
  7195. item.completed_at = datetime.now(timezone.utc)
  7196. await db.commit()
  7197. logger.error(
  7198. "Queue item %s: library file %s is %s",
  7199. item.id,
  7200. item.library_file_id,
  7201. "in the trash" if trashed else "gone",
  7202. )
  7203. await self._power_off_if_needed(db, item)
  7204. return
  7205. # Library files store absolute paths
  7206. lib_path = Path(library_file.file_path)
  7207. file_path = lib_path if lib_path.is_absolute() else settings.base_dir / library_file.file_path
  7208. filename = library_file.filename
  7209. # Create archive from library file so usage tracking has access to the 3MF
  7210. queue_item_id = item.id
  7211. # Held separately: a cleanup dispatch clears item.library_file_id
  7212. # below, and the log line at the end of this block reported that
  7213. # cleared field -- so every consumed print logged "from library
  7214. # file None", which is the one case worth being able to trace.
  7215. source_library_file_id = item.library_file_id
  7216. try:
  7217. from backend.app.services.archive import ArchiveService
  7218. archive_service = ArchiveService(db)
  7219. archive = await archive_service.archive_print(
  7220. printer_id=item.printer_id,
  7221. source_file=file_path,
  7222. original_filename=filename,
  7223. created_by_id=item.created_by_id,
  7224. project_id=item.project_id,
  7225. cost_center_id=item.cost_center_id,
  7226. library_file_id=item.library_file_id, # per-file project progress (#1897)
  7227. plate_id=item.plate_id, # selected plate → Print History (#2603)
  7228. )
  7229. if archive:
  7230. item.archive_id = archive.id
  7231. if item.confirm_outcome:
  7232. archive.confirm_requested = True # ask-for-outcome opt-in (#1898)
  7233. if budget_reservation is not None:
  7234. budget_reservation.print_archive_id = archive.id
  7235. if item.cleanup_library_after_dispatch and not library_file.is_external:
  7236. consumed_library_file_id = library_file.id
  7237. item.library_file_id = None
  7238. cleanup_disk_paths.append(file_path)
  7239. if library_file.thumbnail_path:
  7240. thumb_path = Path(library_file.thumbnail_path)
  7241. if not thumb_path.is_absolute():
  7242. thumb_path = settings.base_dir / library_file.thumbnail_path
  7243. cleanup_disk_paths.append(thumb_path)
  7244. # Before the delete, not after: on PostgreSQL the FK
  7245. # cascade would already have taken these rows (#2819).
  7246. await self._repoint_siblings_at_archive(
  7247. db,
  7248. consumed_library_file_id=consumed_library_file_id,
  7249. archive_id=archive.id,
  7250. dispatched_item_id=item.id,
  7251. )
  7252. # Read while the row is still here; the photos move
  7253. # below, once the delete has actually committed.
  7254. consumed_photos = list(library_file.photos or [])
  7255. await db.delete(library_file)
  7256. file_path = settings.base_dir / archive.file_path
  7257. filename = archive.filename
  7258. # Commit, not flush — flush opens the SQLite write
  7259. # transaction (item.archive_id update + library_file
  7260. # delete) and would hold the WAL writer lock through the
  7261. # FTP upload below, causing "database is locked" cascades
  7262. # for sensor history + concurrent cancels (#1853).
  7263. await db.commit()
  7264. logger.info(
  7265. "Queue item %s: Created archive %s from library file %s",
  7266. item.id,
  7267. archive.id,
  7268. source_library_file_id,
  7269. )
  7270. except Exception as e:
  7271. logger.warning(
  7272. "Queue item %s: Failed to create archive from library file: %s",
  7273. queue_item_id,
  7274. e,
  7275. exc_info=True,
  7276. )
  7277. await db.rollback()
  7278. item = await db.get(PrintQueueItem, queue_item_id)
  7279. if item:
  7280. item.status = "failed"
  7281. item.error_message = "Failed to create archive from library file"
  7282. item.completed_at = datetime.now(timezone.utc)
  7283. await db.commit()
  7284. await self._power_off_if_needed(db, item)
  7285. return
  7286. if not archive:
  7287. item.status = "failed"
  7288. item.error_message = "Failed to create archive from library file"
  7289. item.completed_at = datetime.now(timezone.utc)
  7290. await db.commit()
  7291. logger.error("Queue item %s: Archive creation from library file returned no archive", item.id)
  7292. await self._power_off_if_needed(db, item)
  7293. return
  7294. # The photos follow the file into the archive that replaces it, for
  7295. # the same reason the siblings do (#3077). After the commit above,
  7296. # never before it: that commit can fail ("database is locked",
  7297. # #1853) and roll the library row back, and photos already moved
  7298. # would leave it naming a directory that no longer exists. The
  7299. # file and thumbnail unlinks are deferred for the same reason.
  7300. if consumed_library_file_id is not None and consumed_photos:
  7301. # Held as a plain int, read here while the session is still
  7302. # healthy, because the handler below may not touch an ORM
  7303. # instance at all. The commit it exists for fails inside the
  7304. # FLUSH, not at COMMIT: SQLite takes the write lock at the
  7305. # first DML statement, so a busy writer surfaces as "database
  7306. # is locked" on the UPDATE (#1853). SQLAlchemy rolls that back
  7307. # internally through safe_reraise before re-raising, which
  7308. # expires every loaded instance and leaves the session in
  7309. # pending-rollback state -- so `archive.id` inside the except
  7310. # would itself raise PendingRollbackError and the rollback
  7311. # below would never be reached.
  7312. archive_id = archive.id
  7313. try:
  7314. carried_photos = move_library_photos(
  7315. consumed_library_file_id,
  7316. consumed_photos,
  7317. archive_photos_dir(archive),
  7318. )
  7319. if carried_photos:
  7320. archive.photos = list(archive.photos or []) + carried_photos
  7321. await db.commit()
  7322. except Exception as e:
  7323. # The archive and the delete are already committed; the
  7324. # print goes ahead either way. Worst case the pictures sit
  7325. # unnamed in the archive's own directory.
  7326. #
  7327. # Ints only until the rollback has run, per the note above,
  7328. # which is why this logs queue_item_id and not item.id --
  7329. # the sibling handler forty lines up does the same.
  7330. logger.warning(
  7331. "Queue item %s: failed to carry library photos into archive %s: %s",
  7332. queue_item_id,
  7333. archive_id,
  7334. e,
  7335. )
  7336. await db.rollback()
  7337. # rollback() expires every loaded instance, and in async
  7338. # SQLAlchemy the next plain attribute read is lazy IO
  7339. # outside the greenlet -- MissingGreenlet, which would turn
  7340. # this cosmetic failure into a dispatch crash in exactly the
  7341. # "database is locked" case the block exists for (#1853).
  7342. # The nozzle guard, the upload and the start all keep
  7343. # reading item, archive and printer, so all three go back
  7344. # into the session before falling through.
  7345. item = await db.get(PrintQueueItem, queue_item_id)
  7346. archive = await db.get(PrintArchive, archive_id)
  7347. printer = await db.get(Printer, item.printer_id) if item else None
  7348. if not item or not archive or not printer:
  7349. logger.error(
  7350. "Queue item %s: item, archive %s or printer gone after the photo rollback",
  7351. queue_item_id,
  7352. archive_id,
  7353. )
  7354. return
  7355. else:
  7356. # Neither archive nor library file specified
  7357. item.status = "failed"
  7358. item.error_message = "No source file specified"
  7359. item.completed_at = datetime.now(timezone.utc)
  7360. await db.commit()
  7361. logger.error("Queue item %s: No archive_id or library_file_id specified", item.id)
  7362. await self._power_off_if_needed(db, item)
  7363. return
  7364. # Check file exists on disk
  7365. if not file_path.exists():
  7366. item.status = "failed"
  7367. item.error_message = "Source file not found on disk"
  7368. item.completed_at = datetime.now(timezone.utc)
  7369. await db.commit()
  7370. logger.error("Queue item %s: File not found: %s", item.id, file_path)
  7371. await self._power_off_if_needed(db, item)
  7372. return
  7373. # Nozzle-diameter mismatch guard (#1899). A file sliced for one nozzle
  7374. # size dispatched to a printer with a different nozzle installed is
  7375. # rejected by the firmware with a cryptic HMS ("Failed to get AMS mapping
  7376. # table" 0700_8012, or "nozzle diameter … not consistent" 0500_4038) that
  7377. # gives the user no idea what went wrong. Catch it here, before we spend
  7378. # time preheating and uploading, and fail with an actionable message.
  7379. # Fail-safe by construction: only a POSITIVE mismatch blocks — when the
  7380. # slice carries no nozzle diameter (archive.nozzle_diameter is None) or
  7381. # the printer hasn't reported its nozzles yet, we fall through and let the
  7382. # print proceed exactly as before. On dual-nozzle printers (H2D) a match
  7383. # against EITHER installed nozzle passes, so a 0.6 slice is fine as long
  7384. # as one of the two hotends is a 0.6.
  7385. #
  7386. # On a tool-changer model (H2C) the nozzles parked in the rack count too
  7387. # (#2885): the printer fetches one as part of starting the print, so the
  7388. # set to test against is "reachable", not "currently mounted". This runs
  7389. # well before the rack picker at the bottom of this method, so without
  7390. # the rack in scope here that picker never got the chance to run.
  7391. sliced_nozzle = archive.nozzle_diameter if archive else None
  7392. if sliced_nozzle:
  7393. nozzle_status = printer_manager.get_status(item.printer_id)
  7394. installed = _installed_nozzle_diameters(nozzle_status)
  7395. rack = _rack_nozzle_diameters(nozzle_status)
  7396. mismatch_msg = _nozzle_mismatch_message(sliced_nozzle, installed, rack)
  7397. if mismatch_msg:
  7398. item.status = "failed"
  7399. item.error_message = mismatch_msg
  7400. item.completed_at = datetime.now(timezone.utc)
  7401. await db.commit()
  7402. logger.warning("Queue item %s: nozzle mismatch — %s", item.id, mismatch_msg)
  7403. await notification_service.on_queue_job_failed(
  7404. job_name=filename.replace(".gcode.3mf", "").replace(".3mf", ""),
  7405. printer_id=printer.id,
  7406. printer_name=printer.name,
  7407. reason=mismatch_msg,
  7408. db=db,
  7409. )
  7410. try:
  7411. await ws_manager.send_queue_item_failed(
  7412. user_id=item.created_by_id,
  7413. queue_item_id=item.id,
  7414. printer_id=item.printer_id,
  7415. reason="nozzle_mismatch",
  7416. )
  7417. except Exception:
  7418. pass
  7419. await self._power_off_if_needed(db, item)
  7420. return
  7421. # Preheat / heat-soak (#1468) — fires before upload so the printer's
  7422. # bed (and chamber, if applicable) is at temperature when the firmware
  7423. # starts the actual print routine. Best-effort: any failure logs and
  7424. # falls through to the normal upload+start path rather than turning a
  7425. # configuration issue into a failed queue item.
  7426. # Returns False only when the item was cancelled or deleted while the
  7427. # stage was holding at temperature. Uploading and starting it anyway
  7428. # would print a job the user has already called off, so abandon the
  7429. # dispatch here; `_dispatch_one`'s finally clause unwinds the heaters.
  7430. if not await self._preheat_and_soak(db, item, printer, archive):
  7431. logger.info("Queue item %s: dispatch abandoned — cancelled during preheat", item.id)
  7432. return
  7433. # See `_effective_plate_id` for why this is resolved once here rather
  7434. # than each site below repeating `item.plate_id or 1`.
  7435. effective_plate_id = _effective_plate_id(item.plate_id, file_path)
  7436. # G-code injection for auto-print systems (#422)
  7437. injected_path = None
  7438. # #2547: tracked separately from `injected_path`, which is also set when
  7439. # only a START snippet was injected. Only an END snippet changes what the
  7440. # camera sees at print completion.
  7441. end_gcode_injected = False
  7442. if item.gcode_injection:
  7443. try:
  7444. snippets_raw = await self._get_setting(db, "gcode_snippets")
  7445. if snippets_raw:
  7446. snippets = json.loads(snippets_raw)
  7447. model_snippets = snippets.get(printer.model, {})
  7448. start_gc = (model_snippets.get("start_gcode") or "").strip()
  7449. end_gc = (model_snippets.get("end_gcode") or "").strip()
  7450. if start_gc or end_gc:
  7451. from backend.app.utils.threemf_tools import inject_gcode_into_3mf
  7452. injected_path = inject_gcode_into_3mf(
  7453. file_path, effective_plate_id, start_gc or None, end_gc or None
  7454. )
  7455. if injected_path:
  7456. file_path = injected_path
  7457. end_gcode_injected = bool(end_gc)
  7458. logger.info("Queue item %s: G-code injected for model %s", item.id, printer.model)
  7459. else:
  7460. logger.warning(
  7461. "Queue item %s: G-code injection returned no result, using original", item.id
  7462. )
  7463. except Exception as e:
  7464. logger.warning("Queue item %s: G-code injection failed, using original: %s", item.id, e)
  7465. # #2547: the finish-photo path can't learn from telemetry that this print
  7466. # ends with user End G-code — which means the plate may be gone by the
  7467. # time FINISH arrives (#1867). Flag it here; `on_print_start` binds it to
  7468. # the print once the printer confirms it running.
  7469. if end_gcode_injected:
  7470. print_dispatch_context.mark_pending(printer.id)
  7471. # Upload to root directory (not /cache/) - the start_print command references
  7472. # files by name only (ftp://{filename}), so they must be in the root
  7473. remote_filename = derive_remote_filename(filename)
  7474. remote_path = f"/{remote_filename}"
  7475. # Get FTP retry settings
  7476. ftp_retry_enabled, ftp_retry_count, ftp_retry_delay, ftp_timeout = await get_ftp_retry_settings()
  7477. logger.info(
  7478. f"Queue item {item.id}: FTP upload starting - printer={printer.name} ({printer.model}), "
  7479. f"ip={printer.ip_address}, file={remote_filename}, local_path={file_path}, "
  7480. f"retry_enabled={ftp_retry_enabled}, retry_count={ftp_retry_count}, timeout={ftp_timeout}"
  7481. )
  7482. # Release the pooled DB connection before the FTP delete/upload (#2572).
  7483. # Every read this method needs (printer, archive/library, preheat) is
  7484. # done, and the library-file branch already committed its archive
  7485. # creation. Without this the transaction opened by the first SELECT above
  7486. # stays "idle in transaction" for the entire upload — multiple seconds
  7487. # for a large 3MF — pinning one pooled connection per in-flight dispatch;
  7488. # a farm dispatching many jobs at once then exhausts the pool. This was
  7489. # correlated to an exact idle-in-transaction session on a 93-printer farm
  7490. # (reporter @Jostxxl). expire_on_commit=False keeps item/printer/archive
  7491. # readable; the status writes below (upload-failure path and the
  7492. # pending->printing CAS) transparently open a fresh transaction.
  7493. await db.commit()
  7494. # Delete existing file if present (avoids 553 error on overwrite)
  7495. try:
  7496. logger.debug("Queue item %s: Deleting existing file %s if present...", item.id, remote_path)
  7497. delete_result = await delete_file_async(
  7498. printer.ip_address,
  7499. printer.access_code,
  7500. remote_path,
  7501. socket_timeout=ftp_timeout,
  7502. printer_model=printer.model,
  7503. # This delete and the upload below are one bounded, user-initiated
  7504. # unit -- at most nine connections -- so neither skips on the
  7505. # handshake cool-off the opportunistic sweeps rely on. In #2898's
  7506. # trace this delete took the TLS failure and armed the cool-off,
  7507. # and the upload's four attempts were then spent against it
  7508. # without a socket being opened.
  7509. respect_handshake_cooloff=False,
  7510. )
  7511. logger.debug("Queue item %s: Delete result: %s", item.id, delete_result)
  7512. except Exception as e:
  7513. logger.debug("Queue item %s: Delete failed (may not exist): %s", item.id, e)
  7514. # Dispatch toast — announce the upload start with the total byte
  7515. # count so the frontend can render an honest progress bar.
  7516. toast_uid = item.created_by_id
  7517. toast_file_name = filename.replace(".gcode.3mf", "").replace(".3mf", "")
  7518. try:
  7519. total_bytes = file_path.stat().st_size
  7520. except OSError:
  7521. total_bytes = 0
  7522. try:
  7523. await ws_manager.send_queue_item_uploading(
  7524. user_id=toast_uid,
  7525. queue_item_id=item.id,
  7526. printer_id=item.printer_id,
  7527. printer_name=printer.name,
  7528. file_name=toast_file_name,
  7529. total_bytes=total_bytes,
  7530. )
  7531. except Exception:
  7532. pass # toast is best-effort
  7533. progress_bridge = _UploadProgressBridge(toast_uid, item.id)
  7534. # A deadline expiry gets its own message: "check your SD card" is the
  7535. # wrong advice for a link that was simply too slow to finish (#2529).
  7536. upload_error: str | None = None
  7537. # Why the upload failed, straight from the client rather than inferred.
  7538. # Owned here, so a background fetch for another print cannot overwrite
  7539. # it between the failure and the sentence built from it (#2899).
  7540. upload_failure = FtpFailureReport()
  7541. try:
  7542. if ftp_retry_enabled:
  7543. uploaded = await with_ftp_retry(
  7544. upload_file_async,
  7545. printer.ip_address,
  7546. printer.access_code,
  7547. file_path,
  7548. remote_path,
  7549. socket_timeout=ftp_timeout,
  7550. printer_model=printer.model,
  7551. progress_callback=progress_bridge,
  7552. respect_handshake_cooloff=False,
  7553. failure=upload_failure,
  7554. max_retries=ftp_retry_count,
  7555. retry_delay=ftp_retry_delay,
  7556. operation_name=f"Upload print to {printer.name}",
  7557. )
  7558. else:
  7559. uploaded = await upload_file_async(
  7560. printer.ip_address,
  7561. printer.access_code,
  7562. file_path,
  7563. remote_path,
  7564. socket_timeout=ftp_timeout,
  7565. printer_model=printer.model,
  7566. progress_callback=progress_bridge,
  7567. respect_handshake_cooloff=False,
  7568. failure=upload_failure,
  7569. )
  7570. except UploadCancelled as e:
  7571. uploaded = False
  7572. upload_error = (
  7573. "Upload was too slow to finish and was cancelled. The printer's connection could not sustain "
  7574. "the transfer — check its Wi-Fi signal, or move it closer to the access point."
  7575. )
  7576. logger.error("Queue item %s: upload deadline exceeded: %s", item.id, e)
  7577. except Exception as e:
  7578. uploaded = False
  7579. logger.error("Queue item %s: FTP error: %s (type: %s)", item.id, e, type(e).__name__)
  7580. # Clean up injected temp file after upload attempt
  7581. if injected_path and injected_path.exists():
  7582. injected_path.unlink(missing_ok=True)
  7583. if not uploaded:
  7584. # This used to be one string for every upload failure, telling
  7585. # everyone to check the SD card. The client knows which of seven
  7586. # things went wrong and logs each one differently; it just had no
  7587. # way to say so here, so the card got named even for a TLS
  7588. # handshake that never reached the printer's filesystem (#2899).
  7589. error_msg = upload_error or describe_upload_failure(upload_failure.failure)
  7590. failure = upload_failure.failure
  7591. if upload_error is None and failure is not None and failure.kind in _UPLOAD_REQUEUE_KINDS:
  7592. await self._requeue_after_upload_refused(db, item, printer, error_msg, toast_uid)
  7593. return
  7594. item.status = "failed"
  7595. item.error_message = error_msg
  7596. item.completed_at = datetime.now(timezone.utc)
  7597. await db.commit()
  7598. logger.error(
  7599. f"Queue item {item.id}: FTP upload failed - printer={printer.name}, model={printer.model}, "
  7600. f"ip={printer.ip_address}. Check logs above for storage diagnostics and specific error codes."
  7601. )
  7602. # Send failure notification
  7603. await notification_service.on_queue_job_failed(
  7604. job_name=filename.replace(".gcode.3mf", "").replace(".3mf", ""),
  7605. printer_id=printer.id,
  7606. printer_name=printer.name,
  7607. # The same sentence the queue shows. A push notification saying
  7608. # something different from the UI is its own small bug (#2899).
  7609. reason=error_msg,
  7610. db=db,
  7611. )
  7612. try:
  7613. await ws_manager.send_queue_item_failed(
  7614. user_id=toast_uid,
  7615. queue_item_id=item.id,
  7616. printer_id=item.printer_id,
  7617. reason="upload_failed",
  7618. )
  7619. except Exception:
  7620. pass
  7621. await self._power_off_if_needed(db, item)
  7622. return
  7623. # The printer took a file, so any run of refused uploads is over: the
  7624. # next refusal starts again at the shortest backoff, and notifies (#3210).
  7625. self._upload_refusals.pop(printer.id, None)
  7626. self._upload_refusal_notified.discard(printer.id)
  7627. # Parse AMS mapping if stored
  7628. ams_mapping = None
  7629. if item.ams_mapping:
  7630. try:
  7631. ams_mapping = json.loads(item.ams_mapping)
  7632. except json.JSONDecodeError:
  7633. logger.warning("Queue item %s: Invalid AMS mapping JSON, ignoring", item.id)
  7634. # Register as expected print so we don't create a duplicate archive
  7635. # Only applicable for archive-based prints
  7636. if archive:
  7637. from backend.app.main import register_expected_print
  7638. register_expected_print(
  7639. item.printer_id,
  7640. remote_filename,
  7641. archive.id,
  7642. ams_mapping=ams_mapping,
  7643. created_by_id=item.created_by_id,
  7644. cost_center_id=item.cost_center_id,
  7645. # The plate actually dispatched, not the queue item's raw
  7646. # column: on None, `register_expected_print` stores nothing,
  7647. # and `extract_filament_usage_from_3mf` then books every
  7648. # filament in the file rather than the one plate that printed.
  7649. plate_id=effective_plate_id,
  7650. )
  7651. # Registration happens before the print command by necessity (the
  7652. # printer can report the print before the send returns), so record
  7653. # what to undo if we never get as far as sending. `_dispatch_one`
  7654. # rolls back anything still pending here on every exit — exception,
  7655. # early return, or cancel winning the CAS below.
  7656. self._unconfirmed_expected_print[item.id] = (item.printer_id, remote_filename, archive.id)
  7657. # Propagate the queue item's owner into printer_manager so the
  7658. # print-complete callback can credit the user in the PrintLogEntry
  7659. # (#1670). `created_by_id` is set either at queue-add time (UI-added
  7660. # items) or when the user clicks the manual-start button.
  7661. await self._propagate_owner_to_printer_manager(db, item)
  7662. # IMPORTANT: Set status to "printing" BEFORE sending the print command.
  7663. # This prevents phantom reprints if the backend crashes/restarts after the
  7664. # print command is sent but before the status update is committed.
  7665. # If we crash after this commit but before start_print(), the item will be
  7666. # in "printing" status without actually printing - but that's safer than
  7667. # accidentally reprinting the same file hours later.
  7668. #
  7669. # Atomic CAS (#1853): a user pressing /cancel mid-dispatch (between the
  7670. # initial pending read at the top of check_queue and this point) flips
  7671. # the row to "cancelled" in a separate session. Without the WHERE
  7672. # status='pending' clause, the unconditional update here would silently
  7673. # overwrite that cancellation and we'd ship the MQTT start_print below
  7674. # — printer obeys, user sees "I pressed cancel and the print started".
  7675. # rowcount==0 means the user won the race; bail out, best-effort delete
  7676. # the file we just uploaded, do NOT send start_print.
  7677. now_utc = datetime.now(timezone.utc)
  7678. billing_run_id = str(uuid.uuid4())
  7679. cas = await db.execute(
  7680. update(PrintQueueItem)
  7681. .where(PrintQueueItem.id == item.id)
  7682. .where(PrintQueueItem.status == "pending")
  7683. .values(status="printing", started_at=now_utc, billing_run_id=billing_run_id)
  7684. )
  7685. await db.commit()
  7686. if cas.rowcount == 0:
  7687. logger.info(
  7688. "Queue item %s no longer pending at print-command time "
  7689. "(cancelled or removed mid-dispatch) — aborting before MQTT send (#1853)",
  7690. item.id,
  7691. )
  7692. try:
  7693. await delete_file_async(
  7694. printer.ip_address,
  7695. printer.access_code,
  7696. remote_path,
  7697. socket_timeout=ftp_timeout,
  7698. printer_model=printer.model,
  7699. )
  7700. except Exception as cleanup_err:
  7701. logger.debug(
  7702. "Queue item %s: best-effort cleanup of uploaded file failed: %s",
  7703. item.id,
  7704. cleanup_err,
  7705. )
  7706. try:
  7707. await ws_manager.send_queue_item_failed(
  7708. user_id=toast_uid,
  7709. queue_item_id=item.id,
  7710. printer_id=item.printer_id,
  7711. reason="cancelled_mid_dispatch",
  7712. )
  7713. except Exception:
  7714. pass
  7715. return
  7716. # Sync the in-memory item so subsequent code that reads item.status /
  7717. # item.started_at sees the values we just persisted.
  7718. item.status = "printing"
  7719. item.started_at = now_utc
  7720. item.billing_run_id = billing_run_id
  7721. if archive is not None:
  7722. archive.billing_run_id = billing_run_id
  7723. # Legacy transaction deletion used an archive-wide skip flag.
  7724. # A newly dispatched run has its own UUID/tombstone, so it must be
  7725. # billable independently of any older deleted run on this archive.
  7726. archive.wallet_charge_skipped = False
  7727. # Persist before MQTT send so completion and restart recovery can
  7728. # always recover the internal billing identity.
  7729. await db.commit()
  7730. for cleanup_path in cleanup_disk_paths:
  7731. try:
  7732. if cleanup_path.exists():
  7733. cleanup_path.unlink()
  7734. except OSError as cleanup_err:
  7735. logger.warning(
  7736. "TRANSIENT_LIBRARY_FILE_ORPHAN %s",
  7737. json.dumps(
  7738. {
  7739. "queue_item_id": item.id,
  7740. "path": str(cleanup_path),
  7741. "error": str(cleanup_err),
  7742. },
  7743. sort_keys=True,
  7744. ),
  7745. )
  7746. # Clear the awaiting-plate-clear flag now that we're starting a new print
  7747. printer_manager.set_awaiting_plate_clear(item.printer_id, False)
  7748. # #1898: with the opt-in default-good setting, moving on to the next
  7749. # print resolves the previous print's unanswered outcome prompt as
  7750. # "good" — this path also covers the camera-based plate detection,
  7751. # which releases the gate by allowing dispatch rather than by an
  7752. # explicit acknowledgment. Rides on the dispatch transaction.
  7753. if await self._get_bool_setting(db, "confirm_default_good_on_plate_clear", default=False):
  7754. from backend.app.services.print_confirmation import resolve_pending_confirmation_as_good
  7755. await resolve_pending_confirmation_as_good(db, item.printer_id)
  7756. logger.info("Queue item %s: Status set to 'printing', sending print command...", item.id)
  7757. # Capture state before dispatch so the watchdog can detect whether the
  7758. # printer actually transitioned (#967). Also capture subtask_id so the
  7759. # watchdog can recognise "command landed but state hasn't flipped yet"
  7760. # on slow H2D transitions (#1078).
  7761. pre_status = printer_manager.get_status(item.printer_id)
  7762. pre_state = getattr(pre_status, "state", None) if pre_status else None
  7763. pre_subtask_id = getattr(pre_status, "subtask_id", None) if pre_status else None
  7764. pre_gcode_file = getattr(pre_status, "gcode_file", None) if pre_status else None
  7765. # #1721: respect the user's explicit timelapse choice. The #1397
  7766. # force-on at dispatch was removed because it caused per-layer nozzle
  7767. # parking on slicer profiles with Timelapse Type = Smooth. Finish-photo
  7768. # capture is now driven by the stg_cur=22 transition in bambu_mqtt.py
  7769. # ("Filament unloading", toolhead parked, bed not yet dropped) with a
  7770. # FINISH-state fallback — no need to force a video.
  7771. effective_timelapse = bool(item.timelapse)
  7772. # Nozzle-rack fallback (#2800). A job that never passed through the
  7773. # Virtual Printer carries no Bambu Studio nozzle pick, and an H2C then
  7774. # dispatches with no nozzle field at all and chooses for itself — which
  7775. # is how a print levelled on one hotend and then printed on another,
  7776. # millimetres above the plate. Derive the per-slot extruder assignment
  7777. # from the file being dispatched.
  7778. #
  7779. # Done here rather than at queue time because this is the first point
  7780. # that knows both the actual printer and the actual file: an item can
  7781. # be created without a printer (model-based assignment), reassigned
  7782. # afterwards, or have its file swapped for a G-code-injected copy just
  7783. # above. Every queue-creation path — the print dialog, a bulk library
  7784. # add, the webhook, a pipeline run — is covered by the one call.
  7785. # Skipped when the item already carries a Bambu Studio capture: that
  7786. # one wins downstream anyway, so reading the 3MF again would be work
  7787. # thrown away on every dispatch.
  7788. # Rack position resolution (#1784), tried before the #2800 fallback
  7789. # below because it can express what that one cannot: a plate wanting a
  7790. # *different* hotend off the rack per filament group. The position per
  7791. # group is the operator's pick — the 3MF states it nowhere, proven by
  7792. # sending one plate twice with different picks and finding the two
  7793. # files identical bar float noise — so it is resolved here against the
  7794. # rack as it stands right now, after the upload, not at queue time.
  7795. resolved_nozzle_mapping = None
  7796. if not item.nozzle_mapping and file_path is not None and is_nozzle_rack_model(printer.model):
  7797. rack_plan = extract_rack_plan_from_3mf(file_path, plate_id=effective_plate_id)
  7798. if rack_plan is not None:
  7799. try:
  7800. stored_choice = json.loads(item.nozzle_rack_choice) if item.nozzle_rack_choice else {}
  7801. except (json.JSONDecodeError, TypeError):
  7802. stored_choice = {}
  7803. logger.warning(
  7804. "Queue item %s: unreadable nozzle_rack_choice %r, assigning rack positions instead",
  7805. item.id,
  7806. item.nozzle_rack_choice,
  7807. )
  7808. # JSON object keys are strings; the groups are ints.
  7809. choice: dict[int, int] = {}
  7810. for key, value in (stored_choice or {}).items():
  7811. try:
  7812. choice[int(key)] = int(value)
  7813. except (TypeError, ValueError):
  7814. continue
  7815. live_rack = getattr(printer_manager.get_status(item.printer_id), "nozzle_rack", None) or []
  7816. resolved_nozzle_mapping, rack_error = resolve_rack_plan_mapping(
  7817. rack_plan.slot_groups, rack_plan.group_dicts(), choice, live_rack
  7818. )
  7819. if resolved_nozzle_mapping is None and choice:
  7820. # An explicit pick that no longer holds stops the print. The
  7821. # operator named a hotend; printing from a different one is
  7822. # how a plate gets levelled on one nozzle and drawn with
  7823. # another, millimetres above the bed. Nothing has been sent
  7824. # to the printer yet, so failing here costs only the upload.
  7825. item.status = "failed"
  7826. item.error_message = (
  7827. f"Nozzle rack pick no longer fits the printer: {rack_error}. "
  7828. "Edit the item to choose another position."
  7829. )
  7830. item.completed_at = datetime.now(timezone.utc)
  7831. await db.commit()
  7832. logger.warning(
  7833. "Queue item %s: refusing to dispatch to %s — %s (chose %s, rack %s)",
  7834. item.id,
  7835. printer.name,
  7836. rack_error,
  7837. choice,
  7838. [slot.get("id") for slot in live_rack],
  7839. )
  7840. await notification_service.on_queue_job_failed(
  7841. job_name=filename.replace(".gcode.3mf", "").replace(".3mf", ""),
  7842. printer_id=printer.id,
  7843. printer_name=printer.name,
  7844. reason=item.error_message,
  7845. db=db,
  7846. )
  7847. try:
  7848. await ws_manager.send_queue_item_failed(
  7849. user_id=toast_uid,
  7850. queue_item_id=item.id,
  7851. printer_id=item.printer_id,
  7852. reason="nozzle_rack_pick_stale",
  7853. )
  7854. except Exception:
  7855. pass # Best-effort — don't fail the error handler
  7856. # The file is already on the SD card by this point, and a
  7857. # 3MF left there is a phantom print waiting to be started
  7858. # from the touchscreen. Same cleanup the start_print
  7859. # failure path below does, for the same reason.
  7860. try:
  7861. await delete_file_async(
  7862. printer.ip_address,
  7863. printer.access_code,
  7864. remote_path,
  7865. printer_model=printer.model,
  7866. )
  7867. except Exception:
  7868. pass # Best-effort — don't fail the error handler
  7869. return
  7870. if resolved_nozzle_mapping is None:
  7871. # Nothing was picked and nothing could be assigned. Falls
  7872. # through to the #2800 path, which is what ran before this
  7873. # existed — strictly not worse than today.
  7874. logger.info(
  7875. "Queue item %s: no rack positions assignable (%s); falling back",
  7876. item.id,
  7877. rack_error,
  7878. )
  7879. else:
  7880. logger.info(
  7881. "Queue item %s: rack mapping %s (groups %s, chosen %s)",
  7882. item.id,
  7883. resolved_nozzle_mapping,
  7884. rack_plan.slot_groups,
  7885. choice or "auto",
  7886. )
  7887. nozzle_slot_extruders = None
  7888. if (
  7889. not item.nozzle_mapping
  7890. and resolved_nozzle_mapping is None
  7891. and file_path is not None
  7892. and is_nozzle_rack_model(printer.model)
  7893. ):
  7894. slot_extruders = extract_slot_extruders_from_3mf(file_path, plate_id=effective_plate_id)
  7895. if slot_extruders:
  7896. nozzle_slot_extruders = json.dumps(slot_extruders)
  7897. # Every filament this plate prints is on the external spool -> the print
  7898. # must go out with use_ams=False. The firmware answers use_ams=true plus
  7899. # a mapping it cannot resolve with 07FF_8012 "Failed to get AMS mapping
  7900. # table", which is what held the reporter's P1S at Heatbed preheating
  7901. # for ten minutes before it gave up (#3087). The MQTT command builder
  7902. # already downgrades a mapping that is *only* external ([254]), but a
  7903. # multi-filament project pads the slots this plate does not print with
  7904. # -1 — BambuStudio's own convention — and down there a -1 is
  7905. # indistinguishable from a slot that never resolved, which must never be
  7906. # sent to the spool holder (#2589). Here the plate's filament list says
  7907. # which is which, so the answer is exact rather than a guess.
  7908. #
  7909. # Deliberately narrow: this fires only when every consumed slot is an
  7910. # explicit 254/255. A consumed slot that did not resolve leaves use_ams
  7911. # alone and the firmware still rejects the print, exactly as today. And
  7912. # only for single-nozzle printers, mirroring the builder's own reconcile
  7913. # — on a dual-nozzle machine use_ams is which extruder to feed, not
  7914. # whether to use the AMS, so it is not ours to rewrite.
  7915. effective_use_ams = item.use_ams
  7916. if (
  7917. effective_use_ams
  7918. and ams_mapping
  7919. and file_path is not None
  7920. # Cheap gate before opening the file: with nothing on the spool
  7921. # holder anywhere in the mapping, no subset of it can be all
  7922. # external, so most dispatches never pay for the parse. The
  7923. # isinstance also keeps a malformed stored mapping (a bare number
  7924. # from a hand-edited row) failing where it always failed, in the
  7925. # command builder, rather than here.
  7926. and isinstance(ams_mapping, list)
  7927. and any(_is_external_tray(t) for t in ams_mapping)
  7928. and not _might_be_dual_nozzle(printer.model, pre_status)
  7929. ):
  7930. from backend.app.services.filament_requirements import extract_filament_requirements
  7931. consumed = _consumed_mapping_entries(
  7932. ams_mapping, extract_filament_requirements(file_path, plate_id=effective_plate_id)
  7933. )
  7934. if consumed and all(_is_external_tray(t) for t in consumed):
  7935. effective_use_ams = False
  7936. logger.info(
  7937. "Queue item %s: every filament plate %s prints is on the external spool "
  7938. "(mapping %s) — dispatching with use_ams=False (#3087)",
  7939. item.id,
  7940. effective_plate_id,
  7941. ams_mapping,
  7942. )
  7943. # A slot that lost its K-profile selection (a power cycle resets every
  7944. # slot to the default K) would print on the default. The idle check in
  7945. # the status handler normally restores it first, but nothing guarantees
  7946. # it ran before this dispatch (#3219). Unthrottled: once per job.
  7947. # Never allowed to stop the print it is protecting.
  7948. try:
  7949. await kprofile_drift.reapply_lost_kprofiles(item.printer_id, _used_global_tray_ids(item), throttle=False)
  7950. except Exception:
  7951. logger.exception("Queue item %s: K-profile check before dispatch failed", item.id)
  7952. # Start the print with AMS mapping, plate_id and print options.
  7953. # nozzle_mapping rides through verbatim — JSON string captured from
  7954. # Bambu Studio's project_file on VP intake (#1780); the MQTT layer
  7955. # parses + injects it only for dual-nozzle models so a null on every
  7956. # other model is a transparent pass-through. The rack fallback is
  7957. # resolved down there too, where the live rack position is known.
  7958. started = printer_manager.start_print(
  7959. item.printer_id,
  7960. remote_filename,
  7961. plate_id=effective_plate_id,
  7962. ams_mapping=ams_mapping,
  7963. bed_levelling=item.bed_levelling,
  7964. flow_cali=item.flow_cali,
  7965. vibration_cali=item.vibration_cali,
  7966. layer_inspect=item.layer_inspect,
  7967. timelapse=effective_timelapse,
  7968. use_ams=effective_use_ams,
  7969. nozzle_offset_cali=item.nozzle_offset_cali,
  7970. nozzle_mapping=item.nozzle_mapping
  7971. or (json.dumps(resolved_nozzle_mapping) if resolved_nozzle_mapping else None),
  7972. nozzle_slot_extruders=nozzle_slot_extruders,
  7973. )
  7974. if started:
  7975. # The command is away, so the expectation is now legitimate and must
  7976. # survive. Anything still in this dict when _dispatch_one exits gets
  7977. # rolled back.
  7978. self._unconfirmed_expected_print.pop(item.id, None)
  7979. self._unconfirmed_budget_reservations.discard(item.id)
  7980. # Handoff to the print's own gcode: keep whatever preheat set (bed
  7981. # target, chamber target, airduct heating) — the gcode owns
  7982. # heater/flap control from here. Clearing the pin prevents
  7983. # `_dispatch_one`'s finally from unwinding a live print.
  7984. self._preheat_pin.pop(item.printer_id, None)
  7985. self._preheat_pin_bed.pop(item.printer_id, None)
  7986. logger.info("Queue item %s: Print started successfully - %s", item.id, filename)
  7987. # No dispatch-toast event here: the legacy bg-dispatch path kept
  7988. # status='processing' from upload start until the printer acked
  7989. # (or timed out). The frontend derives "Awaiting printer…" purely
  7990. # from upload_progress_pct >= 99.9; an explicit 'dispatched' WS
  7991. # event would push the status chip out of 'PROCESSING' prematurely
  7992. # — which is exactly what the screenshot at #1625-followup
  7993. # complained about.
  7994. # Register the local 3MF in the cover-cache so /cover skips FTP
  7995. # (#1166 follow-up). file_path was resolved earlier from either the
  7996. # archive or the library file row.
  7997. if file_path is not None:
  7998. cache_3mf_download(item.printer_id, remote_filename, file_path)
  7999. # Hold the printer against further dispatches until the watchdog
  8000. # confirms the printer transitioned (or until the hard timeout).
  8001. # Prevents multi-plate batches from triple-dispatching onto the
  8002. # same H2D Pro while it digests the first project_file (#1157).
  8003. self._mark_printer_dispatched(item.printer_id, pre_state, pre_subtask_id)
  8004. # Watchdog: if the printer never transitions out of pre_state AND
  8005. # never advances subtask_id, the MQTT publish was accepted locally but
  8006. # didn't reach the printer (half-broken session — same shape as
  8007. # #887/#936). Revert the queue item so the next dispatch can pick it
  8008. # up instead of leaving it stuck in "printing" (#967). subtask_id
  8009. # check avoids false reverts on slow H2D FINISH→PREPARE transitions
  8010. # that would otherwise cause the item to re-dispatch as a reprint
  8011. # of the just-finished job (#1078).
  8012. if pre_state:
  8013. spawn_background_task(
  8014. self._watchdog_print_start(
  8015. item.id,
  8016. item.printer_id,
  8017. pre_state,
  8018. pre_subtask_id,
  8019. pre_gcode_file,
  8020. created_by_id=toast_uid,
  8021. ),
  8022. name=f"watchdog-print-start-{item.id}",
  8023. )
  8024. # Get estimated time for notification.
  8025. #
  8026. # This used to fall back to `library_file.print_time_seconds`, a column
  8027. # LibraryFile does not have — the print time it knows about lives in
  8028. # `file_metadata`. So a library print whose archive carried no parseable
  8029. # print time (a plain .gcode, or a 3MF the parser could not read) raised
  8030. # AttributeError right here, *after* the printer had already been sent
  8031. # the job: the started-notification never fired, and the exception
  8032. # unwound the whole queue pass, so every other printer still waiting to
  8033. # be dispatched on that tick silently missed its turn.
  8034. #
  8035. # The queue item caches the print time at creation ("Cached from
  8036. # archive/library"), which is the value this was reaching for.
  8037. estimated_time = None
  8038. if archive and archive.print_time_seconds:
  8039. estimated_time = archive.print_time_seconds
  8040. elif item.print_time_seconds:
  8041. estimated_time = item.print_time_seconds
  8042. # Send job started notification
  8043. await notification_service.on_queue_job_started(
  8044. job_name=filename.replace(".gcode.3mf", "").replace(".3mf", ""),
  8045. printer_id=printer.id,
  8046. printer_name=printer.name,
  8047. db=db,
  8048. estimated_time=estimated_time,
  8049. )
  8050. # MQTT relay - publish queue job started
  8051. try:
  8052. from backend.app.services.mqtt_relay import mqtt_relay
  8053. await mqtt_relay.on_queue_job_started(
  8054. job_id=item.id,
  8055. filename=filename,
  8056. printer_id=printer.id,
  8057. printer_name=printer.name,
  8058. printer_serial=printer.serial_number,
  8059. )
  8060. except Exception:
  8061. pass # Don't fail if MQTT fails
  8062. else:
  8063. # Clean up uploaded file from SD card to prevent phantom prints
  8064. try:
  8065. await delete_file_async(
  8066. printer.ip_address,
  8067. printer.access_code,
  8068. remote_path,
  8069. printer_model=printer.model,
  8070. )
  8071. except Exception:
  8072. pass # Best-effort — don't fail the error handler
  8073. # Busy-refusal is a deferral, not a failure (#2598). The printer's
  8074. # state can flip from idle to active in the window between the
  8075. # pre-dispatch check above and this publish (the FTP upload takes
  8076. # seconds); start_print() then refuses to send project_file to the
  8077. # now-busy printer and returns False. Failing the item here would be
  8078. # wrong — the printer is fine, it is simply busy — so revert to
  8079. # pending and let a later tick dispatch it once the printer is idle,
  8080. # exactly like the pre-dispatch guard. Only a start_print() False on
  8081. # an idle/unknown printer is a genuine command failure.
  8082. post_dispatch_state = getattr(printer_manager.get_status(item.printer_id), "state", None)
  8083. if post_dispatch_state in _ACTIVE_PRINT_STATES:
  8084. logger.info(
  8085. "Queue item %s: printer %s became busy (state=%s) before the start "
  8086. "command was sent — deferring, reverting item to pending (#2598)",
  8087. item.id,
  8088. item.printer_id,
  8089. post_dispatch_state,
  8090. )
  8091. item.status = "pending"
  8092. item.started_at = None
  8093. await db.commit()
  8094. return
  8095. # Print command failed - revert status
  8096. item.status = "failed"
  8097. item.error_message = "Failed to send print command to printer"
  8098. item.completed_at = datetime.now(timezone.utc)
  8099. await db.commit()
  8100. logger.error(
  8101. f"Queue item {item.id}: Failed to start print on {printer.name} ({printer.model}) - "
  8102. f"printer_manager.start_print() returned False. "
  8103. f"This may indicate: printer not connected, MQTT error, unsupported model configuration, or firmware issue. "
  8104. f"Check printer status and backend logs for details."
  8105. )
  8106. # Send failure notification
  8107. await notification_service.on_queue_job_failed(
  8108. job_name=filename.replace(".gcode.3mf", "").replace(".3mf", ""),
  8109. printer_id=printer.id,
  8110. printer_name=printer.name,
  8111. reason="Failed to send print command to printer - check printer connection and status",
  8112. db=db,
  8113. )
  8114. try:
  8115. await ws_manager.send_queue_item_failed(
  8116. user_id=toast_uid,
  8117. queue_item_id=item.id,
  8118. printer_id=item.printer_id,
  8119. reason="start_command_failed",
  8120. )
  8121. except Exception:
  8122. pass
  8123. await self._power_off_if_needed(db, item)
  8124. @staticmethod
  8125. async def _watchdog_print_start(
  8126. queue_item_id: int,
  8127. printer_id: int,
  8128. pre_state: str,
  8129. pre_subtask_id: str | None = None,
  8130. pre_gcode_file: str | None = None,
  8131. timeout: float = 90.0,
  8132. phase_b_timeout: float = 180.0,
  8133. poll_interval: float = 3.0,
  8134. created_by_id: int | None = None,
  8135. ) -> None:
  8136. """Revert a queue item if the printer never acknowledges the start command.
  8137. Bambuddy optimistically marks the queue item as "printing" right after the
  8138. MQTT project_file publish succeeds locally. The watchdog runs in two phases:
  8139. Phase A (up to ``timeout``): wait for either an active-state transition
  8140. or a ``subtask_id`` advance past ``pre_subtask_id``. State alone is the
  8141. primary signal; subtask_id advance handles the H2D case where state can
  8142. sit at FINISH for ~50 s after the printer accepted ``project_file``
  8143. before flipping to PREPARE (#1078). If neither happens, the MQTT publish
  8144. was lost on a half-broken session (#887/#936) — revert and force
  8145. reconnect (the #967 recovery path).
  8146. Phase B (up to ``phase_b_timeout``, only if Phase A exited on subtask_id
  8147. alone): keep watching for the active-state transition. subtask_id alone
  8148. proves the file landed but not that the printer started — and a printer
  8149. that accepts the command but stays at IDLE/FINISH indefinitely (e.g.
  8150. cloud+LAN re-auth dance after a power cycle on old firmware, #1678)
  8151. used to leave the queue item stuck in 'printing' forever because the
  8152. old watchdog returned success as soon as subtask_id advanced. If Phase
  8153. B times out, revert the queue item so the user can retry without
  8154. restarting Bambuddy. Skip ``force_reconnect`` here: the file landed and
  8155. a forced reconnect mid-parse triggers 0500_4003 (#1150).
  8156. Phase A timeout raised from 45 s → 90 s as belt-and-braces for slow
  8157. transitions that also don't emit an early subtask_id tick.
  8158. Both phases also watch for ``HMS_MQTT_VERIFY_FAILED``. A printer that
  8159. refuses to verify our commands will never start this job or any other,
  8160. so waiting out the full 270 s and re-uploading the 3MF twice more only
  8161. burns an upload slot the rest of the farm is queued behind — that path
  8162. is for a printer that might still come good, which this one cannot
  8163. (#2732). It fails the item on the spot with the actual reason instead.
  8164. """
  8165. last_status = None
  8166. landed_on_subtask = False
  8167. # Latched, not level-tested: state.hms_errors is rebuilt from scratch on
  8168. # every push carrying an `hms` key, so the fault can come and go between
  8169. # 3-second polls. Seeing it once inside the dispatch window is enough.
  8170. command_rejected = False
  8171. # Latched for the same reason as command_rejected: drying can finish, or
  8172. # be stopped by the user, part-way through the dispatch window. Seeing it
  8173. # once is what matters — it is the state the printer was in when it
  8174. # declined to start (#2758).
  8175. drying_ams_ids: list[int] = []
  8176. deadline = time.monotonic() + timeout
  8177. while time.monotonic() < deadline:
  8178. await asyncio.sleep(poll_interval)
  8179. status = printer_manager.get_status(printer_id)
  8180. if not status:
  8181. # Printer disconnected — don't mess with the DB. Drop the
  8182. # in-memory dispatch hold too so a fresh dispatch can retry
  8183. # once the printer comes back; the hard timeout would
  8184. # otherwise hold the printer unnecessarily.
  8185. scheduler._release_dispatch_hold(printer_id)
  8186. return
  8187. last_status = status
  8188. if status.state in _ACTIVE_PRINT_STATES:
  8189. # Printer is actively processing the job — release the
  8190. # post-dispatch hold so the next pending item for this printer
  8191. # can be evaluated normally. We do NOT accept arbitrary state
  8192. # transitions: a printer going FINISH -> IDLE (user dismissed
  8193. # the post-print prompt without accepting our project_file)
  8194. # would otherwise look like "command landed" and leave the
  8195. # queue item stuck in 'printing' forever (#1370).
  8196. scheduler._release_dispatch_hold(printer_id)
  8197. try:
  8198. await ws_manager.send_queue_item_acked(
  8199. user_id=created_by_id,
  8200. queue_item_id=queue_item_id,
  8201. printer_id=printer_id,
  8202. )
  8203. except Exception:
  8204. pass
  8205. return
  8206. drying_ams_ids = drying_ams_ids or _drying_ams_ids(status)
  8207. # Checked only after the active-state exit above: a stale HMS left
  8208. # over from an earlier job must never abort a print that is visibly
  8209. # running. An actually-refused command leaves the printer idle, so
  8210. # this ordering costs the detection nothing.
  8211. if _mqtt_commands_rejected(status):
  8212. command_rejected = True
  8213. break
  8214. if pre_subtask_id is not None and status.subtask_id is not None and status.subtask_id != pre_subtask_id:
  8215. # Phase A exit — printer accepted the file (subtask_id flipped
  8216. # to our submission id). Don't return yet: the printer may
  8217. # have accepted the command but never actually start (e.g.
  8218. # cloud+LAN re-auth dance after a power cycle, #1678). Phase
  8219. # B watches for the active-state transition.
  8220. landed_on_subtask = True
  8221. break
  8222. if landed_on_subtask and not command_rejected:
  8223. phase_b_deadline = time.monotonic() + phase_b_timeout
  8224. while time.monotonic() < phase_b_deadline:
  8225. await asyncio.sleep(poll_interval)
  8226. status = printer_manager.get_status(printer_id)
  8227. if not status:
  8228. scheduler._release_dispatch_hold(printer_id)
  8229. return
  8230. last_status = status
  8231. if status.state in _ACTIVE_PRINT_STATES:
  8232. scheduler._release_dispatch_hold(printer_id)
  8233. try:
  8234. await ws_manager.send_queue_item_acked(
  8235. user_id=created_by_id,
  8236. queue_item_id=queue_item_id,
  8237. printer_id=printer_id,
  8238. )
  8239. except Exception:
  8240. pass
  8241. return
  8242. drying_ams_ids = drying_ams_ids or _drying_ams_ids(status)
  8243. # Same ordering rule as Phase A: a running print wins over a
  8244. # lingering HMS.
  8245. if _mqtt_commands_rejected(status):
  8246. command_rejected = True
  8247. break
  8248. # No active-state transition. Revert the item so the scheduler can retry.
  8249. # Drop the in-memory hold so the retry isn't blocked by it.
  8250. scheduler._release_dispatch_hold(printer_id)
  8251. # Logged on every failed dispatch window, not just the last one, so a
  8252. # support bundle shows the correlation from the first attempt rather than
  8253. # only after the retry budget is spent (#2758).
  8254. if drying_ams_ids:
  8255. logger.info(
  8256. "Queue item %s: printer %d never started while AMS %s drying — this may be why, see #2758",
  8257. queue_item_id,
  8258. printer_id,
  8259. ", ".join(str(i) for i in drying_ams_ids),
  8260. )
  8261. # Four outcomes from the revert attempt, each routed differently:
  8262. # "reverted": row flipped from printing -> pending, run recovery
  8263. # "gave_up": same, but the retry budget is spent — row failed
  8264. # rather than pending, so it stops going round again
  8265. # "already_moved_on": item.status != 'printing' (completed/cancelled by
  8266. # on_print_complete or user). Skip recovery entirely
  8267. # — the print clearly landed somewhere even if the
  8268. # watchdog didn't see the active-state transition.
  8269. # "revert_failed": SQLite contention exhausted retries. Still run
  8270. # recovery so the MQTT session gets a fresh client_id
  8271. # on the half-broken-session path.
  8272. #
  8273. # The retry budget (#2555): reverting to 'pending' hands the item straight
  8274. # back to the next queue pass, which re-uploads the whole 3MF and waits out
  8275. # the watchdog again. For a printer that is genuinely wedged that loop never
  8276. # ends — the reporter had one printer "since this morning still not launch"
  8277. # — and each lap also consumes an upload slot that the other printers in the
  8278. # farm are waiting on. Retrying is right; retrying forever is not.
  8279. async def _do_revert(db):
  8280. item = await db.get(PrintQueueItem, queue_item_id)
  8281. if not item or item.status != "printing":
  8282. return "already_moved_on"
  8283. item.dispatch_attempts = (item.dispatch_attempts or 0) + 1
  8284. item.started_at = None
  8285. # Charge the attempt to the candidate that was actually dispatched, so
  8286. # a cross-model item (#671) reaches for its other file next lap instead
  8287. # of retrying the printer that just failed to start. Matched by file
  8288. # because that is what the resolver copied onto the row.
  8289. if item.library_file_id is not None:
  8290. await db.execute(
  8291. update(PrintQueueVariant)
  8292. .where(PrintQueueVariant.queue_item_id == item.id)
  8293. .where(PrintQueueVariant.library_file_id == item.library_file_id)
  8294. .values(attempt_count=PrintQueueVariant.attempt_count + 1)
  8295. )
  8296. if command_rejected:
  8297. # No retry budget for this one: the printer refused to verify the
  8298. # command, and re-uploading the same 3MF to the same printer will
  8299. # be refused the same way. Fail now with the fix rather than after
  8300. # three laps of a message about SD cards (#2732).
  8301. item.status = "failed"
  8302. item.error_message = (
  8303. "The printer rejected the print command: MQTT command verification failed "
  8304. "(HMS 0500-0500-0001-0007). Enable Developer Mode on the printer, restart it, "
  8305. "then start the job again."
  8306. )
  8307. item.completed_at = datetime.now(timezone.utc)
  8308. await db.commit()
  8309. return "command_rejected"
  8310. if item.dispatch_attempts >= DISPATCH_MAX_ATTEMPTS:
  8311. item.status = "failed"
  8312. if drying_ams_ids:
  8313. # #2758: the generic message below sent the reporter looking
  8314. # at the SD card while the actual obstacle — AMS units in a
  8315. # drying cycle — was on screen the whole time. Name what we
  8316. # observed and let the user judge it; Bambuddy does not stop
  8317. # the cycle itself, because on this hardware drying can run
  8318. # alongside a print and stopping it may not be the fix.
  8319. units = ", ".join(f"AMS {i}" for i in drying_ams_ids)
  8320. item.error_message = (
  8321. f"The printer accepted the file but never started printing, after "
  8322. f"{item.dispatch_attempts} attempts. {units} "
  8323. f"{'was' if len(drying_ams_ids) == 1 else 'were'} drying throughout — "
  8324. f"some printers refuse to begin a print while an AMS is in a drying "
  8325. f"cycle, and an AMS drying without its external power supply can also "
  8326. f"leave too little power for the start-of-print calibration. Stop the "
  8327. f"drying, or connect the AMS power supply, and start the job again."
  8328. )
  8329. else:
  8330. item.error_message = (
  8331. f"The printer accepted the file but never started printing, after "
  8332. f"{item.dispatch_attempts} attempts. Check the printer's screen for a "
  8333. f"prompt or error, confirm its SD card is readable, and start the job again."
  8334. )
  8335. item.completed_at = datetime.now(timezone.utc)
  8336. await release_budget_reservation(
  8337. db,
  8338. source_type="print_queue",
  8339. source_id=item.id,
  8340. status="released",
  8341. )
  8342. await db.commit()
  8343. return "gave_up"
  8344. item.status = "pending"
  8345. await db.commit()
  8346. return "reverted"
  8347. try:
  8348. revert_outcome = await run_with_retry(_do_revert, label=f"watchdog revert item={queue_item_id}")
  8349. except Exception as e:
  8350. logger.warning(
  8351. "Queue item %s: failed to revert to 'pending' (printer %d): %s — "
  8352. "scheduler may keep treating this item as in-flight",
  8353. queue_item_id,
  8354. printer_id,
  8355. e,
  8356. )
  8357. revert_outcome = "revert_failed"
  8358. if revert_outcome == "already_moved_on":
  8359. # Preserves the pre-#1370 early-return: if on_print_complete (or any
  8360. # other path) already moved the item past 'printing', don't run the
  8361. # MQTT session-recovery logic below — a forced reconnect on a healthy
  8362. # session breaks ongoing prints on the same printer.
  8363. return
  8364. total_timeout = timeout + (phase_b_timeout if landed_on_subtask else 0.0)
  8365. if revert_outcome == "command_rejected":
  8366. logger.error(
  8367. "Queue item %s: printer %d reported HMS %s (MQTT command verification "
  8368. "failed) — the print command was rejected, not lost. Failing the item "
  8369. "without retrying; enable Developer Mode on the printer and restart it (#2732)",
  8370. queue_item_id,
  8371. printer_id,
  8372. HMS_MQTT_VERIFY_FAILED,
  8373. )
  8374. await scheduler._notify_dispatch_gave_up(
  8375. queue_item_id,
  8376. printer_id,
  8377. created_by_id,
  8378. reason="Printer rejected the print command (MQTT command verification failed)",
  8379. )
  8380. # Same reasoning as the landed_on_subtask path below: the file is on
  8381. # the printer and a forced reconnect would only add 0500_4003 to a
  8382. # problem that has nothing to do with the MQTT session (#1150).
  8383. return
  8384. if revert_outcome == "gave_up":
  8385. logger.error(
  8386. "Queue item %s: printer %d never started the print after %d dispatch "
  8387. "attempts (last one waited %.0fs) — marking the item failed instead of "
  8388. "re-uploading it again (#2555)",
  8389. queue_item_id,
  8390. printer_id,
  8391. DISPATCH_MAX_ATTEMPTS,
  8392. total_timeout,
  8393. )
  8394. await scheduler._notify_dispatch_gave_up(queue_item_id, printer_id, created_by_id)
  8395. elif revert_outcome == "reverted":
  8396. if landed_on_subtask:
  8397. logger.warning(
  8398. "Queue item %s: printer %d accepted project_file (subtask_id "
  8399. "advanced) but never transitioned to an active state within "
  8400. "%.0fs — printer wedged post-acceptance; reverted to 'pending' "
  8401. "for retry (#1678)",
  8402. queue_item_id,
  8403. printer_id,
  8404. total_timeout,
  8405. )
  8406. else:
  8407. logger.warning(
  8408. "Queue item %s: printer %d did not respond to print command within "
  8409. "%.0fs (state still %s, subtask_id still %s) — reverted to 'pending' "
  8410. "for retry (#967)",
  8411. queue_item_id,
  8412. printer_id,
  8413. timeout,
  8414. pre_state,
  8415. pre_subtask_id,
  8416. )
  8417. # Phase B was entered iff subtask_id advanced, which means the
  8418. # project_file landed on the printer. A forced reconnect at this point
  8419. # would interrupt the printer's parse and trigger 0500_4003 (#1150) —
  8420. # skip the recovery entirely.
  8421. if landed_on_subtask:
  8422. return
  8423. # Phase A timeout path: if the printer's gcode_file changed since
  8424. # pre-dispatch, the project_file command landed and the printer is
  8425. # parsing — a forced reconnect mid-parse triggers 0500_4003 (#1150).
  8426. # If gcode_file is unchanged, the publish was silently swallowed
  8427. # (#887/#936) and force_reconnect recovery is what we want.
  8428. client = printer_manager.get_client(printer_id)
  8429. current_gcode_file = getattr(last_status, "gcode_file", None) if last_status else None
  8430. publish_landed = current_gcode_file is not None and current_gcode_file != pre_gcode_file
  8431. if publish_landed:
  8432. logger.warning(
  8433. "Queue item %s: gcode_file changed to %r (was %r) — printer "
  8434. "received the command and is parsing slowly. Skipping forced "
  8435. "MQTT reconnect to avoid 0500_4003 mid-parse (#1150).",
  8436. queue_item_id,
  8437. current_gcode_file,
  8438. pre_gcode_file,
  8439. )
  8440. elif client and hasattr(client, "force_reconnect_stale_session"):
  8441. client.force_reconnect_stale_session(
  8442. f"queue print command unacknowledged after {timeout:.0f}s "
  8443. f"(state still {pre_state}, gcode_file {current_gcode_file!r})"
  8444. )
  8445. # Global scheduler instance
  8446. scheduler = PrintScheduler()