| 1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401240224032404240524062407240824092410241124122413241424152416241724182419242024212422242324242425242624272428242924302431243224332434243524362437243824392440244124422443244424452446244724482449245024512452245324542455245624572458245924602461246224632464246524662467246824692470247124722473247424752476247724782479248024812482248324842485248624872488248924902491249224932494249524962497249824992500250125022503250425052506250725082509251025112512251325142515251625172518251925202521252225232524252525262527252825292530253125322533253425352536253725382539254025412542254325442545254625472548254925502551255225532554255525562557255825592560256125622563256425652566256725682569257025712572257325742575257625772578257925802581258225832584258525862587258825892590259125922593259425952596259725982599260026012602260326042605260626072608260926102611261226132614261526162617261826192620262126222623262426252626262726282629263026312632263326342635263626372638263926402641264226432644264526462647264826492650265126522653265426552656265726582659266026612662266326642665266626672668266926702671267226732674267526762677267826792680268126822683268426852686268726882689269026912692269326942695269626972698269927002701270227032704270527062707270827092710271127122713271427152716271727182719272027212722272327242725272627272728272927302731273227332734273527362737273827392740274127422743274427452746274727482749275027512752275327542755275627572758275927602761276227632764276527662767276827692770277127722773277427752776277727782779278027812782278327842785278627872788278927902791279227932794279527962797279827992800280128022803280428052806280728082809281028112812281328142815281628172818281928202821282228232824282528262827282828292830283128322833283428352836283728382839284028412842284328442845284628472848284928502851285228532854285528562857285828592860286128622863286428652866286728682869287028712872287328742875287628772878287928802881288228832884288528862887288828892890289128922893289428952896289728982899290029012902290329042905290629072908290929102911291229132914291529162917291829192920292129222923292429252926292729282929293029312932293329342935293629372938293929402941294229432944294529462947294829492950295129522953295429552956295729582959296029612962296329642965296629672968296929702971297229732974297529762977297829792980298129822983298429852986298729882989299029912992299329942995299629972998299930003001300230033004300530063007300830093010301130123013301430153016301730183019302030213022302330243025302630273028302930303031303230333034303530363037303830393040304130423043304430453046304730483049305030513052305330543055305630573058305930603061306230633064306530663067306830693070307130723073307430753076307730783079308030813082308330843085308630873088308930903091309230933094309530963097309830993100310131023103310431053106310731083109311031113112311331143115311631173118311931203121312231233124312531263127312831293130313131323133313431353136313731383139314031413142314331443145314631473148314931503151315231533154315531563157315831593160316131623163316431653166316731683169317031713172317331743175317631773178317931803181318231833184318531863187318831893190319131923193319431953196319731983199320032013202320332043205320632073208320932103211321232133214321532163217321832193220322132223223322432253226322732283229323032313232323332343235323632373238323932403241324232433244324532463247324832493250325132523253325432553256325732583259326032613262326332643265326632673268326932703271327232733274327532763277327832793280328132823283328432853286328732883289329032913292329332943295329632973298329933003301330233033304330533063307330833093310331133123313331433153316331733183319332033213322332333243325332633273328332933303331333233333334333533363337333833393340334133423343334433453346334733483349335033513352335333543355335633573358335933603361336233633364336533663367336833693370337133723373337433753376337733783379338033813382338333843385338633873388338933903391339233933394339533963397339833993400340134023403340434053406340734083409341034113412341334143415341634173418341934203421342234233424342534263427342834293430343134323433343434353436343734383439344034413442344334443445344634473448344934503451345234533454345534563457345834593460346134623463346434653466346734683469347034713472347334743475347634773478347934803481348234833484348534863487348834893490349134923493349434953496349734983499350035013502350335043505350635073508350935103511351235133514351535163517351835193520352135223523352435253526352735283529353035313532353335343535353635373538353935403541354235433544354535463547354835493550355135523553355435553556355735583559356035613562356335643565356635673568356935703571357235733574357535763577357835793580358135823583358435853586358735883589359035913592359335943595359635973598359936003601360236033604360536063607360836093610361136123613361436153616361736183619362036213622362336243625362636273628362936303631363236333634363536363637363836393640364136423643364436453646364736483649365036513652365336543655365636573658365936603661366236633664366536663667366836693670367136723673367436753676367736783679368036813682368336843685368636873688368936903691369236933694369536963697369836993700370137023703370437053706370737083709371037113712371337143715371637173718371937203721372237233724372537263727372837293730373137323733373437353736373737383739374037413742374337443745374637473748374937503751375237533754375537563757375837593760376137623763376437653766376737683769377037713772377337743775377637773778377937803781378237833784378537863787378837893790379137923793379437953796379737983799380038013802380338043805380638073808380938103811381238133814381538163817381838193820382138223823382438253826382738283829383038313832383338343835383638373838383938403841384238433844384538463847384838493850385138523853385438553856385738583859386038613862386338643865386638673868386938703871387238733874387538763877387838793880388138823883388438853886388738883889389038913892389338943895389638973898389939003901390239033904390539063907390839093910391139123913391439153916391739183919392039213922392339243925392639273928392939303931393239333934393539363937393839393940394139423943394439453946394739483949395039513952395339543955395639573958395939603961396239633964396539663967396839693970397139723973397439753976397739783979398039813982398339843985398639873988398939903991399239933994399539963997399839994000400140024003400440054006400740084009401040114012401340144015401640174018401940204021402240234024402540264027402840294030403140324033403440354036403740384039404040414042404340444045404640474048404940504051405240534054405540564057405840594060406140624063406440654066406740684069407040714072407340744075407640774078407940804081408240834084408540864087408840894090409140924093409440954096409740984099410041014102410341044105410641074108410941104111411241134114411541164117411841194120412141224123412441254126412741284129413041314132413341344135413641374138413941404141414241434144414541464147414841494150415141524153415441554156415741584159416041614162416341644165416641674168416941704171417241734174417541764177417841794180418141824183418441854186418741884189419041914192419341944195419641974198419942004201420242034204420542064207420842094210421142124213421442154216421742184219422042214222422342244225422642274228422942304231423242334234423542364237423842394240424142424243424442454246424742484249425042514252425342544255425642574258425942604261426242634264426542664267426842694270427142724273427442754276427742784279428042814282428342844285428642874288428942904291429242934294429542964297429842994300430143024303430443054306430743084309431043114312431343144315431643174318431943204321432243234324432543264327432843294330433143324333433443354336433743384339434043414342434343444345434643474348434943504351435243534354435543564357435843594360436143624363436443654366436743684369437043714372437343744375437643774378437943804381438243834384438543864387438843894390439143924393439443954396439743984399440044014402440344044405440644074408440944104411441244134414441544164417441844194420442144224423442444254426442744284429443044314432443344344435443644374438443944404441444244434444444544464447444844494450445144524453445444554456445744584459446044614462446344644465446644674468446944704471447244734474447544764477447844794480448144824483448444854486448744884489449044914492449344944495449644974498449945004501450245034504450545064507450845094510451145124513451445154516451745184519452045214522452345244525452645274528452945304531453245334534453545364537453845394540454145424543454445454546454745484549455045514552455345544555455645574558455945604561456245634564456545664567456845694570457145724573457445754576457745784579458045814582458345844585458645874588458945904591459245934594459545964597459845994600460146024603460446054606460746084609461046114612461346144615461646174618461946204621462246234624462546264627462846294630463146324633463446354636463746384639464046414642464346444645464646474648464946504651465246534654465546564657465846594660466146624663466446654666466746684669467046714672467346744675467646774678467946804681468246834684468546864687468846894690469146924693469446954696469746984699470047014702470347044705470647074708470947104711471247134714471547164717471847194720472147224723472447254726472747284729473047314732473347344735473647374738473947404741474247434744474547464747474847494750475147524753475447554756475747584759476047614762476347644765476647674768476947704771477247734774477547764777477847794780478147824783478447854786478747884789479047914792479347944795479647974798479948004801480248034804480548064807480848094810481148124813481448154816481748184819482048214822482348244825482648274828482948304831483248334834483548364837483848394840484148424843484448454846484748484849485048514852485348544855485648574858485948604861486248634864486548664867486848694870487148724873487448754876487748784879488048814882488348844885488648874888488948904891489248934894489548964897489848994900490149024903490449054906490749084909491049114912491349144915491649174918491949204921492249234924492549264927492849294930493149324933493449354936493749384939494049414942494349444945494649474948494949504951495249534954495549564957495849594960496149624963496449654966496749684969497049714972497349744975497649774978497949804981498249834984498549864987498849894990499149924993499449954996499749984999500050015002500350045005500650075008500950105011501250135014501550165017501850195020502150225023502450255026502750285029503050315032503350345035503650375038503950405041504250435044504550465047504850495050505150525053505450555056505750585059506050615062506350645065506650675068506950705071507250735074507550765077507850795080508150825083508450855086508750885089509050915092509350945095509650975098509951005101510251035104510551065107510851095110511151125113511451155116511751185119512051215122512351245125512651275128512951305131513251335134513551365137513851395140514151425143514451455146514751485149515051515152515351545155515651575158515951605161516251635164516551665167516851695170517151725173517451755176517751785179518051815182518351845185518651875188518951905191519251935194519551965197519851995200520152025203520452055206520752085209521052115212521352145215521652175218521952205221522252235224522552265227522852295230523152325233523452355236523752385239524052415242524352445245524652475248524952505251525252535254525552565257525852595260526152625263526452655266526752685269527052715272527352745275527652775278527952805281528252835284528552865287528852895290529152925293529452955296529752985299530053015302530353045305530653075308530953105311531253135314531553165317531853195320532153225323532453255326532753285329533053315332533353345335533653375338533953405341534253435344534553465347534853495350535153525353535453555356535753585359536053615362536353645365536653675368536953705371537253735374537553765377537853795380538153825383538453855386538753885389539053915392539353945395539653975398539954005401540254035404540554065407540854095410541154125413541454155416541754185419542054215422542354245425542654275428542954305431543254335434543554365437543854395440544154425443544454455446544754485449545054515452545354545455545654575458545954605461546254635464546554665467546854695470547154725473547454755476547754785479548054815482548354845485548654875488548954905491549254935494549554965497549854995500550155025503550455055506550755085509551055115512551355145515551655175518551955205521552255235524552555265527552855295530553155325533553455355536553755385539554055415542554355445545554655475548554955505551555255535554555555565557555855595560556155625563556455655566556755685569557055715572557355745575557655775578557955805581558255835584558555865587558855895590559155925593559455955596559755985599560056015602560356045605560656075608560956105611561256135614561556165617561856195620562156225623562456255626562756285629563056315632563356345635563656375638563956405641564256435644564556465647564856495650565156525653565456555656565756585659566056615662566356645665566656675668566956705671567256735674567556765677567856795680568156825683568456855686568756885689569056915692569356945695569656975698569957005701570257035704570557065707570857095710571157125713571457155716571757185719572057215722572357245725572657275728572957305731573257335734573557365737573857395740574157425743574457455746574757485749575057515752575357545755575657575758575957605761576257635764576557665767576857695770577157725773577457755776577757785779578057815782578357845785578657875788578957905791579257935794579557965797579857995800580158025803580458055806580758085809581058115812581358145815581658175818581958205821582258235824582558265827582858295830583158325833583458355836583758385839584058415842584358445845584658475848584958505851585258535854585558565857585858595860586158625863586458655866586758685869587058715872587358745875587658775878587958805881588258835884588558865887588858895890589158925893589458955896589758985899590059015902590359045905590659075908590959105911591259135914591559165917591859195920592159225923592459255926592759285929593059315932593359345935593659375938593959405941594259435944594559465947594859495950595159525953595459555956595759585959596059615962596359645965596659675968596959705971597259735974597559765977597859795980598159825983598459855986598759885989599059915992599359945995599659975998599960006001600260036004600560066007600860096010601160126013601460156016601760186019602060216022602360246025602660276028602960306031603260336034603560366037603860396040604160426043604460456046604760486049605060516052605360546055605660576058605960606061606260636064606560666067606860696070607160726073607460756076607760786079608060816082608360846085608660876088608960906091609260936094609560966097609860996100610161026103610461056106610761086109611061116112611361146115611661176118611961206121612261236124612561266127612861296130613161326133613461356136613761386139614061416142614361446145614661476148614961506151615261536154615561566157615861596160616161626163616461656166616761686169617061716172617361746175617661776178617961806181618261836184618561866187618861896190619161926193619461956196619761986199620062016202620362046205620662076208620962106211621262136214621562166217621862196220622162226223622462256226622762286229623062316232623362346235623662376238623962406241624262436244624562466247624862496250625162526253625462556256625762586259626062616262626362646265626662676268626962706271627262736274627562766277627862796280628162826283628462856286628762886289629062916292629362946295629662976298629963006301630263036304630563066307630863096310631163126313631463156316631763186319632063216322632363246325632663276328632963306331633263336334633563366337633863396340634163426343634463456346634763486349635063516352635363546355635663576358635963606361636263636364636563666367636863696370637163726373637463756376637763786379638063816382638363846385638663876388638963906391639263936394639563966397639863996400640164026403640464056406640764086409641064116412641364146415641664176418641964206421642264236424642564266427642864296430643164326433643464356436643764386439644064416442644364446445644664476448644964506451645264536454645564566457645864596460646164626463646464656466646764686469647064716472647364746475647664776478647964806481648264836484648564866487648864896490649164926493649464956496649764986499650065016502650365046505650665076508650965106511651265136514651565166517651865196520652165226523652465256526652765286529653065316532653365346535653665376538653965406541654265436544654565466547654865496550655165526553655465556556655765586559656065616562656365646565656665676568656965706571657265736574657565766577657865796580658165826583658465856586658765886589659065916592659365946595659665976598659966006601660266036604660566066607660866096610661166126613661466156616661766186619662066216622662366246625662666276628662966306631663266336634663566366637663866396640664166426643664466456646664766486649665066516652665366546655665666576658665966606661666266636664666566666667666866696670667166726673667466756676667766786679668066816682668366846685668666876688668966906691669266936694669566966697669866996700670167026703670467056706670767086709671067116712671367146715671667176718671967206721672267236724672567266727672867296730673167326733673467356736673767386739674067416742674367446745674667476748674967506751675267536754675567566757675867596760676167626763676467656766676767686769677067716772677367746775677667776778677967806781678267836784678567866787678867896790679167926793679467956796679767986799680068016802680368046805680668076808680968106811681268136814681568166817681868196820682168226823682468256826682768286829683068316832683368346835683668376838683968406841684268436844684568466847684868496850685168526853685468556856685768586859686068616862686368646865686668676868686968706871687268736874687568766877687868796880688168826883688468856886688768886889689068916892689368946895689668976898689969006901690269036904690569066907690869096910691169126913691469156916691769186919692069216922692369246925692669276928692969306931693269336934693569366937693869396940694169426943694469456946694769486949695069516952695369546955695669576958695969606961696269636964696569666967696869696970697169726973697469756976697769786979698069816982698369846985698669876988698969906991699269936994699569966997699869997000700170027003700470057006700770087009701070117012701370147015701670177018701970207021702270237024702570267027702870297030703170327033703470357036703770387039704070417042704370447045704670477048704970507051705270537054705570567057705870597060706170627063706470657066706770687069707070717072707370747075707670777078707970807081708270837084708570867087708870897090709170927093709470957096709770987099710071017102710371047105710671077108710971107111711271137114711571167117711871197120712171227123712471257126712771287129713071317132713371347135713671377138713971407141714271437144714571467147714871497150715171527153715471557156715771587159716071617162716371647165716671677168716971707171717271737174717571767177717871797180718171827183718471857186718771887189719071917192719371947195719671977198719972007201720272037204720572067207720872097210721172127213721472157216721772187219722072217222722372247225722672277228722972307231723272337234723572367237723872397240724172427243724472457246724772487249725072517252725372547255725672577258725972607261726272637264726572667267726872697270727172727273727472757276727772787279728072817282728372847285728672877288728972907291729272937294729572967297729872997300730173027303730473057306730773087309731073117312731373147315731673177318731973207321732273237324732573267327732873297330733173327333733473357336733773387339734073417342734373447345734673477348734973507351735273537354735573567357735873597360736173627363736473657366736773687369737073717372737373747375737673777378737973807381738273837384738573867387738873897390739173927393739473957396739773987399740074017402740374047405740674077408740974107411741274137414741574167417741874197420742174227423742474257426742774287429743074317432743374347435743674377438743974407441744274437444744574467447744874497450745174527453745474557456745774587459746074617462746374647465746674677468746974707471747274737474747574767477747874797480748174827483748474857486748774887489749074917492749374947495749674977498749975007501750275037504750575067507750875097510751175127513751475157516751775187519752075217522752375247525752675277528752975307531753275337534753575367537753875397540754175427543754475457546754775487549755075517552755375547555755675577558755975607561756275637564756575667567756875697570757175727573757475757576757775787579758075817582758375847585758675877588758975907591759275937594759575967597759875997600760176027603760476057606760776087609761076117612761376147615761676177618761976207621762276237624762576267627762876297630763176327633763476357636763776387639764076417642764376447645764676477648764976507651765276537654765576567657765876597660766176627663766476657666766776687669767076717672767376747675767676777678767976807681768276837684768576867687768876897690769176927693769476957696769776987699770077017702770377047705770677077708770977107711771277137714771577167717771877197720772177227723772477257726772777287729773077317732773377347735773677377738773977407741774277437744774577467747774877497750775177527753775477557756775777587759776077617762776377647765776677677768776977707771777277737774777577767777777877797780778177827783778477857786778777887789779077917792779377947795779677977798779978007801780278037804780578067807780878097810781178127813781478157816781778187819782078217822782378247825782678277828782978307831783278337834783578367837783878397840784178427843784478457846784778487849785078517852785378547855785678577858785978607861786278637864786578667867786878697870787178727873787478757876787778787879788078817882788378847885788678877888788978907891789278937894789578967897789878997900790179027903790479057906790779087909791079117912791379147915791679177918791979207921792279237924792579267927792879297930793179327933793479357936793779387939794079417942794379447945794679477948794979507951795279537954795579567957795879597960796179627963796479657966796779687969797079717972797379747975797679777978797979807981798279837984798579867987798879897990799179927993799479957996799779987999800080018002800380048005800680078008800980108011801280138014801580168017801880198020802180228023802480258026802780288029803080318032803380348035803680378038803980408041804280438044804580468047804880498050805180528053805480558056805780588059806080618062806380648065806680678068806980708071807280738074807580768077807880798080808180828083808480858086808780888089809080918092809380948095809680978098809981008101810281038104810581068107810881098110811181128113811481158116811781188119812081218122812381248125812681278128812981308131813281338134813581368137813881398140814181428143814481458146814781488149815081518152815381548155815681578158815981608161816281638164816581668167816881698170817181728173817481758176817781788179818081818182818381848185818681878188818981908191819281938194819581968197819881998200820182028203820482058206820782088209821082118212821382148215821682178218821982208221822282238224822582268227822882298230823182328233823482358236823782388239824082418242824382448245824682478248824982508251825282538254825582568257825882598260826182628263826482658266826782688269827082718272827382748275827682778278827982808281828282838284828582868287828882898290829182928293829482958296829782988299830083018302830383048305830683078308830983108311831283138314831583168317831883198320832183228323832483258326832783288329833083318332833383348335833683378338833983408341834283438344834583468347834883498350835183528353835483558356835783588359836083618362836383648365836683678368836983708371837283738374837583768377837883798380838183828383838483858386838783888389839083918392839383948395839683978398839984008401840284038404840584068407840884098410841184128413841484158416841784188419842084218422842384248425842684278428842984308431843284338434843584368437843884398440844184428443844484458446844784488449845084518452845384548455845684578458845984608461846284638464846584668467846884698470847184728473847484758476847784788479848084818482848384848485848684878488848984908491849284938494849584968497849884998500850185028503850485058506850785088509851085118512851385148515851685178518851985208521852285238524852585268527852885298530853185328533853485358536853785388539854085418542854385448545854685478548854985508551855285538554855585568557855885598560856185628563856485658566856785688569857085718572857385748575857685778578857985808581858285838584858585868587858885898590859185928593859485958596859785988599860086018602860386048605860686078608860986108611861286138614861586168617861886198620862186228623862486258626862786288629863086318632863386348635863686378638863986408641864286438644864586468647864886498650865186528653865486558656865786588659866086618662866386648665866686678668866986708671867286738674867586768677867886798680868186828683868486858686868786888689869086918692869386948695869686978698869987008701870287038704870587068707870887098710871187128713871487158716871787188719872087218722872387248725872687278728872987308731873287338734873587368737873887398740874187428743874487458746874787488749875087518752875387548755875687578758875987608761876287638764876587668767876887698770877187728773877487758776877787788779878087818782878387848785878687878788878987908791879287938794879587968797879887998800880188028803880488058806880788088809881088118812881388148815881688178818881988208821882288238824882588268827882888298830883188328833883488358836883788388839884088418842884388448845884688478848884988508851885288538854885588568857885888598860886188628863886488658866886788688869887088718872887388748875887688778878887988808881888288838884888588868887888888898890889188928893889488958896889788988899890089018902890389048905890689078908890989108911891289138914891589168917891889198920892189228923892489258926892789288929893089318932893389348935893689378938893989408941894289438944894589468947894889498950895189528953895489558956895789588959896089618962896389648965896689678968896989708971897289738974897589768977897889798980898189828983898489858986898789888989899089918992899389948995899689978998899990009001900290039004900590069007900890099010901190129013901490159016901790189019902090219022902390249025902690279028902990309031903290339034903590369037903890399040904190429043904490459046904790489049905090519052905390549055905690579058905990609061906290639064906590669067906890699070907190729073907490759076907790789079908090819082908390849085908690879088908990909091909290939094909590969097909890999100910191029103910491059106910791089109911091119112911391149115911691179118911991209121912291239124912591269127912891299130913191329133913491359136913791389139914091419142914391449145914691479148914991509151915291539154915591569157915891599160916191629163916491659166916791689169917091719172917391749175917691779178917991809181918291839184918591869187918891899190919191929193919491959196919791989199920092019202920392049205920692079208920992109211921292139214921592169217921892199220922192229223922492259226922792289229923092319232923392349235923692379238923992409241924292439244924592469247924892499250925192529253925492559256925792589259926092619262926392649265926692679268926992709271927292739274927592769277927892799280928192829283928492859286928792889289929092919292929392949295929692979298929993009301930293039304930593069307930893099310931193129313931493159316931793189319932093219322932393249325932693279328932993309331933293339334933593369337933893399340934193429343934493459346934793489349935093519352935393549355935693579358935993609361936293639364936593669367936893699370937193729373937493759376937793789379938093819382938393849385938693879388938993909391939293939394939593969397939893999400940194029403940494059406940794089409941094119412941394149415941694179418941994209421942294239424942594269427942894299430943194329433943494359436943794389439944094419442944394449445944694479448944994509451945294539454945594569457945894599460946194629463946494659466946794689469947094719472947394749475947694779478947994809481948294839484948594869487948894899490949194929493949494959496949794989499950095019502950395049505950695079508950995109511951295139514951595169517951895199520952195229523952495259526952795289529953095319532953395349535953695379538953995409541954295439544954595469547954895499550955195529553955495559556955795589559956095619562956395649565956695679568956995709571957295739574957595769577957895799580958195829583958495859586958795889589959095919592959395949595959695979598959996009601960296039604960596069607960896099610961196129613961496159616961796189619962096219622962396249625962696279628962996309631963296339634963596369637963896399640964196429643964496459646964796489649965096519652965396549655965696579658965996609661966296639664966596669667966896699670967196729673967496759676967796789679968096819682968396849685968696879688968996909691969296939694969596969697969896999700970197029703970497059706970797089709971097119712971397149715971697179718971997209721972297239724972597269727972897299730973197329733973497359736973797389739974097419742974397449745974697479748974997509751975297539754975597569757975897599760976197629763976497659766976797689769977097719772977397749775977697779778977997809781978297839784978597869787978897899790979197929793979497959796979797989799980098019802980398049805980698079808980998109811981298139814981598169817981898199820982198229823982498259826982798289829983098319832983398349835983698379838983998409841984298439844984598469847984898499850985198529853985498559856985798589859986098619862986398649865986698679868986998709871987298739874987598769877987898799880988198829883988498859886988798889889989098919892989398949895989698979898989999009901990299039904990599069907990899099910991199129913991499159916991799189919992099219922992399249925992699279928992999309931993299339934993599369937993899399940994199429943994499459946994799489949995099519952995399549955995699579958995999609961996299639964996599669967996899699970997199729973997499759976997799789979998099819982998399849985998699879988998999909991999299939994999599969997999899991000010001100021000310004100051000610007100081000910010100111001210013100141001510016100171001810019100201002110022100231002410025100261002710028100291003010031100321003310034100351003610037100381003910040100411004210043100441004510046100471004810049100501005110052100531005410055100561005710058100591006010061100621006310064100651006610067100681006910070100711007210073100741007510076100771007810079100801008110082100831008410085100861008710088100891009010091100921009310094100951009610097100981009910100101011010210103101041010510106101071010810109101101011110112101131011410115101161011710118101191012010121101221012310124101251012610127101281012910130101311013210133101341013510136101371013810139101401014110142101431014410145101461014710148101491015010151101521015310154101551015610157101581015910160101611016210163101641016510166101671016810169101701017110172101731017410175101761017710178101791018010181101821018310184101851018610187101881018910190101911019210193101941019510196101971019810199102001020110202102031020410205102061020710208102091021010211102121021310214102151021610217102181021910220102211022210223102241022510226102271022810229102301023110232102331023410235102361023710238102391024010241102421024310244102451024610247102481024910250102511025210253102541025510256102571025810259102601026110262102631026410265102661026710268102691027010271102721027310274102751027610277102781027910280102811028210283102841028510286102871028810289102901029110292102931029410295102961029710298102991030010301103021030310304103051030610307103081030910310103111031210313103141031510316103171031810319103201032110322103231032410325103261032710328103291033010331103321033310334103351033610337103381033910340103411034210343103441034510346103471034810349103501035110352103531035410355103561035710358103591036010361103621036310364103651036610367103681036910370103711037210373103741037510376103771037810379103801038110382103831038410385103861038710388103891039010391103921039310394103951039610397103981039910400104011040210403104041040510406104071040810409104101041110412104131041410415104161041710418104191042010421104221042310424104251042610427104281042910430104311043210433104341043510436104371043810439104401044110442104431044410445104461044710448104491045010451104521045310454104551045610457104581045910460104611046210463104641046510466104671046810469104701047110472104731047410475104761047710478104791048010481104821048310484104851048610487104881048910490104911049210493104941049510496104971049810499105001050110502105031050410505105061050710508105091051010511105121051310514105151051610517105181051910520105211052210523105241052510526105271052810529105301053110532105331053410535105361053710538105391054010541105421054310544105451054610547105481054910550105511055210553105541055510556105571055810559105601056110562105631056410565105661056710568105691057010571105721057310574105751057610577105781057910580105811058210583105841058510586105871058810589105901059110592105931059410595105961059710598105991060010601106021060310604106051060610607106081060910610106111061210613106141061510616106171061810619106201062110622106231062410625106261062710628106291063010631106321063310634106351063610637106381063910640106411064210643106441064510646106471064810649106501065110652106531065410655106561065710658106591066010661106621066310664106651066610667106681066910670106711067210673106741067510676106771067810679106801068110682106831068410685106861068710688106891069010691106921069310694106951069610697106981069910700107011070210703107041070510706107071070810709107101071110712107131071410715107161071710718107191072010721107221072310724107251072610727107281072910730107311073210733107341073510736107371073810739107401074110742107431074410745107461074710748107491075010751107521075310754107551075610757107581075910760107611076210763107641076510766107671076810769107701077110772107731077410775107761077710778107791078010781107821078310784107851078610787107881078910790107911079210793107941079510796107971079810799108001080110802108031080410805108061080710808108091081010811108121081310814108151081610817108181081910820108211082210823108241082510826108271082810829108301083110832108331083410835108361083710838108391084010841108421084310844108451084610847108481084910850108511085210853108541085510856108571085810859108601086110862108631086410865108661086710868108691087010871108721087310874108751087610877108781087910880108811088210883108841088510886108871088810889108901089110892108931089410895108961089710898108991090010901109021090310904109051090610907109081090910910109111091210913109141091510916109171091810919109201092110922109231092410925109261092710928109291093010931109321093310934109351093610937109381093910940109411094210943109441094510946109471094810949109501095110952109531095410955109561095710958109591096010961109621096310964109651096610967109681096910970109711097210973109741097510976109771097810979109801098110982109831098410985109861098710988109891099010991109921099310994109951099610997109981099911000110011100211003110041100511006110071100811009110101101111012110131101411015110161101711018110191102011021110221102311024110251102611027110281102911030110311103211033110341103511036110371103811039110401104111042110431104411045110461104711048110491105011051110521105311054110551105611057110581105911060110611106211063110641106511066110671106811069110701107111072110731107411075110761107711078110791108011081110821108311084110851108611087110881108911090110911109211093110941109511096110971109811099111001110111102111031110411105111061110711108111091111011111111121111311114111151111611117111181111911120111211112211123111241112511126111271112811129111301113111132111331113411135111361113711138111391114011141111421114311144111451114611147111481114911150111511115211153111541115511156111571115811159111601116111162111631116411165111661116711168111691117011171111721117311174111751117611177111781117911180111811118211183111841118511186111871118811189111901119111192111931119411195111961119711198111991120011201112021120311204112051120611207112081120911210112111121211213112141121511216112171121811219112201122111222112231122411225112261122711228112291123011231112321123311234112351123611237112381123911240112411124211243112441124511246112471124811249112501125111252112531125411255112561125711258112591126011261112621126311264112651126611267112681126911270112711127211273112741127511276112771127811279112801128111282112831128411285112861128711288112891129011291112921129311294112951129611297112981129911300113011130211303113041130511306113071130811309113101131111312113131131411315113161131711318113191132011321113221132311324113251132611327113281132911330113311133211333113341133511336113371133811339113401134111342113431134411345113461134711348113491135011351113521135311354113551135611357113581135911360113611136211363113641136511366113671136811369113701137111372113731137411375113761137711378113791138011381113821138311384113851138611387113881138911390113911139211393113941139511396113971139811399114001140111402114031140411405114061140711408114091141011411114121141311414114151141611417114181141911420114211142211423114241142511426114271142811429114301143111432114331143411435114361143711438114391144011441114421144311444114451144611447114481144911450114511145211453114541145511456114571145811459114601146111462114631146411465114661146711468114691147011471114721147311474114751147611477114781147911480114811148211483114841148511486114871148811489114901149111492114931149411495114961149711498114991150011501115021150311504115051150611507115081150911510115111151211513115141151511516115171151811519115201152111522115231152411525115261152711528115291153011531115321153311534115351153611537115381153911540115411154211543115441154511546115471154811549115501155111552115531155411555115561155711558115591156011561115621156311564115651156611567115681156911570115711157211573115741157511576115771157811579115801158111582115831158411585115861158711588115891159011591115921159311594115951159611597115981159911600116011160211603116041160511606116071160811609116101161111612116131161411615116161161711618116191162011621116221162311624116251162611627116281162911630116311163211633116341163511636116371163811639116401164111642116431164411645116461164711648116491165011651116521165311654116551165611657116581165911660116611166211663116641166511666116671166811669116701167111672116731167411675116761167711678116791168011681116821168311684116851168611687116881168911690116911169211693116941169511696116971169811699117001170111702117031170411705117061170711708117091171011711117121171311714117151171611717117181171911720117211172211723117241172511726117271172811729117301173111732117331173411735117361173711738117391174011741117421174311744117451174611747117481174911750117511175211753117541175511756117571175811759117601176111762117631176411765117661176711768117691177011771117721177311774117751177611777117781177911780117811178211783117841178511786117871178811789117901179111792117931179411795117961179711798117991180011801118021180311804118051180611807118081180911810118111181211813118141181511816118171181811819118201182111822118231182411825118261182711828118291183011831118321183311834118351183611837118381183911840118411184211843118441184511846118471184811849118501185111852118531185411855118561185711858118591186011861118621186311864118651186611867118681186911870118711187211873118741187511876118771187811879118801188111882118831188411885118861188711888118891189011891118921189311894118951189611897118981189911900119011190211903119041190511906119071190811909119101191111912119131191411915119161191711918119191192011921119221192311924119251192611927119281192911930119311193211933119341193511936119371193811939119401194111942119431194411945119461194711948119491195011951119521195311954119551195611957119581195911960119611196211963119641196511966119671196811969119701197111972119731197411975119761197711978119791198011981119821198311984119851198611987119881198911990119911199211993119941199511996119971199811999120001200112002120031200412005120061200712008120091201012011120121201312014120151201612017120181201912020120211202212023120241202512026120271202812029120301203112032120331203412035120361203712038120391204012041120421204312044120451204612047120481204912050120511205212053120541205512056120571205812059120601206112062120631206412065120661206712068120691207012071120721207312074120751207612077120781207912080120811208212083120841208512086120871208812089120901209112092120931209412095120961209712098120991210012101121021210312104121051210612107121081210912110121111211212113121141211512116121171211812119121201212112122121231212412125121261212712128121291213012131121321213312134121351213612137121381213912140121411214212143121441214512146121471214812149121501215112152121531215412155121561215712158121591216012161121621216312164121651216612167121681216912170121711217212173121741217512176121771217812179121801218112182121831218412185121861218712188121891219012191121921219312194121951219612197121981219912200122011220212203122041220512206122071220812209122101221112212122131221412215122161221712218122191222012221122221222312224122251222612227122281222912230122311223212233122341223512236122371223812239122401224112242122431224412245122461224712248122491225012251122521225312254122551225612257122581225912260122611226212263122641226512266122671226812269122701227112272122731227412275122761227712278122791228012281122821228312284122851228612287122881228912290122911229212293122941229512296122971229812299123001230112302123031230412305123061230712308123091231012311123121231312314123151231612317123181231912320123211232212323123241232512326123271232812329123301233112332123331233412335123361233712338123391234012341123421234312344123451234612347123481234912350123511235212353123541235512356123571235812359123601236112362123631236412365123661236712368123691237012371123721237312374123751237612377123781237912380123811238212383123841238512386123871238812389123901239112392123931239412395123961239712398123991240012401124021240312404124051240612407124081240912410124111241212413124141241512416124171241812419124201242112422124231242412425124261242712428124291243012431124321243312434124351243612437124381243912440124411244212443124441244512446124471244812449124501245112452124531245412455124561245712458124591246012461124621246312464124651246612467124681246912470124711247212473124741247512476124771247812479124801248112482124831248412485124861248712488124891249012491124921249312494124951249612497124981249912500125011250212503125041250512506125071250812509125101251112512125131251412515125161251712518125191252012521125221252312524125251252612527125281252912530125311253212533125341253512536125371253812539125401254112542125431254412545125461254712548125491255012551125521255312554125551255612557125581255912560125611256212563125641256512566125671256812569125701257112572125731257412575125761257712578125791258012581125821258312584125851258612587125881258912590125911259212593125941259512596125971259812599126001260112602126031260412605126061260712608126091261012611126121261312614126151261612617126181261912620126211262212623126241262512626126271262812629126301263112632126331263412635126361263712638126391264012641126421264312644126451264612647126481264912650126511265212653126541265512656126571265812659126601266112662126631266412665126661266712668126691267012671126721267312674126751267612677126781267912680126811268212683126841268512686126871268812689126901269112692126931269412695126961269712698126991270012701127021270312704127051270612707127081270912710127111271212713127141271512716127171271812719127201272112722127231272412725127261272712728127291273012731127321273312734127351273612737127381273912740127411274212743127441274512746127471274812749127501275112752127531275412755127561275712758127591276012761127621276312764127651276612767127681276912770127711277212773127741277512776127771277812779127801278112782127831278412785127861278712788127891279012791127921279312794127951279612797127981279912800128011280212803128041280512806128071280812809128101281112812128131281412815128161281712818128191282012821128221282312824128251282612827128281282912830128311283212833128341283512836128371283812839128401284112842128431284412845128461284712848128491285012851128521285312854128551285612857128581285912860128611286212863128641286512866128671286812869128701287112872128731287412875128761287712878128791288012881128821288312884128851288612887128881288912890128911289212893128941289512896128971289812899129001290112902129031290412905129061290712908129091291012911129121291312914129151291612917129181291912920129211292212923129241292512926129271292812929129301293112932129331293412935129361293712938129391294012941129421294312944129451294612947129481294912950129511295212953129541295512956129571295812959129601296112962129631296412965129661296712968129691297012971129721297312974129751297612977129781297912980129811298212983129841298512986129871298812989129901299112992129931299412995129961299712998129991300013001130021300313004130051300613007130081300913010130111301213013130141301513016130171301813019130201302113022130231302413025130261302713028130291303013031130321303313034130351303613037130381303913040130411304213043130441304513046130471304813049130501305113052130531305413055130561305713058130591306013061130621306313064130651306613067130681306913070130711307213073130741307513076130771307813079130801308113082130831308413085130861308713088130891309013091130921309313094130951309613097130981309913100131011310213103131041310513106131071310813109131101311113112131131311413115131161311713118131191312013121131221312313124131251312613127131281312913130131311313213133131341313513136131371313813139131401314113142131431314413145131461314713148131491315013151131521315313154131551315613157131581315913160131611316213163131641316513166131671316813169131701317113172131731317413175131761317713178131791318013181131821318313184131851318613187131881318913190131911319213193131941319513196131971319813199132001320113202132031320413205132061320713208132091321013211132121321313214132151321613217132181321913220132211322213223132241322513226132271322813229132301323113232132331323413235132361323713238132391324013241132421324313244132451324613247132481324913250132511325213253132541325513256132571325813259132601326113262132631326413265132661326713268132691327013271132721327313274132751327613277132781327913280132811328213283132841328513286132871328813289132901329113292132931329413295132961329713298132991330013301133021330313304133051330613307133081330913310133111331213313133141331513316133171331813319133201332113322133231332413325133261332713328133291333013331133321333313334133351333613337133381333913340133411334213343133441334513346133471334813349133501335113352133531335413355133561335713358133591336013361133621336313364133651336613367133681336913370133711337213373133741337513376133771337813379133801338113382133831338413385133861338713388133891339013391133921339313394133951339613397133981339913400134011340213403134041340513406134071340813409134101341113412134131341413415134161341713418134191342013421134221342313424134251342613427134281342913430134311343213433134341343513436134371343813439134401344113442134431344413445134461344713448134491345013451134521345313454134551345613457134581345913460134611346213463134641346513466134671346813469134701347113472134731347413475134761347713478134791348013481134821348313484134851348613487134881348913490134911349213493134941349513496134971349813499135001350113502135031350413505135061350713508 |
- /*
- MIT License http://www.opensource.org/licenses/mit-license.php
- Author Raj Aryan (based on SWC parser by Alexander Akait)
- */
- "use strict";
- const {
- CSS_TYPE,
- HTML_TYPE,
- JAVASCRIPT_TYPE
- } = require("../ModuleSourceTypeConstants");
- const GenericSourceProcessor = require("../util/SourceProcessor");
- const { deferredWrite } = GenericSourceProcessor;
- /** @import { CssEnvironment, CssProcessOptions, CssTransformOptions } from "../css/syntax" */
- /** @import { OutputHtmlOptions } from "../../declarations/WebpackOptions" */
- /**
- * Renders one nested body this document embeds — an inline `<style>`, every
- * `style=""` (handed over as a whole stylesheet), a `<script>` holding JSON or
- * JavaScript, an `<svg>` subtree, and the document an `<iframe srcdoc>` holds. Returning the input unchanged declines
- * it. Synchronous: it is called from inside the print walk, so a caller with an
- * async minifier collects the bodies in a walk-only pass first and looks the
- * results up here.
- *
- * What comes back is spliced in as-is, so an `<svg>` renderer must return
- * markup — everything else is literal text. A `<style>` / `<script>` body may
- * therefore carry its own `sourceMappingURL`, which is the only way a map
- * reaches inline content; nothing else here can hold one.
- * @typedef {(source: string, info: { type: string, hostType: string, as?: string }) => string | undefined} EmbeddedSourceRenderer
- */
- /**
- * One embedded source recorded for a caller that can only answer
- * asynchronously, and the text to print once it has.
- * @typedef {import("../util/dataURL").DeferredEmbeddedSource} DeferredEmbeddedSource
- */
- // Neither a JSON `<script>` body nor an `<svg>` subtree is a module source
- // type, so neither has a constant in `ModuleSourceTypeConstants` — they name
- // themselves here.
- const JSON_TYPE = "json";
- const SVG_TYPE = "svg";
- // A `style=""` holds a block's contents, not a stylesheet. It is still CSS, so
- // it is offered as CSS — with `as` saying which production it is, the same
- // vocabulary `module.parser.css.as` uses.
- const BLOCK_CONTENTS = "block-contents";
- const {
- askEmbeddedRenderer,
- collectEmbeddedDiagnostics,
- embeddedText
- } = require("../util/dataURL");
- // Every language `renderEmbeddedSource` is offered from a document, in the
- // order the offer sites below reach them. A new site adds its language here, so
- // a caller that has to say up front which it handles reads one list.
- const EMBEDDED_LANGUAGES = [
- CSS_TYPE,
- JAVASCRIPT_TYPE,
- JSON_TYPE,
- SVG_TYPE,
- HTML_TYPE
- ];
- const {
- ADDRESS_DIV_P,
- APPLET_MARQUEE_OBJECT,
- BLOCK_END,
- BLOCK_LEVEL_ELEMENTS,
- BLOCK_START,
- BODY_HTML_BR,
- BODY_START_KEPT_BEFORE,
- BOOLEAN_ATTRIBUTES,
- CAPTION_IGNORED_ENDS,
- CAPTION_TABLE_STARTS,
- CELL_IGNORED_ENDS,
- CLEAR_TABLE,
- CLEAR_TABLE_BODY,
- CLEAR_TABLE_ROW,
- COMMA_LIST_ATTRIBUTES,
- DOM_TOKEN_LIST_ATTRIBUTES,
- EMPTY_ELEMENT_KEPT,
- EMPTY_METADATA_ELEMENTS,
- EMPTY_REMOVABLE_ATTRIBUTES,
- ENUMERATED_ATTRIBUTE_NAMES,
- ENUMERATED_KEYWORDS,
- FONT_BREAKOUT_ATTRS,
- FOREIGN_ATTR_NS,
- FOREIGN_BREAKOUT,
- FORMATTING,
- FOSTER_KEPT,
- HEADING,
- HEAD_BODY_HTML_BR,
- HEAD_ELEMENTS,
- HEAD_VOID_ELEMENTS,
- HTML_SCOPE,
- IGNORED_BODY_TABLE_STARTS,
- IMPLIED,
- IMPLIED_THOROUGH,
- INTEGER_ATTRIBUTES,
- IN_HEAD_NOSCRIPT_PASSTHROUGH,
- IN_TABLE_IGNORED_ENDS,
- JAVASCRIPT_SCRIPT_TYPES,
- JSON_SCRIPT_TYPES,
- LEADING_NEWLINE_ELEMENTS,
- LITERAL_TEXT_PARENTS,
- MATHML_SPECIAL,
- MATHML_TEXT_INTEGRATION,
- NOFRAMES_STYLE_NOSCRIPT,
- NO_DECODE_TEXT,
- OPTIONAL_END_TAG_AT_END,
- OPTIONAL_END_TAG_FOLLOWERS,
- OPTIONAL_END_TAG_UNLESS_TRAILING_NODE,
- PARAM_SOURCE_TRACK,
- P_ENDS_ON_PARENT_END_TAG,
- QUIRKY_EXACT,
- QUIRKY_PREFIXES,
- RAW_TEXT_ELEMENTS,
- REDUNDANT_DEFAULT_ATTRIBUTES,
- REDUNDANT_TYPE_ATTRIBUTES,
- RESERVED_ELEMENT_NAMES,
- REWRITABLE_ATTRIBUTES,
- ROW_IGNORED_ENDS,
- ROW_TRIGGER_STARTS,
- SCOPE_ENDED_BY,
- SHADOW_HOSTS,
- SHELL_ELEMENTS,
- SIGNED_INTEGER_ATTRIBUTES,
- SPECIAL,
- SRCSET_ATTRIBUTES,
- STYLE_SCRIPT_TEMPLATE,
- SVG_ATTR_ADJUST,
- SVG_SPECIAL,
- SVG_TAG_ADJUST,
- TABLE_CONTEXT,
- TABLE_SCOPE_STOP,
- TBODY_GROUP,
- TBODY_IGNORED_ENDS,
- TBODY_TRIGGER_STARTS,
- TD_TH,
- TD_TH_TR,
- TEMPLATE_START_TAG_MODES: TEMPLATE_START_TAG_MODE_NAMES,
- TOKEN_LIST_ATTRIBUTES,
- TRANSPARENT_IMPLIED_ELEMENTS,
- URL_ATTRIBUTES,
- VOID,
- VOID_FORMATTING
- } = require("./data");
- // cspell:ignore apos notpre noncharacter noncharacters DFFF FFFE CCLS ALNUM
- // #region html entities
- // The contents of this region are auto-generated by
- // `tooling/generate-html-entities.js` from `tooling/html-entities.json`.
- // Do not edit by hand — re-run the generator (via `yarn fix:special`) to refresh.
- //
- // WHATWG named character references. Keys are entity names WITHOUT the
- // leading `&` (some end with `;`, others omit it for legacy entities that
- // match without a closing semicolon). Values are the decoded character
- // strings (1–2 UTF-16 code units).
- // Built on a null prototype so bracket lookups (`HTML_ENTITIES[name]`)
- // can't be poisoned by inherited `Object.prototype` keys like `toString`,
- // `constructor`, or `__proto__` — without this, `&toString;` would falsely
- // look like a matched named character reference.
- // prettier-ignore
- // cspell:disable-next-line
- const HTML_ENTITIES = /** @type {Readonly<Record<string, string>>} */ (Object.freeze(Object.assign(Object.create(null), {"AElig":"Æ","AElig;":"Æ","AMP":"&","AMP;":"&","Aacute":"Á","Aacute;":"Á","Abreve;":"Ă","Acirc":"Â","Acirc;":"Â","Acy;":"А","Afr;":"𝔄","Agrave":"À","Agrave;":"À","Alpha;":"Α","Amacr;":"Ā","And;":"⩓","Aogon;":"Ą","Aopf;":"𝔸","ApplyFunction;":"","Aring":"Å","Aring;":"Å","Ascr;":"𝒜","Assign;":"≔","Atilde":"Ã","Atilde;":"Ã","Auml":"Ä","Auml;":"Ä","Backslash;":"∖","Barv;":"⫧","Barwed;":"⌆","Bcy;":"Б","Because;":"∵","Bernoullis;":"ℬ","Beta;":"Β","Bfr;":"𝔅","Bopf;":"𝔹","Breve;":"˘","Bscr;":"ℬ","Bumpeq;":"≎","CHcy;":"Ч","COPY":"©","COPY;":"©","Cacute;":"Ć","Cap;":"⋒","CapitalDifferentialD;":"ⅅ","Cayleys;":"ℭ","Ccaron;":"Č","Ccedil":"Ç","Ccedil;":"Ç","Ccirc;":"Ĉ","Cconint;":"∰","Cdot;":"Ċ","Cedilla;":"¸","CenterDot;":"·","Cfr;":"ℭ","Chi;":"Χ","CircleDot;":"⊙","CircleMinus;":"⊖","CirclePlus;":"⊕","CircleTimes;":"⊗","ClockwiseContourIntegral;":"∲","CloseCurlyDoubleQuote;":"”","CloseCurlyQuote;":"’","Colon;":"∷","Colone;":"⩴","Congruent;":"≡","Conint;":"∯","ContourIntegral;":"∮","Copf;":"ℂ","Coproduct;":"∐","CounterClockwiseContourIntegral;":"∳","Cross;":"⨯","Cscr;":"𝒞","Cup;":"⋓","CupCap;":"≍","DD;":"ⅅ","DDotrahd;":"⤑","DJcy;":"Ђ","DScy;":"Ѕ","DZcy;":"Џ","Dagger;":"‡","Darr;":"↡","Dashv;":"⫤","Dcaron;":"Ď","Dcy;":"Д","Del;":"∇","Delta;":"Δ","Dfr;":"𝔇","DiacriticalAcute;":"´","DiacriticalDot;":"˙","DiacriticalDoubleAcute;":"˝","DiacriticalGrave;":"`","DiacriticalTilde;":"˜","Diamond;":"⋄","DifferentialD;":"ⅆ","Dopf;":"𝔻","Dot;":"¨","DotDot;":"⃜","DotEqual;":"≐","DoubleContourIntegral;":"∯","DoubleDot;":"¨","DoubleDownArrow;":"⇓","DoubleLeftArrow;":"⇐","DoubleLeftRightArrow;":"⇔","DoubleLeftTee;":"⫤","DoubleLongLeftArrow;":"⟸","DoubleLongLeftRightArrow;":"⟺","DoubleLongRightArrow;":"⟹","DoubleRightArrow;":"⇒","DoubleRightTee;":"⊨","DoubleUpArrow;":"⇑","DoubleUpDownArrow;":"⇕","DoubleVerticalBar;":"∥","DownArrow;":"↓","DownArrowBar;":"⤓","DownArrowUpArrow;":"⇵","DownBreve;":"̑","DownLeftRightVector;":"⥐","DownLeftTeeVector;":"⥞","DownLeftVector;":"↽","DownLeftVectorBar;":"⥖","DownRightTeeVector;":"⥟","DownRightVector;":"⇁","DownRightVectorBar;":"⥗","DownTee;":"⊤","DownTeeArrow;":"↧","Downarrow;":"⇓","Dscr;":"𝒟","Dstrok;":"Đ","ENG;":"Ŋ","ETH":"Ð","ETH;":"Ð","Eacute":"É","Eacute;":"É","Ecaron;":"Ě","Ecirc":"Ê","Ecirc;":"Ê","Ecy;":"Э","Edot;":"Ė","Efr;":"𝔈","Egrave":"È","Egrave;":"È","Element;":"∈","Emacr;":"Ē","EmptySmallSquare;":"◻","EmptyVerySmallSquare;":"▫","Eogon;":"Ę","Eopf;":"𝔼","Epsilon;":"Ε","Equal;":"⩵","EqualTilde;":"≂","Equilibrium;":"⇌","Escr;":"ℰ","Esim;":"⩳","Eta;":"Η","Euml":"Ë","Euml;":"Ë","Exists;":"∃","ExponentialE;":"ⅇ","Fcy;":"Ф","Ffr;":"𝔉","FilledSmallSquare;":"◼","FilledVerySmallSquare;":"▪","Fopf;":"𝔽","ForAll;":"∀","Fouriertrf;":"ℱ","Fscr;":"ℱ","GJcy;":"Ѓ","GT":">","GT;":">","Gamma;":"Γ","Gammad;":"Ϝ","Gbreve;":"Ğ","Gcedil;":"Ģ","Gcirc;":"Ĝ","Gcy;":"Г","Gdot;":"Ġ","Gfr;":"𝔊","Gg;":"⋙","Gopf;":"𝔾","GreaterEqual;":"≥","GreaterEqualLess;":"⋛","GreaterFullEqual;":"≧","GreaterGreater;":"⪢","GreaterLess;":"≷","GreaterSlantEqual;":"⩾","GreaterTilde;":"≳","Gscr;":"𝒢","Gt;":"≫","HARDcy;":"Ъ","Hacek;":"ˇ","Hat;":"^","Hcirc;":"Ĥ","Hfr;":"ℌ","HilbertSpace;":"ℋ","Hopf;":"ℍ","HorizontalLine;":"─","Hscr;":"ℋ","Hstrok;":"Ħ","HumpDownHump;":"≎","HumpEqual;":"≏","IEcy;":"Е","IJlig;":"IJ","IOcy;":"Ё","Iacute":"Í","Iacute;":"Í","Icirc":"Î","Icirc;":"Î","Icy;":"И","Idot;":"İ","Ifr;":"ℑ","Igrave":"Ì","Igrave;":"Ì","Im;":"ℑ","Imacr;":"Ī","ImaginaryI;":"ⅈ","Implies;":"⇒","Int;":"∬","Integral;":"∫","Intersection;":"⋂","InvisibleComma;":"","InvisibleTimes;":"","Iogon;":"Į","Iopf;":"𝕀","Iota;":"Ι","Iscr;":"ℐ","Itilde;":"Ĩ","Iukcy;":"І","Iuml":"Ï","Iuml;":"Ï","Jcirc;":"Ĵ","Jcy;":"Й","Jfr;":"𝔍","Jopf;":"𝕁","Jscr;":"𝒥","Jsercy;":"Ј","Jukcy;":"Є","KHcy;":"Х","KJcy;":"Ќ","Kappa;":"Κ","Kcedil;":"Ķ","Kcy;":"К","Kfr;":"𝔎","Kopf;":"𝕂","Kscr;":"𝒦","LJcy;":"Љ","LT":"<","LT;":"<","Lacute;":"Ĺ","Lambda;":"Λ","Lang;":"⟪","Laplacetrf;":"ℒ","Larr;":"↞","Lcaron;":"Ľ","Lcedil;":"Ļ","Lcy;":"Л","LeftAngleBracket;":"⟨","LeftArrow;":"←","LeftArrowBar;":"⇤","LeftArrowRightArrow;":"⇆","LeftCeiling;":"⌈","LeftDoubleBracket;":"⟦","LeftDownTeeVector;":"⥡","LeftDownVector;":"⇃","LeftDownVectorBar;":"⥙","LeftFloor;":"⌊","LeftRightArrow;":"↔","LeftRightVector;":"⥎","LeftTee;":"⊣","LeftTeeArrow;":"↤","LeftTeeVector;":"⥚","LeftTriangle;":"⊲","LeftTriangleBar;":"⧏","LeftTriangleEqual;":"⊴","LeftUpDownVector;":"⥑","LeftUpTeeVector;":"⥠","LeftUpVector;":"↿","LeftUpVectorBar;":"⥘","LeftVector;":"↼","LeftVectorBar;":"⥒","Leftarrow;":"⇐","Leftrightarrow;":"⇔","LessEqualGreater;":"⋚","LessFullEqual;":"≦","LessGreater;":"≶","LessLess;":"⪡","LessSlantEqual;":"⩽","LessTilde;":"≲","Lfr;":"𝔏","Ll;":"⋘","Lleftarrow;":"⇚","Lmidot;":"Ŀ","LongLeftArrow;":"⟵","LongLeftRightArrow;":"⟷","LongRightArrow;":"⟶","Longleftarrow;":"⟸","Longleftrightarrow;":"⟺","Longrightarrow;":"⟹","Lopf;":"𝕃","LowerLeftArrow;":"↙","LowerRightArrow;":"↘","Lscr;":"ℒ","Lsh;":"↰","Lstrok;":"Ł","Lt;":"≪","Map;":"⤅","Mcy;":"М","MediumSpace;":" ","Mellintrf;":"ℳ","Mfr;":"𝔐","MinusPlus;":"∓","Mopf;":"𝕄","Mscr;":"ℳ","Mu;":"Μ","NJcy;":"Њ","Nacute;":"Ń","Ncaron;":"Ň","Ncedil;":"Ņ","Ncy;":"Н","NegativeMediumSpace;":"","NegativeThickSpace;":"","NegativeThinSpace;":"","NegativeVeryThinSpace;":"","NestedGreaterGreater;":"≫","NestedLessLess;":"≪","NewLine;":"\n","Nfr;":"𝔑","NoBreak;":"","NonBreakingSpace;":" ","Nopf;":"ℕ","Not;":"⫬","NotCongruent;":"≢","NotCupCap;":"≭","NotDoubleVerticalBar;":"∦","NotElement;":"∉","NotEqual;":"≠","NotEqualTilde;":"≂̸","NotExists;":"∄","NotGreater;":"≯","NotGreaterEqual;":"≱","NotGreaterFullEqual;":"≧̸","NotGreaterGreater;":"≫̸","NotGreaterLess;":"≹","NotGreaterSlantEqual;":"⩾̸","NotGreaterTilde;":"≵","NotHumpDownHump;":"≎̸","NotHumpEqual;":"≏̸","NotLeftTriangle;":"⋪","NotLeftTriangleBar;":"⧏̸","NotLeftTriangleEqual;":"⋬","NotLess;":"≮","NotLessEqual;":"≰","NotLessGreater;":"≸","NotLessLess;":"≪̸","NotLessSlantEqual;":"⩽̸","NotLessTilde;":"≴","NotNestedGreaterGreater;":"⪢̸","NotNestedLessLess;":"⪡̸","NotPrecedes;":"⊀","NotPrecedesEqual;":"⪯̸","NotPrecedesSlantEqual;":"⋠","NotReverseElement;":"∌","NotRightTriangle;":"⋫","NotRightTriangleBar;":"⧐̸","NotRightTriangleEqual;":"⋭","NotSquareSubset;":"⊏̸","NotSquareSubsetEqual;":"⋢","NotSquareSuperset;":"⊐̸","NotSquareSupersetEqual;":"⋣","NotSubset;":"⊂⃒","NotSubsetEqual;":"⊈","NotSucceeds;":"⊁","NotSucceedsEqual;":"⪰̸","NotSucceedsSlantEqual;":"⋡","NotSucceedsTilde;":"≿̸","NotSuperset;":"⊃⃒","NotSupersetEqual;":"⊉","NotTilde;":"≁","NotTildeEqual;":"≄","NotTildeFullEqual;":"≇","NotTildeTilde;":"≉","NotVerticalBar;":"∤","Nscr;":"𝒩","Ntilde":"Ñ","Ntilde;":"Ñ","Nu;":"Ν","OElig;":"Œ","Oacute":"Ó","Oacute;":"Ó","Ocirc":"Ô","Ocirc;":"Ô","Ocy;":"О","Odblac;":"Ő","Ofr;":"𝔒","Ograve":"Ò","Ograve;":"Ò","Omacr;":"Ō","Omega;":"Ω","Omicron;":"Ο","Oopf;":"𝕆","OpenCurlyDoubleQuote;":"“","OpenCurlyQuote;":"‘","Or;":"⩔","Oscr;":"𝒪","Oslash":"Ø","Oslash;":"Ø","Otilde":"Õ","Otilde;":"Õ","Otimes;":"⨷","Ouml":"Ö","Ouml;":"Ö","OverBar;":"‾","OverBrace;":"⏞","OverBracket;":"⎴","OverParenthesis;":"⏜","PartialD;":"∂","Pcy;":"П","Pfr;":"𝔓","Phi;":"Φ","Pi;":"Π","PlusMinus;":"±","Poincareplane;":"ℌ","Popf;":"ℙ","Pr;":"⪻","Precedes;":"≺","PrecedesEqual;":"⪯","PrecedesSlantEqual;":"≼","PrecedesTilde;":"≾","Prime;":"″","Product;":"∏","Proportion;":"∷","Proportional;":"∝","Pscr;":"𝒫","Psi;":"Ψ","QUOT":"\"","QUOT;":"\"","Qfr;":"𝔔","Qopf;":"ℚ","Qscr;":"𝒬","RBarr;":"⤐","REG":"®","REG;":"®","Racute;":"Ŕ","Rang;":"⟫","Rarr;":"↠","Rarrtl;":"⤖","Rcaron;":"Ř","Rcedil;":"Ŗ","Rcy;":"Р","Re;":"ℜ","ReverseElement;":"∋","ReverseEquilibrium;":"⇋","ReverseUpEquilibrium;":"⥯","Rfr;":"ℜ","Rho;":"Ρ","RightAngleBracket;":"⟩","RightArrow;":"→","RightArrowBar;":"⇥","RightArrowLeftArrow;":"⇄","RightCeiling;":"⌉","RightDoubleBracket;":"⟧","RightDownTeeVector;":"⥝","RightDownVector;":"⇂","RightDownVectorBar;":"⥕","RightFloor;":"⌋","RightTee;":"⊢","RightTeeArrow;":"↦","RightTeeVector;":"⥛","RightTriangle;":"⊳","RightTriangleBar;":"⧐","RightTriangleEqual;":"⊵","RightUpDownVector;":"⥏","RightUpTeeVector;":"⥜","RightUpVector;":"↾","RightUpVectorBar;":"⥔","RightVector;":"⇀","RightVectorBar;":"⥓","Rightarrow;":"⇒","Ropf;":"ℝ","RoundImplies;":"⥰","Rrightarrow;":"⇛","Rscr;":"ℛ","Rsh;":"↱","RuleDelayed;":"⧴","SHCHcy;":"Щ","SHcy;":"Ш","SOFTcy;":"Ь","Sacute;":"Ś","Sc;":"⪼","Scaron;":"Š","Scedil;":"Ş","Scirc;":"Ŝ","Scy;":"С","Sfr;":"𝔖","ShortDownArrow;":"↓","ShortLeftArrow;":"←","ShortRightArrow;":"→","ShortUpArrow;":"↑","Sigma;":"Σ","SmallCircle;":"∘","Sopf;":"𝕊","Sqrt;":"√","Square;":"□","SquareIntersection;":"⊓","SquareSubset;":"⊏","SquareSubsetEqual;":"⊑","SquareSuperset;":"⊐","SquareSupersetEqual;":"⊒","SquareUnion;":"⊔","Sscr;":"𝒮","Star;":"⋆","Sub;":"⋐","Subset;":"⋐","SubsetEqual;":"⊆","Succeeds;":"≻","SucceedsEqual;":"⪰","SucceedsSlantEqual;":"≽","SucceedsTilde;":"≿","SuchThat;":"∋","Sum;":"∑","Sup;":"⋑","Superset;":"⊃","SupersetEqual;":"⊇","Supset;":"⋑","THORN":"Þ","THORN;":"Þ","TRADE;":"™","TSHcy;":"Ћ","TScy;":"Ц","Tab;":"\t","Tau;":"Τ","Tcaron;":"Ť","Tcedil;":"Ţ","Tcy;":"Т","Tfr;":"𝔗","Therefore;":"∴","Theta;":"Θ","ThickSpace;":" ","ThinSpace;":" ","Tilde;":"∼","TildeEqual;":"≃","TildeFullEqual;":"≅","TildeTilde;":"≈","Topf;":"𝕋","TripleDot;":"⃛","Tscr;":"𝒯","Tstrok;":"Ŧ","Uacute":"Ú","Uacute;":"Ú","Uarr;":"↟","Uarrocir;":"⥉","Ubrcy;":"Ў","Ubreve;":"Ŭ","Ucirc":"Û","Ucirc;":"Û","Ucy;":"У","Udblac;":"Ű","Ufr;":"𝔘","Ugrave":"Ù","Ugrave;":"Ù","Umacr;":"Ū","UnderBar;":"_","UnderBrace;":"⏟","UnderBracket;":"⎵","UnderParenthesis;":"⏝","Union;":"⋃","UnionPlus;":"⊎","Uogon;":"Ų","Uopf;":"𝕌","UpArrow;":"↑","UpArrowBar;":"⤒","UpArrowDownArrow;":"⇅","UpDownArrow;":"↕","UpEquilibrium;":"⥮","UpTee;":"⊥","UpTeeArrow;":"↥","Uparrow;":"⇑","Updownarrow;":"⇕","UpperLeftArrow;":"↖","UpperRightArrow;":"↗","Upsi;":"ϒ","Upsilon;":"Υ","Uring;":"Ů","Uscr;":"𝒰","Utilde;":"Ũ","Uuml":"Ü","Uuml;":"Ü","VDash;":"⊫","Vbar;":"⫫","Vcy;":"В","Vdash;":"⊩","Vdashl;":"⫦","Vee;":"⋁","Verbar;":"‖","Vert;":"‖","VerticalBar;":"∣","VerticalLine;":"|","VerticalSeparator;":"❘","VerticalTilde;":"≀","VeryThinSpace;":" ","Vfr;":"𝔙","Vopf;":"𝕍","Vscr;":"𝒱","Vvdash;":"⊪","Wcirc;":"Ŵ","Wedge;":"⋀","Wfr;":"𝔚","Wopf;":"𝕎","Wscr;":"𝒲","Xfr;":"𝔛","Xi;":"Ξ","Xopf;":"𝕏","Xscr;":"𝒳","YAcy;":"Я","YIcy;":"Ї","YUcy;":"Ю","Yacute":"Ý","Yacute;":"Ý","Ycirc;":"Ŷ","Ycy;":"Ы","Yfr;":"𝔜","Yopf;":"𝕐","Yscr;":"𝒴","Yuml;":"Ÿ","ZHcy;":"Ж","Zacute;":"Ź","Zcaron;":"Ž","Zcy;":"З","Zdot;":"Ż","ZeroWidthSpace;":"","Zeta;":"Ζ","Zfr;":"ℨ","Zopf;":"ℤ","Zscr;":"𝒵","aacute":"á","aacute;":"á","abreve;":"ă","ac;":"∾","acE;":"∾̳","acd;":"∿","acirc":"â","acirc;":"â","acute":"´","acute;":"´","acy;":"а","aelig":"æ","aelig;":"æ","af;":"","afr;":"𝔞","agrave":"à","agrave;":"à","alefsym;":"ℵ","aleph;":"ℵ","alpha;":"α","amacr;":"ā","amalg;":"⨿","amp":"&","amp;":"&","and;":"∧","andand;":"⩕","andd;":"⩜","andslope;":"⩘","andv;":"⩚","ang;":"∠","ange;":"⦤","angle;":"∠","angmsd;":"∡","angmsdaa;":"⦨","angmsdab;":"⦩","angmsdac;":"⦪","angmsdad;":"⦫","angmsdae;":"⦬","angmsdaf;":"⦭","angmsdag;":"⦮","angmsdah;":"⦯","angrt;":"∟","angrtvb;":"⊾","angrtvbd;":"⦝","angsph;":"∢","angst;":"Å","angzarr;":"⍼","aogon;":"ą","aopf;":"𝕒","ap;":"≈","apE;":"⩰","apacir;":"⩯","ape;":"≊","apid;":"≋","apos;":"'","approx;":"≈","approxeq;":"≊","aring":"å","aring;":"å","ascr;":"𝒶","ast;":"*","asymp;":"≈","asympeq;":"≍","atilde":"ã","atilde;":"ã","auml":"ä","auml;":"ä","awconint;":"∳","awint;":"⨑","bNot;":"⫭","backcong;":"≌","backepsilon;":"϶","backprime;":"‵","backsim;":"∽","backsimeq;":"⋍","barvee;":"⊽","barwed;":"⌅","barwedge;":"⌅","bbrk;":"⎵","bbrktbrk;":"⎶","bcong;":"≌","bcy;":"б","bdquo;":"„","becaus;":"∵","because;":"∵","bemptyv;":"⦰","bepsi;":"϶","bernou;":"ℬ","beta;":"β","beth;":"ℶ","between;":"≬","bfr;":"𝔟","bigcap;":"⋂","bigcirc;":"◯","bigcup;":"⋃","bigodot;":"⨀","bigoplus;":"⨁","bigotimes;":"⨂","bigsqcup;":"⨆","bigstar;":"★","bigtriangledown;":"▽","bigtriangleup;":"△","biguplus;":"⨄","bigvee;":"⋁","bigwedge;":"⋀","bkarow;":"⤍","blacklozenge;":"⧫","blacksquare;":"▪","blacktriangle;":"▴","blacktriangledown;":"▾","blacktriangleleft;":"◂","blacktriangleright;":"▸","blank;":"␣","blk12;":"▒","blk14;":"░","blk34;":"▓","block;":"█","bne;":"=⃥","bnequiv;":"≡⃥","bnot;":"⌐","bopf;":"𝕓","bot;":"⊥","bottom;":"⊥","bowtie;":"⋈","boxDL;":"╗","boxDR;":"╔","boxDl;":"╖","boxDr;":"╓","boxH;":"═","boxHD;":"╦","boxHU;":"╩","boxHd;":"╤","boxHu;":"╧","boxUL;":"╝","boxUR;":"╚","boxUl;":"╜","boxUr;":"╙","boxV;":"║","boxVH;":"╬","boxVL;":"╣","boxVR;":"╠","boxVh;":"╫","boxVl;":"╢","boxVr;":"╟","boxbox;":"⧉","boxdL;":"╕","boxdR;":"╒","boxdl;":"┐","boxdr;":"┌","boxh;":"─","boxhD;":"╥","boxhU;":"╨","boxhd;":"┬","boxhu;":"┴","boxminus;":"⊟","boxplus;":"⊞","boxtimes;":"⊠","boxuL;":"╛","boxuR;":"╘","boxul;":"┘","boxur;":"└","boxv;":"│","boxvH;":"╪","boxvL;":"╡","boxvR;":"╞","boxvh;":"┼","boxvl;":"┤","boxvr;":"├","bprime;":"‵","breve;":"˘","brvbar":"¦","brvbar;":"¦","bscr;":"𝒷","bsemi;":"⁏","bsim;":"∽","bsime;":"⋍","bsol;":"\\","bsolb;":"⧅","bsolhsub;":"⟈","bull;":"•","bullet;":"•","bump;":"≎","bumpE;":"⪮","bumpe;":"≏","bumpeq;":"≏","cacute;":"ć","cap;":"∩","capand;":"⩄","capbrcup;":"⩉","capcap;":"⩋","capcup;":"⩇","capdot;":"⩀","caps;":"∩︀","caret;":"⁁","caron;":"ˇ","ccaps;":"⩍","ccaron;":"č","ccedil":"ç","ccedil;":"ç","ccirc;":"ĉ","ccups;":"⩌","ccupssm;":"⩐","cdot;":"ċ","cedil":"¸","cedil;":"¸","cemptyv;":"⦲","cent":"¢","cent;":"¢","centerdot;":"·","cfr;":"𝔠","chcy;":"ч","check;":"✓","checkmark;":"✓","chi;":"χ","cir;":"○","cirE;":"⧃","circ;":"ˆ","circeq;":"≗","circlearrowleft;":"↺","circlearrowright;":"↻","circledR;":"®","circledS;":"Ⓢ","circledast;":"⊛","circledcirc;":"⊚","circleddash;":"⊝","cire;":"≗","cirfnint;":"⨐","cirmid;":"⫯","cirscir;":"⧂","clubs;":"♣","clubsuit;":"♣","colon;":":","colone;":"≔","coloneq;":"≔","comma;":",","commat;":"@","comp;":"∁","compfn;":"∘","complement;":"∁","complexes;":"ℂ","cong;":"≅","congdot;":"⩭","conint;":"∮","copf;":"𝕔","coprod;":"∐","copy":"©","copy;":"©","copysr;":"℗","crarr;":"↵","cross;":"✗","cscr;":"𝒸","csub;":"⫏","csube;":"⫑","csup;":"⫐","csupe;":"⫒","ctdot;":"⋯","cudarrl;":"⤸","cudarrr;":"⤵","cuepr;":"⋞","cuesc;":"⋟","cularr;":"↶","cularrp;":"⤽","cup;":"∪","cupbrcap;":"⩈","cupcap;":"⩆","cupcup;":"⩊","cupdot;":"⊍","cupor;":"⩅","cups;":"∪︀","curarr;":"↷","curarrm;":"⤼","curlyeqprec;":"⋞","curlyeqsucc;":"⋟","curlyvee;":"⋎","curlywedge;":"⋏","curren":"¤","curren;":"¤","curvearrowleft;":"↶","curvearrowright;":"↷","cuvee;":"⋎","cuwed;":"⋏","cwconint;":"∲","cwint;":"∱","cylcty;":"⌭","dArr;":"⇓","dHar;":"⥥","dagger;":"†","daleth;":"ℸ","darr;":"↓","dash;":"‐","dashv;":"⊣","dbkarow;":"⤏","dblac;":"˝","dcaron;":"ď","dcy;":"д","dd;":"ⅆ","ddagger;":"‡","ddarr;":"⇊","ddotseq;":"⩷","deg":"°","deg;":"°","delta;":"δ","demptyv;":"⦱","dfisht;":"⥿","dfr;":"𝔡","dharl;":"⇃","dharr;":"⇂","diam;":"⋄","diamond;":"⋄","diamondsuit;":"♦","diams;":"♦","die;":"¨","digamma;":"ϝ","disin;":"⋲","div;":"÷","divide":"÷","divide;":"÷","divideontimes;":"⋇","divonx;":"⋇","djcy;":"ђ","dlcorn;":"⌞","dlcrop;":"⌍","dollar;":"$","dopf;":"𝕕","dot;":"˙","doteq;":"≐","doteqdot;":"≑","dotminus;":"∸","dotplus;":"∔","dotsquare;":"⊡","doublebarwedge;":"⌆","downarrow;":"↓","downdownarrows;":"⇊","downharpoonleft;":"⇃","downharpoonright;":"⇂","drbkarow;":"⤐","drcorn;":"⌟","drcrop;":"⌌","dscr;":"𝒹","dscy;":"ѕ","dsol;":"⧶","dstrok;":"đ","dtdot;":"⋱","dtri;":"▿","dtrif;":"▾","duarr;":"⇵","duhar;":"⥯","dwangle;":"⦦","dzcy;":"џ","dzigrarr;":"⟿","eDDot;":"⩷","eDot;":"≑","eacute":"é","eacute;":"é","easter;":"⩮","ecaron;":"ě","ecir;":"≖","ecirc":"ê","ecirc;":"ê","ecolon;":"≕","ecy;":"э","edot;":"ė","ee;":"ⅇ","efDot;":"≒","efr;":"𝔢","eg;":"⪚","egrave":"è","egrave;":"è","egs;":"⪖","egsdot;":"⪘","el;":"⪙","elinters;":"⏧","ell;":"ℓ","els;":"⪕","elsdot;":"⪗","emacr;":"ē","empty;":"∅","emptyset;":"∅","emptyv;":"∅","emsp13;":" ","emsp14;":" ","emsp;":" ","eng;":"ŋ","ensp;":" ","eogon;":"ę","eopf;":"𝕖","epar;":"⋕","eparsl;":"⧣","eplus;":"⩱","epsi;":"ε","epsilon;":"ε","epsiv;":"ϵ","eqcirc;":"≖","eqcolon;":"≕","eqsim;":"≂","eqslantgtr;":"⪖","eqslantless;":"⪕","equals;":"=","equest;":"≟","equiv;":"≡","equivDD;":"⩸","eqvparsl;":"⧥","erDot;":"≓","erarr;":"⥱","escr;":"ℯ","esdot;":"≐","esim;":"≂","eta;":"η","eth":"ð","eth;":"ð","euml":"ë","euml;":"ë","euro;":"€","excl;":"!","exist;":"∃","expectation;":"ℰ","exponentiale;":"ⅇ","fallingdotseq;":"≒","fcy;":"ф","female;":"♀","ffilig;":"ffi","fflig;":"ff","ffllig;":"ffl","ffr;":"𝔣","filig;":"fi","fjlig;":"fj","flat;":"♭","fllig;":"fl","fltns;":"▱","fnof;":"ƒ","fopf;":"𝕗","forall;":"∀","fork;":"⋔","forkv;":"⫙","fpartint;":"⨍","frac12":"½","frac12;":"½","frac13;":"⅓","frac14":"¼","frac14;":"¼","frac15;":"⅕","frac16;":"⅙","frac18;":"⅛","frac23;":"⅔","frac25;":"⅖","frac34":"¾","frac34;":"¾","frac35;":"⅗","frac38;":"⅜","frac45;":"⅘","frac56;":"⅚","frac58;":"⅝","frac78;":"⅞","frasl;":"⁄","frown;":"⌢","fscr;":"𝒻","gE;":"≧","gEl;":"⪌","gacute;":"ǵ","gamma;":"γ","gammad;":"ϝ","gap;":"⪆","gbreve;":"ğ","gcirc;":"ĝ","gcy;":"г","gdot;":"ġ","ge;":"≥","gel;":"⋛","geq;":"≥","geqq;":"≧","geqslant;":"⩾","ges;":"⩾","gescc;":"⪩","gesdot;":"⪀","gesdoto;":"⪂","gesdotol;":"⪄","gesl;":"⋛︀","gesles;":"⪔","gfr;":"𝔤","gg;":"≫","ggg;":"⋙","gimel;":"ℷ","gjcy;":"ѓ","gl;":"≷","glE;":"⪒","gla;":"⪥","glj;":"⪤","gnE;":"≩","gnap;":"⪊","gnapprox;":"⪊","gne;":"⪈","gneq;":"⪈","gneqq;":"≩","gnsim;":"⋧","gopf;":"𝕘","grave;":"`","gscr;":"ℊ","gsim;":"≳","gsime;":"⪎","gsiml;":"⪐","gt":">","gt;":">","gtcc;":"⪧","gtcir;":"⩺","gtdot;":"⋗","gtlPar;":"⦕","gtquest;":"⩼","gtrapprox;":"⪆","gtrarr;":"⥸","gtrdot;":"⋗","gtreqless;":"⋛","gtreqqless;":"⪌","gtrless;":"≷","gtrsim;":"≳","gvertneqq;":"≩︀","gvnE;":"≩︀","hArr;":"⇔","hairsp;":" ","half;":"½","hamilt;":"ℋ","hardcy;":"ъ","harr;":"↔","harrcir;":"⥈","harrw;":"↭","hbar;":"ℏ","hcirc;":"ĥ","hearts;":"♥","heartsuit;":"♥","hellip;":"…","hercon;":"⊹","hfr;":"𝔥","hksearow;":"⤥","hkswarow;":"⤦","hoarr;":"⇿","homtht;":"∻","hookleftarrow;":"↩","hookrightarrow;":"↪","hopf;":"𝕙","horbar;":"―","hscr;":"𝒽","hslash;":"ℏ","hstrok;":"ħ","hybull;":"⁃","hyphen;":"‐","iacute":"í","iacute;":"í","ic;":"","icirc":"î","icirc;":"î","icy;":"и","iecy;":"е","iexcl":"¡","iexcl;":"¡","iff;":"⇔","ifr;":"𝔦","igrave":"ì","igrave;":"ì","ii;":"ⅈ","iiiint;":"⨌","iiint;":"∭","iinfin;":"⧜","iiota;":"℩","ijlig;":"ij","imacr;":"ī","image;":"ℑ","imagline;":"ℐ","imagpart;":"ℑ","imath;":"ı","imof;":"⊷","imped;":"Ƶ","in;":"∈","incare;":"℅","infin;":"∞","infintie;":"⧝","inodot;":"ı","int;":"∫","intcal;":"⊺","integers;":"ℤ","intercal;":"⊺","intlarhk;":"⨗","intprod;":"⨼","iocy;":"ё","iogon;":"į","iopf;":"𝕚","iota;":"ι","iprod;":"⨼","iquest":"¿","iquest;":"¿","iscr;":"𝒾","isin;":"∈","isinE;":"⋹","isindot;":"⋵","isins;":"⋴","isinsv;":"⋳","isinv;":"∈","it;":"","itilde;":"ĩ","iukcy;":"і","iuml":"ï","iuml;":"ï","jcirc;":"ĵ","jcy;":"й","jfr;":"𝔧","jmath;":"ȷ","jopf;":"𝕛","jscr;":"𝒿","jsercy;":"ј","jukcy;":"є","kappa;":"κ","kappav;":"ϰ","kcedil;":"ķ","kcy;":"к","kfr;":"𝔨","kgreen;":"ĸ","khcy;":"х","kjcy;":"ќ","kopf;":"𝕜","kscr;":"𝓀","lAarr;":"⇚","lArr;":"⇐","lAtail;":"⤛","lBarr;":"⤎","lE;":"≦","lEg;":"⪋","lHar;":"⥢","lacute;":"ĺ","laemptyv;":"⦴","lagran;":"ℒ","lambda;":"λ","lang;":"⟨","langd;":"⦑","langle;":"⟨","lap;":"⪅","laquo":"«","laquo;":"«","larr;":"←","larrb;":"⇤","larrbfs;":"⤟","larrfs;":"⤝","larrhk;":"↩","larrlp;":"↫","larrpl;":"⤹","larrsim;":"⥳","larrtl;":"↢","lat;":"⪫","latail;":"⤙","late;":"⪭","lates;":"⪭︀","lbarr;":"⤌","lbbrk;":"❲","lbrace;":"{","lbrack;":"[","lbrke;":"⦋","lbrksld;":"⦏","lbrkslu;":"⦍","lcaron;":"ľ","lcedil;":"ļ","lceil;":"⌈","lcub;":"{","lcy;":"л","ldca;":"⤶","ldquo;":"“","ldquor;":"„","ldrdhar;":"⥧","ldrushar;":"⥋","ldsh;":"↲","le;":"≤","leftarrow;":"←","leftarrowtail;":"↢","leftharpoondown;":"↽","leftharpoonup;":"↼","leftleftarrows;":"⇇","leftrightarrow;":"↔","leftrightarrows;":"⇆","leftrightharpoons;":"⇋","leftrightsquigarrow;":"↭","leftthreetimes;":"⋋","leg;":"⋚","leq;":"≤","leqq;":"≦","leqslant;":"⩽","les;":"⩽","lescc;":"⪨","lesdot;":"⩿","lesdoto;":"⪁","lesdotor;":"⪃","lesg;":"⋚︀","lesges;":"⪓","lessapprox;":"⪅","lessdot;":"⋖","lesseqgtr;":"⋚","lesseqqgtr;":"⪋","lessgtr;":"≶","lesssim;":"≲","lfisht;":"⥼","lfloor;":"⌊","lfr;":"𝔩","lg;":"≶","lgE;":"⪑","lhard;":"↽","lharu;":"↼","lharul;":"⥪","lhblk;":"▄","ljcy;":"љ","ll;":"≪","llarr;":"⇇","llcorner;":"⌞","llhard;":"⥫","lltri;":"◺","lmidot;":"ŀ","lmoust;":"⎰","lmoustache;":"⎰","lnE;":"≨","lnap;":"⪉","lnapprox;":"⪉","lne;":"⪇","lneq;":"⪇","lneqq;":"≨","lnsim;":"⋦","loang;":"⟬","loarr;":"⇽","lobrk;":"⟦","longleftarrow;":"⟵","longleftrightarrow;":"⟷","longmapsto;":"⟼","longrightarrow;":"⟶","looparrowleft;":"↫","looparrowright;":"↬","lopar;":"⦅","lopf;":"𝕝","loplus;":"⨭","lotimes;":"⨴","lowast;":"∗","lowbar;":"_","loz;":"◊","lozenge;":"◊","lozf;":"⧫","lpar;":"(","lparlt;":"⦓","lrarr;":"⇆","lrcorner;":"⌟","lrhar;":"⇋","lrhard;":"⥭","lrm;":"","lrtri;":"⊿","lsaquo;":"‹","lscr;":"𝓁","lsh;":"↰","lsim;":"≲","lsime;":"⪍","lsimg;":"⪏","lsqb;":"[","lsquo;":"‘","lsquor;":"‚","lstrok;":"ł","lt":"<","lt;":"<","ltcc;":"⪦","ltcir;":"⩹","ltdot;":"⋖","lthree;":"⋋","ltimes;":"⋉","ltlarr;":"⥶","ltquest;":"⩻","ltrPar;":"⦖","ltri;":"◃","ltrie;":"⊴","ltrif;":"◂","lurdshar;":"⥊","luruhar;":"⥦","lvertneqq;":"≨︀","lvnE;":"≨︀","mDDot;":"∺","macr":"¯","macr;":"¯","male;":"♂","malt;":"✠","maltese;":"✠","map;":"↦","mapsto;":"↦","mapstodown;":"↧","mapstoleft;":"↤","mapstoup;":"↥","marker;":"▮","mcomma;":"⨩","mcy;":"м","mdash;":"—","measuredangle;":"∡","mfr;":"𝔪","mho;":"℧","micro":"µ","micro;":"µ","mid;":"∣","midast;":"*","midcir;":"⫰","middot":"·","middot;":"·","minus;":"−","minusb;":"⊟","minusd;":"∸","minusdu;":"⨪","mlcp;":"⫛","mldr;":"…","mnplus;":"∓","models;":"⊧","mopf;":"𝕞","mp;":"∓","mscr;":"𝓂","mstpos;":"∾","mu;":"μ","multimap;":"⊸","mumap;":"⊸","nGg;":"⋙̸","nGt;":"≫⃒","nGtv;":"≫̸","nLeftarrow;":"⇍","nLeftrightarrow;":"⇎","nLl;":"⋘̸","nLt;":"≪⃒","nLtv;":"≪̸","nRightarrow;":"⇏","nVDash;":"⊯","nVdash;":"⊮","nabla;":"∇","nacute;":"ń","nang;":"∠⃒","nap;":"≉","napE;":"⩰̸","napid;":"≋̸","napos;":"ʼn","napprox;":"≉","natur;":"♮","natural;":"♮","naturals;":"ℕ","nbsp":" ","nbsp;":" ","nbump;":"≎̸","nbumpe;":"≏̸","ncap;":"⩃","ncaron;":"ň","ncedil;":"ņ","ncong;":"≇","ncongdot;":"⩭̸","ncup;":"⩂","ncy;":"н","ndash;":"–","ne;":"≠","neArr;":"⇗","nearhk;":"⤤","nearr;":"↗","nearrow;":"↗","nedot;":"≐̸","nequiv;":"≢","nesear;":"⤨","nesim;":"≂̸","nexist;":"∄","nexists;":"∄","nfr;":"𝔫","ngE;":"≧̸","nge;":"≱","ngeq;":"≱","ngeqq;":"≧̸","ngeqslant;":"⩾̸","nges;":"⩾̸","ngsim;":"≵","ngt;":"≯","ngtr;":"≯","nhArr;":"⇎","nharr;":"↮","nhpar;":"⫲","ni;":"∋","nis;":"⋼","nisd;":"⋺","niv;":"∋","njcy;":"њ","nlArr;":"⇍","nlE;":"≦̸","nlarr;":"↚","nldr;":"‥","nle;":"≰","nleftarrow;":"↚","nleftrightarrow;":"↮","nleq;":"≰","nleqq;":"≦̸","nleqslant;":"⩽̸","nles;":"⩽̸","nless;":"≮","nlsim;":"≴","nlt;":"≮","nltri;":"⋪","nltrie;":"⋬","nmid;":"∤","nopf;":"𝕟","not":"¬","not;":"¬","notin;":"∉","notinE;":"⋹̸","notindot;":"⋵̸","notinva;":"∉","notinvb;":"⋷","notinvc;":"⋶","notni;":"∌","notniva;":"∌","notnivb;":"⋾","notnivc;":"⋽","npar;":"∦","nparallel;":"∦","nparsl;":"⫽⃥","npart;":"∂̸","npolint;":"⨔","npr;":"⊀","nprcue;":"⋠","npre;":"⪯̸","nprec;":"⊀","npreceq;":"⪯̸","nrArr;":"⇏","nrarr;":"↛","nrarrc;":"⤳̸","nrarrw;":"↝̸","nrightarrow;":"↛","nrtri;":"⋫","nrtrie;":"⋭","nsc;":"⊁","nsccue;":"⋡","nsce;":"⪰̸","nscr;":"𝓃","nshortmid;":"∤","nshortparallel;":"∦","nsim;":"≁","nsime;":"≄","nsimeq;":"≄","nsmid;":"∤","nspar;":"∦","nsqsube;":"⋢","nsqsupe;":"⋣","nsub;":"⊄","nsubE;":"⫅̸","nsube;":"⊈","nsubset;":"⊂⃒","nsubseteq;":"⊈","nsubseteqq;":"⫅̸","nsucc;":"⊁","nsucceq;":"⪰̸","nsup;":"⊅","nsupE;":"⫆̸","nsupe;":"⊉","nsupset;":"⊃⃒","nsupseteq;":"⊉","nsupseteqq;":"⫆̸","ntgl;":"≹","ntilde":"ñ","ntilde;":"ñ","ntlg;":"≸","ntriangleleft;":"⋪","ntrianglelefteq;":"⋬","ntriangleright;":"⋫","ntrianglerighteq;":"⋭","nu;":"ν","num;":"#","numero;":"№","numsp;":" ","nvDash;":"⊭","nvHarr;":"⤄","nvap;":"≍⃒","nvdash;":"⊬","nvge;":"≥⃒","nvgt;":">⃒","nvinfin;":"⧞","nvlArr;":"⤂","nvle;":"≤⃒","nvlt;":"<⃒","nvltrie;":"⊴⃒","nvrArr;":"⤃","nvrtrie;":"⊵⃒","nvsim;":"∼⃒","nwArr;":"⇖","nwarhk;":"⤣","nwarr;":"↖","nwarrow;":"↖","nwnear;":"⤧","oS;":"Ⓢ","oacute":"ó","oacute;":"ó","oast;":"⊛","ocir;":"⊚","ocirc":"ô","ocirc;":"ô","ocy;":"о","odash;":"⊝","odblac;":"ő","odiv;":"⨸","odot;":"⊙","odsold;":"⦼","oelig;":"œ","ofcir;":"⦿","ofr;":"𝔬","ogon;":"˛","ograve":"ò","ograve;":"ò","ogt;":"⧁","ohbar;":"⦵","ohm;":"Ω","oint;":"∮","olarr;":"↺","olcir;":"⦾","olcross;":"⦻","oline;":"‾","olt;":"⧀","omacr;":"ō","omega;":"ω","omicron;":"ο","omid;":"⦶","ominus;":"⊖","oopf;":"𝕠","opar;":"⦷","operp;":"⦹","oplus;":"⊕","or;":"∨","orarr;":"↻","ord;":"⩝","order;":"ℴ","orderof;":"ℴ","ordf":"ª","ordf;":"ª","ordm":"º","ordm;":"º","origof;":"⊶","oror;":"⩖","orslope;":"⩗","orv;":"⩛","oscr;":"ℴ","oslash":"ø","oslash;":"ø","osol;":"⊘","otilde":"õ","otilde;":"õ","otimes;":"⊗","otimesas;":"⨶","ouml":"ö","ouml;":"ö","ovbar;":"⌽","par;":"∥","para":"¶","para;":"¶","parallel;":"∥","parsim;":"⫳","parsl;":"⫽","part;":"∂","pcy;":"п","percnt;":"%","period;":".","permil;":"‰","perp;":"⊥","pertenk;":"‱","pfr;":"𝔭","phi;":"φ","phiv;":"ϕ","phmmat;":"ℳ","phone;":"☎","pi;":"π","pitchfork;":"⋔","piv;":"ϖ","planck;":"ℏ","planckh;":"ℎ","plankv;":"ℏ","plus;":"+","plusacir;":"⨣","plusb;":"⊞","pluscir;":"⨢","plusdo;":"∔","plusdu;":"⨥","pluse;":"⩲","plusmn":"±","plusmn;":"±","plussim;":"⨦","plustwo;":"⨧","pm;":"±","pointint;":"⨕","popf;":"𝕡","pound":"£","pound;":"£","pr;":"≺","prE;":"⪳","prap;":"⪷","prcue;":"≼","pre;":"⪯","prec;":"≺","precapprox;":"⪷","preccurlyeq;":"≼","preceq;":"⪯","precnapprox;":"⪹","precneqq;":"⪵","precnsim;":"⋨","precsim;":"≾","prime;":"′","primes;":"ℙ","prnE;":"⪵","prnap;":"⪹","prnsim;":"⋨","prod;":"∏","profalar;":"⌮","profline;":"⌒","profsurf;":"⌓","prop;":"∝","propto;":"∝","prsim;":"≾","prurel;":"⊰","pscr;":"𝓅","psi;":"ψ","puncsp;":" ","qfr;":"𝔮","qint;":"⨌","qopf;":"𝕢","qprime;":"⁗","qscr;":"𝓆","quaternions;":"ℍ","quatint;":"⨖","quest;":"?","questeq;":"≟","quot":"\"","quot;":"\"","rAarr;":"⇛","rArr;":"⇒","rAtail;":"⤜","rBarr;":"⤏","rHar;":"⥤","race;":"∽̱","racute;":"ŕ","radic;":"√","raemptyv;":"⦳","rang;":"⟩","rangd;":"⦒","range;":"⦥","rangle;":"⟩","raquo":"»","raquo;":"»","rarr;":"→","rarrap;":"⥵","rarrb;":"⇥","rarrbfs;":"⤠","rarrc;":"⤳","rarrfs;":"⤞","rarrhk;":"↪","rarrlp;":"↬","rarrpl;":"⥅","rarrsim;":"⥴","rarrtl;":"↣","rarrw;":"↝","ratail;":"⤚","ratio;":"∶","rationals;":"ℚ","rbarr;":"⤍","rbbrk;":"❳","rbrace;":"}","rbrack;":"]","rbrke;":"⦌","rbrksld;":"⦎","rbrkslu;":"⦐","rcaron;":"ř","rcedil;":"ŗ","rceil;":"⌉","rcub;":"}","rcy;":"р","rdca;":"⤷","rdldhar;":"⥩","rdquo;":"”","rdquor;":"”","rdsh;":"↳","real;":"ℜ","realine;":"ℛ","realpart;":"ℜ","reals;":"ℝ","rect;":"▭","reg":"®","reg;":"®","rfisht;":"⥽","rfloor;":"⌋","rfr;":"𝔯","rhard;":"⇁","rharu;":"⇀","rharul;":"⥬","rho;":"ρ","rhov;":"ϱ","rightarrow;":"→","rightarrowtail;":"↣","rightharpoondown;":"⇁","rightharpoonup;":"⇀","rightleftarrows;":"⇄","rightleftharpoons;":"⇌","rightrightarrows;":"⇉","rightsquigarrow;":"↝","rightthreetimes;":"⋌","ring;":"˚","risingdotseq;":"≓","rlarr;":"⇄","rlhar;":"⇌","rlm;":"","rmoust;":"⎱","rmoustache;":"⎱","rnmid;":"⫮","roang;":"⟭","roarr;":"⇾","robrk;":"⟧","ropar;":"⦆","ropf;":"𝕣","roplus;":"⨮","rotimes;":"⨵","rpar;":")","rpargt;":"⦔","rppolint;":"⨒","rrarr;":"⇉","rsaquo;":"›","rscr;":"𝓇","rsh;":"↱","rsqb;":"]","rsquo;":"’","rsquor;":"’","rthree;":"⋌","rtimes;":"⋊","rtri;":"▹","rtrie;":"⊵","rtrif;":"▸","rtriltri;":"⧎","ruluhar;":"⥨","rx;":"℞","sacute;":"ś","sbquo;":"‚","sc;":"≻","scE;":"⪴","scap;":"⪸","scaron;":"š","sccue;":"≽","sce;":"⪰","scedil;":"ş","scirc;":"ŝ","scnE;":"⪶","scnap;":"⪺","scnsim;":"⋩","scpolint;":"⨓","scsim;":"≿","scy;":"с","sdot;":"⋅","sdotb;":"⊡","sdote;":"⩦","seArr;":"⇘","searhk;":"⤥","searr;":"↘","searrow;":"↘","sect":"§","sect;":"§","semi;":";","seswar;":"⤩","setminus;":"∖","setmn;":"∖","sext;":"✶","sfr;":"𝔰","sfrown;":"⌢","sharp;":"♯","shchcy;":"щ","shcy;":"ш","shortmid;":"∣","shortparallel;":"∥","shy":"","shy;":"","sigma;":"σ","sigmaf;":"ς","sigmav;":"ς","sim;":"∼","simdot;":"⩪","sime;":"≃","simeq;":"≃","simg;":"⪞","simgE;":"⪠","siml;":"⪝","simlE;":"⪟","simne;":"≆","simplus;":"⨤","simrarr;":"⥲","slarr;":"←","smallsetminus;":"∖","smashp;":"⨳","smeparsl;":"⧤","smid;":"∣","smile;":"⌣","smt;":"⪪","smte;":"⪬","smtes;":"⪬︀","softcy;":"ь","sol;":"/","solb;":"⧄","solbar;":"⌿","sopf;":"𝕤","spades;":"♠","spadesuit;":"♠","spar;":"∥","sqcap;":"⊓","sqcaps;":"⊓︀","sqcup;":"⊔","sqcups;":"⊔︀","sqsub;":"⊏","sqsube;":"⊑","sqsubset;":"⊏","sqsubseteq;":"⊑","sqsup;":"⊐","sqsupe;":"⊒","sqsupset;":"⊐","sqsupseteq;":"⊒","squ;":"□","square;":"□","squarf;":"▪","squf;":"▪","srarr;":"→","sscr;":"𝓈","ssetmn;":"∖","ssmile;":"⌣","sstarf;":"⋆","star;":"☆","starf;":"★","straightepsilon;":"ϵ","straightphi;":"ϕ","strns;":"¯","sub;":"⊂","subE;":"⫅","subdot;":"⪽","sube;":"⊆","subedot;":"⫃","submult;":"⫁","subnE;":"⫋","subne;":"⊊","subplus;":"⪿","subrarr;":"⥹","subset;":"⊂","subseteq;":"⊆","subseteqq;":"⫅","subsetneq;":"⊊","subsetneqq;":"⫋","subsim;":"⫇","subsub;":"⫕","subsup;":"⫓","succ;":"≻","succapprox;":"⪸","succcurlyeq;":"≽","succeq;":"⪰","succnapprox;":"⪺","succneqq;":"⪶","succnsim;":"⋩","succsim;":"≿","sum;":"∑","sung;":"♪","sup1":"¹","sup1;":"¹","sup2":"²","sup2;":"²","sup3":"³","sup3;":"³","sup;":"⊃","supE;":"⫆","supdot;":"⪾","supdsub;":"⫘","supe;":"⊇","supedot;":"⫄","suphsol;":"⟉","suphsub;":"⫗","suplarr;":"⥻","supmult;":"⫂","supnE;":"⫌","supne;":"⊋","supplus;":"⫀","supset;":"⊃","supseteq;":"⊇","supseteqq;":"⫆","supsetneq;":"⊋","supsetneqq;":"⫌","supsim;":"⫈","supsub;":"⫔","supsup;":"⫖","swArr;":"⇙","swarhk;":"⤦","swarr;":"↙","swarrow;":"↙","swnwar;":"⤪","szlig":"ß","szlig;":"ß","target;":"⌖","tau;":"τ","tbrk;":"⎴","tcaron;":"ť","tcedil;":"ţ","tcy;":"т","tdot;":"⃛","telrec;":"⌕","tfr;":"𝔱","there4;":"∴","therefore;":"∴","theta;":"θ","thetasym;":"ϑ","thetav;":"ϑ","thickapprox;":"≈","thicksim;":"∼","thinsp;":" ","thkap;":"≈","thksim;":"∼","thorn":"þ","thorn;":"þ","tilde;":"˜","times":"×","times;":"×","timesb;":"⊠","timesbar;":"⨱","timesd;":"⨰","tint;":"∭","toea;":"⤨","top;":"⊤","topbot;":"⌶","topcir;":"⫱","topf;":"𝕥","topfork;":"⫚","tosa;":"⤩","tprime;":"‴","trade;":"™","triangle;":"▵","triangledown;":"▿","triangleleft;":"◃","trianglelefteq;":"⊴","triangleq;":"≜","triangleright;":"▹","trianglerighteq;":"⊵","tridot;":"◬","trie;":"≜","triminus;":"⨺","triplus;":"⨹","trisb;":"⧍","tritime;":"⨻","trpezium;":"⏢","tscr;":"𝓉","tscy;":"ц","tshcy;":"ћ","tstrok;":"ŧ","twixt;":"≬","twoheadleftarrow;":"↞","twoheadrightarrow;":"↠","uArr;":"⇑","uHar;":"⥣","uacute":"ú","uacute;":"ú","uarr;":"↑","ubrcy;":"ў","ubreve;":"ŭ","ucirc":"û","ucirc;":"û","ucy;":"у","udarr;":"⇅","udblac;":"ű","udhar;":"⥮","ufisht;":"⥾","ufr;":"𝔲","ugrave":"ù","ugrave;":"ù","uharl;":"↿","uharr;":"↾","uhblk;":"▀","ulcorn;":"⌜","ulcorner;":"⌜","ulcrop;":"⌏","ultri;":"◸","umacr;":"ū","uml":"¨","uml;":"¨","uogon;":"ų","uopf;":"𝕦","uparrow;":"↑","updownarrow;":"↕","upharpoonleft;":"↿","upharpoonright;":"↾","uplus;":"⊎","upsi;":"υ","upsih;":"ϒ","upsilon;":"υ","upuparrows;":"⇈","urcorn;":"⌝","urcorner;":"⌝","urcrop;":"⌎","uring;":"ů","urtri;":"◹","uscr;":"𝓊","utdot;":"⋰","utilde;":"ũ","utri;":"▵","utrif;":"▴","uuarr;":"⇈","uuml":"ü","uuml;":"ü","uwangle;":"⦧","vArr;":"⇕","vBar;":"⫨","vBarv;":"⫩","vDash;":"⊨","vangrt;":"⦜","varepsilon;":"ϵ","varkappa;":"ϰ","varnothing;":"∅","varphi;":"ϕ","varpi;":"ϖ","varpropto;":"∝","varr;":"↕","varrho;":"ϱ","varsigma;":"ς","varsubsetneq;":"⊊︀","varsubsetneqq;":"⫋︀","varsupsetneq;":"⊋︀","varsupsetneqq;":"⫌︀","vartheta;":"ϑ","vartriangleleft;":"⊲","vartriangleright;":"⊳","vcy;":"в","vdash;":"⊢","vee;":"∨","veebar;":"⊻","veeeq;":"≚","vellip;":"⋮","verbar;":"|","vert;":"|","vfr;":"𝔳","vltri;":"⊲","vnsub;":"⊂⃒","vnsup;":"⊃⃒","vopf;":"𝕧","vprop;":"∝","vrtri;":"⊳","vscr;":"𝓋","vsubnE;":"⫋︀","vsubne;":"⊊︀","vsupnE;":"⫌︀","vsupne;":"⊋︀","vzigzag;":"⦚","wcirc;":"ŵ","wedbar;":"⩟","wedge;":"∧","wedgeq;":"≙","weierp;":"℘","wfr;":"𝔴","wopf;":"𝕨","wp;":"℘","wr;":"≀","wreath;":"≀","wscr;":"𝓌","xcap;":"⋂","xcirc;":"◯","xcup;":"⋃","xdtri;":"▽","xfr;":"𝔵","xhArr;":"⟺","xharr;":"⟷","xi;":"ξ","xlArr;":"⟸","xlarr;":"⟵","xmap;":"⟼","xnis;":"⋻","xodot;":"⨀","xopf;":"𝕩","xoplus;":"⨁","xotime;":"⨂","xrArr;":"⟹","xrarr;":"⟶","xscr;":"𝓍","xsqcup;":"⨆","xuplus;":"⨄","xutri;":"△","xvee;":"⋁","xwedge;":"⋀","yacute":"ý","yacute;":"ý","yacy;":"я","ycirc;":"ŷ","ycy;":"ы","yen":"¥","yen;":"¥","yfr;":"𝔶","yicy;":"ї","yopf;":"𝕪","yscr;":"𝓎","yucy;":"ю","yuml":"ÿ","yuml;":"ÿ","zacute;":"ź","zcaron;":"ž","zcy;":"з","zdot;":"ż","zeetrf;":"ℨ","zeta;":"ζ","zfr;":"𝔷","zhcy;":"ж","zigrarr;":"⇝","zopf;":"𝕫","zscr;":"𝓏","zwj;":"","zwnj;":""})));
- // #endregion
- const STATE_DATA = 0;
- const STATE_TAG_OPEN = 1;
- const STATE_END_TAG_OPEN = 2;
- const STATE_TAG_NAME = 3;
- const STATE_BEFORE_ATTRIBUTE_NAME = 4;
- const STATE_ATTRIBUTE_NAME = 5;
- const STATE_AFTER_ATTRIBUTE_NAME = 6;
- const STATE_BEFORE_ATTRIBUTE_VALUE = 7;
- const STATE_ATTRIBUTE_VALUE_DOUBLE_QUOTED = 8;
- const STATE_ATTRIBUTE_VALUE_SINGLE_QUOTED = 9;
- const STATE_ATTRIBUTE_VALUE_UNQUOTED = 10;
- const STATE_AFTER_ATTRIBUTE_VALUE_QUOTED = 11;
- const STATE_SELF_CLOSING_START_TAG = 12;
- const STATE_MARKUP_DECLARATION_OPEN = 13;
- const STATE_COMMENT_START = 14;
- const STATE_COMMENT_START_DASH = 15;
- const STATE_COMMENT = 16;
- const STATE_COMMENT_END_DASH = 17;
- const STATE_COMMENT_END = 18;
- const STATE_COMMENT_END_BANG = 19;
- const STATE_BOGUS_COMMENT = 20;
- const STATE_COMMENT_LESS_THAN_SIGN = 21;
- const STATE_COMMENT_LESS_THAN_SIGN_BANG = 22;
- const STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH = 23;
- const STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH_DASH = 24;
- const STATE_DOCTYPE = 25;
- const STATE_BEFORE_DOCTYPE_NAME = 26;
- const STATE_DOCTYPE_NAME = 27;
- const STATE_AFTER_DOCTYPE_NAME = 28;
- const STATE_AFTER_DOCTYPE_PUBLIC_KEYWORD = 29;
- const STATE_BEFORE_DOCTYPE_PUBLIC_IDENTIFIER = 30;
- const STATE_DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED = 31;
- const STATE_DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED = 32;
- const STATE_AFTER_DOCTYPE_PUBLIC_IDENTIFIER = 33;
- const STATE_BETWEEN_DOCTYPE_PUBLIC_AND_SYSTEM_IDENTIFIERS = 34;
- const STATE_AFTER_DOCTYPE_SYSTEM_KEYWORD = 35;
- const STATE_BEFORE_DOCTYPE_SYSTEM_IDENTIFIER = 36;
- const STATE_DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED = 37;
- const STATE_DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED = 38;
- const STATE_AFTER_DOCTYPE_SYSTEM_IDENTIFIER = 39;
- const STATE_BOGUS_DOCTYPE = 40;
- const STATE_CDATA_SECTION = 41;
- const STATE_CDATA_SECTION_BRACKET = 42;
- const STATE_CDATA_SECTION_END = 43;
- const STATE_RCDATA = 44;
- const STATE_RCDATA_LESS_THAN_SIGN = 45;
- const STATE_RCDATA_END_TAG_OPEN = 46;
- const STATE_RCDATA_END_TAG_NAME = 47;
- const STATE_RAWTEXT = 48;
- const STATE_RAWTEXT_LESS_THAN_SIGN = 49;
- const STATE_RAWTEXT_END_TAG_OPEN = 50;
- const STATE_RAWTEXT_END_TAG_NAME = 51;
- const STATE_SCRIPT_DATA = 52;
- const STATE_SCRIPT_DATA_LESS_THAN_SIGN = 53;
- const STATE_SCRIPT_DATA_END_TAG_OPEN = 54;
- const STATE_SCRIPT_DATA_END_TAG_NAME = 55;
- const STATE_SCRIPT_DATA_ESCAPE_START = 56;
- const STATE_SCRIPT_DATA_ESCAPE_START_DASH = 57;
- const STATE_SCRIPT_DATA_ESCAPED = 58;
- const STATE_SCRIPT_DATA_ESCAPED_DASH = 59;
- const STATE_SCRIPT_DATA_ESCAPED_DASH_DASH = 60;
- const STATE_SCRIPT_DATA_ESCAPED_LESS_THAN_SIGN = 61;
- const STATE_SCRIPT_DATA_ESCAPED_END_TAG_OPEN = 62;
- const STATE_SCRIPT_DATA_ESCAPED_END_TAG_NAME = 63;
- const STATE_SCRIPT_DATA_DOUBLE_ESCAPE_START = 64;
- const STATE_SCRIPT_DATA_DOUBLE_ESCAPED = 65;
- const STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH = 66;
- const STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH_DASH = 67;
- const STATE_SCRIPT_DATA_DOUBLE_ESCAPED_LESS_THAN_SIGN = 68;
- const STATE_SCRIPT_DATA_DOUBLE_ESCAPE_END = 69;
- const STATE_PLAINTEXT = 70;
- // https://html.spec.whatwg.org/multipage/parsing.html#character-reference-state
- const STATE_CHARACTER_REFERENCE = 71;
- // https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state
- const STATE_NAMED_CHARACTER_REFERENCE = 72;
- // https://html.spec.whatwg.org/multipage/parsing.html#ambiguous-ampersand-state
- const STATE_AMBIGUOUS_AMPERSAND = 73;
- // https://html.spec.whatwg.org/multipage/parsing.html#numeric-character-reference-state
- const STATE_NUMERIC_CHARACTER_REFERENCE = 74;
- // https://html.spec.whatwg.org/multipage/parsing.html#hexadecimal-character-reference-start-state
- const STATE_HEXADECIMAL_CHARACTER_REFERENCE_START = 75;
- // https://html.spec.whatwg.org/multipage/parsing.html#decimal-character-reference-start-state
- const STATE_DECIMAL_CHARACTER_REFERENCE_START = 76;
- // https://html.spec.whatwg.org/multipage/parsing.html#hexadecimal-character-reference-state
- const STATE_HEXADECIMAL_CHARACTER_REFERENCE = 77;
- // https://html.spec.whatwg.org/multipage/parsing.html#decimal-character-reference-state
- const STATE_DECIMAL_CHARACTER_REFERENCE = 78;
- // https://html.spec.whatwg.org/multipage/parsing.html#numeric-character-reference-end-state
- const STATE_NUMERIC_CHARACTER_REFERENCE_END = 79;
- const CC_TAB = 0x09;
- const CC_LF = 0x0a;
- const CC_FF = 0x0c;
- const CC_CR = 0x0d;
- const CC_SPACE = 0x20;
- const CC_NULL = 0x00;
- const CC_EXCLAMATION_MARK = 0x21;
- const CC_QUOTATION_MARK = 0x22;
- const CC_NUMBER_SIGN = 0x23;
- const CC_AMPERSAND = 0x26;
- const CC_APOSTROPHE = 0x27;
- const CC_HYPHEN_MINUS = 0x2d;
- const CC_SOLIDUS = 0x2f;
- const CC_SEMICOLON = 0x3b;
- const CC_LESS_THAN = 0x3c;
- const CC_EQUALS = 0x3d;
- const CC_GREATER_THAN = 0x3e;
- const CC_QUESTION_MARK = 0x3f;
- const CC_LEFT_SQUARE_BRACKET = 0x5b;
- const CC_RIGHT_SQUARE_BRACKET = 0x5d;
- const CC_LOW_LINE = 0x5f;
- const CC_GRAVE_ACCENT = 0x60;
- const CC_NO_BREAK_SPACE = 0xa0;
- const QUOTE_DOUBLE = 1;
- const QUOTE_SINGLE = 2;
- const QUOTE_NONE = 0;
- // Longest WHATWG named entity name *including* the trailing `;` is 32 chars
- // (`CounterClockwiseContourIntegral;`); without the trailing `;` it's 31.
- // Used to cap both the tokenizer's named-character-reference run length and
- // the decoder's longest-prefix backtrack so pathological inputs (e.g. `&`
- // followed by thousands of alphanumerics) stay linear-time.
- const MAX_ENTITY_NAME_LEN = 32;
- // ASCII character-class bit flags packed into one lookup table. The tokenizer
- // runs these predicates per code point (tag names, attribute names, character
- // references, whitespace skipping), so a single table load + mask replaces the
- // per-call comparison chains. Code points >= 0x80 are never in any of these
- // classes, so callers short-circuit on `cc < 0x80` before indexing.
- const CCLS_SPACE = 1;
- const CCLS_DIGIT = 2;
- const CCLS_UPPER = 4;
- const CCLS_LOWER = 8;
- const CCLS_HEX = 16;
- // Terminator sets of the tag-name / attribute-name states' fast-forward scans,
- // mirroring those states' arcs; one mask test replaces their compare chains.
- const CCLS_TAG_NAME_TERM = 32;
- const CCLS_ATTR_NAME_TERM = 64;
- const CCLS_ALPHA = CCLS_UPPER | CCLS_LOWER;
- const CCLS_ALNUM = CCLS_ALPHA | CCLS_DIGIT;
- const CHAR_CLASS = new Uint8Array(128);
- for (let i = 0; i < 128; i++) {
- let f = 0;
- if (
- i === CC_TAB ||
- i === CC_LF ||
- i === CC_FF ||
- i === CC_CR ||
- i === CC_SPACE
- ) {
- f |= CCLS_SPACE;
- }
- if (i >= 0x30 && i <= 0x39) f |= CCLS_DIGIT | CCLS_HEX;
- if (i >= 0x41 && i <= 0x5a) f |= CCLS_UPPER;
- if (i >= 0x61 && i <= 0x7a) f |= CCLS_LOWER;
- if ((i >= 0x41 && i <= 0x46) || (i >= 0x61 && i <= 0x66)) f |= CCLS_HEX;
- if (
- (f & CCLS_SPACE) !== 0 ||
- i === CC_SOLIDUS ||
- i === CC_GREATER_THAN ||
- i === CC_NULL
- ) {
- f |= CCLS_TAG_NAME_TERM | CCLS_ATTR_NAME_TERM;
- }
- if (
- i === CC_EQUALS ||
- i === CC_QUOTATION_MARK ||
- i === CC_APOSTROPHE ||
- i === CC_LESS_THAN
- ) {
- f |= CCLS_ATTR_NAME_TERM;
- }
- CHAR_CLASS[i] = f;
- }
- /**
- * @param {number} cc character code
- * @returns {boolean} is ascii alpha
- */
- const isAsciiAlpha = (cc) => cc < 0x80 && (CHAR_CLASS[cc] & CCLS_ALPHA) !== 0;
- /**
- * @param {number} cc character code
- * @returns {boolean} is ascii alphanumeric
- */
- const isAsciiAlphanumeric = (cc) =>
- cc < 0x80 && (CHAR_CLASS[cc] & CCLS_ALNUM) !== 0;
- /**
- * @param {number} cc character code
- * @returns {boolean} is ascii digit
- */
- const isAsciiDigit = (cc) => cc < 0x80 && (CHAR_CLASS[cc] & CCLS_DIGIT) !== 0;
- /**
- * @param {number} cc character code
- * @returns {boolean} is ascii hex digit
- */
- const isAsciiHexDigit = (cc) => cc < 0x80 && (CHAR_CLASS[cc] & CCLS_HEX) !== 0;
- /**
- * @param {number} cc character code
- * @returns {boolean} is ascii upper alpha
- */
- const isAsciiUpperAlpha = (cc) =>
- cc < 0x80 && (CHAR_CLASS[cc] & CCLS_UPPER) !== 0;
- /**
- * @param {number} cc character code
- * @returns {boolean} is ascii lower alpha
- */
- const isAsciiLowerAlpha = (cc) =>
- cc < 0x80 && (CHAR_CLASS[cc] & CCLS_LOWER) !== 0;
- /**
- * Tokenizer whitespace. U+000D CARRIAGE RETURN is included because the spec's
- * input-stream preprocessing converts CR (and CRLF) to LF before tokenizing;
- * this scanner keeps the original offsets, so it treats a raw CR as whitespace
- * to match the post-preprocessing behaviour.
- * @param {number} cc character code
- * @returns {boolean} is space
- */
- const isSpace = (cc) => cc < 0x80 && (CHAR_CLASS[cc] & CCLS_SPACE) !== 0;
- /**
- * @param {number} code numeric character reference code point
- * @returns {boolean} is a Unicode noncharacter
- */
- const isNoncharacter = (code) =>
- (code >= 0xfdd0 && code <= 0xfdef) || (code & 0xfffe) === 0xfffe;
- /**
- * @param {number} code numeric character reference code point
- * @returns {boolean} is a C0/C1 control that is not ASCII whitespace
- */
- const isControlReference = (code) =>
- code === 0x0d ||
- ((code <= 0x1f || (code >= 0x7f && code <= 0x9f)) &&
- code !== CC_TAB &&
- code !== CC_LF &&
- code !== CC_FF &&
- code !== CC_SPACE);
- /**
- * Severity of a tokenizer-detected parse error. `"warning"` is recoverable
- * (the tokenizer continued and the emitted token is still well-formed, e.g.
- * missing-attribute-value); `"error"` means the emitted token's offset
- * range is incomplete or does not match what the spec would produce, e.g.
- * eof-in-tag.
- *
- * Token offsets are JS string indices (UTF-16 code-unit offsets into
- * `input`), not byte offsets — relevant for inputs containing non-BMP
- * code points where one code point spans two indices.
- * @typedef {"warning" | "error"} ParseErrorSeverity
- */
- /**
- * @typedef {object} HtmlTokenCallbacks
- * @property {(input: string, start: number, end: number, nameStart: number, nameEnd: number, selfClosing: boolean) => number=} openTag
- * @property {(input: string, start: number, end: number, nameStart: number, nameEnd: number) => number=} closeTag
- * @property {(input: string, start: number, end: number) => number=} text
- * @property {(input: string, nameStart: number, nameEnd: number, valueStart: number, valueEnd: number, quoteType: number) => number=} attribute
- * @property {(input: string, start: number, end: number, dataStart: number, dataEnd: number) => number=} comment
- * @property {(input: string, start: number, end: number, nameStart: number, nameEnd: number, publicStart: number, publicEnd: number, systemStart: number, systemEnd: number, forceQuirks: boolean) => number=} doctype
- * @property {(input: string, code: string, start: number, end: number, severity: ParseErrorSeverity) => void=} parseError
- * @property {(() => boolean)=} isForeign returns true when the adjusted current node is in a foreign (SVG/MathML) namespace, which is where a `<![CDATA[` opens a CDATA section
- * @property {((name: string) => boolean)=} vetoesContentMode returns true when the tree builder will not run the algorithm that switches the tokenizer for this start tag, so it stays in the data state
- * @property {string=} fragmentContext context element tag name for fragment parsing; seeds the initial content mode
- */
- /**
- * @param {string} name tag name (lowercase)
- * @returns {number} content mode state for this tag, or STATE_DATA
- */
- const getContentModeForTag = (name) => {
- switch (name) {
- case "textarea":
- case "title":
- return STATE_RCDATA;
- case "style":
- case "xmp":
- case "iframe":
- case "noembed":
- case "noframes":
- return STATE_RAWTEXT;
- case "script":
- return STATE_SCRIPT_DATA;
- case "plaintext":
- return STATE_PLAINTEXT;
- default:
- return STATE_DATA;
- }
- };
- /**
- * Case-insensitive comparison of `input[start..end)` to a lowercase ASCII
- * literal, without allocating the slice.
- * @param {string} input input
- * @param {number} start range start
- * @param {number} end range end
- * @param {string} lit lowercase ASCII literal
- * @returns {boolean} true if the range equals `lit` ignoring ASCII case
- */
- const rangeEqualsLowerCase = (input, start, end, lit) => {
- if (end - start !== lit.length) return false;
- for (let i = 0; i < lit.length; i++) {
- let c = input.charCodeAt(start + i);
- if (c >= 0x41 && c <= 0x5a) c += 0x20;
- if (c !== lit.charCodeAt(i)) return false;
- }
- return true;
- };
- /**
- * Content mode for the just-opened tag whose name spans `input[start..end)`,
- * matched on the raw range so ordinary tags need neither a slice nor a
- * `toLowerCase`. Mirrors `getContentModeForTag`.
- * @param {string} input input
- * @param {number} start tag-name start
- * @param {number} end tag-name end
- * @returns {number} content mode state, or STATE_DATA
- */
- const getContentModeForRange = (input, start, end) => {
- switch (end - start) {
- case 3:
- if (rangeEqualsLowerCase(input, start, end, "xmp")) return STATE_RAWTEXT;
- return STATE_DATA;
- case 5:
- if (rangeEqualsLowerCase(input, start, end, "title")) return STATE_RCDATA;
- if (rangeEqualsLowerCase(input, start, end, "style")) {
- return STATE_RAWTEXT;
- }
- return STATE_DATA;
- case 6:
- if (rangeEqualsLowerCase(input, start, end, "script")) {
- return STATE_SCRIPT_DATA;
- }
- if (rangeEqualsLowerCase(input, start, end, "iframe")) {
- return STATE_RAWTEXT;
- }
- return STATE_DATA;
- case 7:
- if (rangeEqualsLowerCase(input, start, end, "noembed")) {
- return STATE_RAWTEXT;
- }
- return STATE_DATA;
- case 8:
- if (rangeEqualsLowerCase(input, start, end, "textarea")) {
- return STATE_RCDATA;
- }
- if (rangeEqualsLowerCase(input, start, end, "noframes")) {
- return STATE_RAWTEXT;
- }
- return STATE_DATA;
- case 9:
- if (rangeEqualsLowerCase(input, start, end, "plaintext")) {
- return STATE_PLAINTEXT;
- }
- return STATE_DATA;
- default:
- return STATE_DATA;
- }
- };
- /**
- * @param {string} input input string
- * @param {number} pos current position
- * @param {HtmlTokenCallbacks} callbacks callbacks
- * @returns {number} final position
- */
- const tokenize = (input, pos = 0, callbacks = {}) => {
- const len = input.length;
- let state = STATE_DATA;
- let returnState = STATE_DATA;
- let textStart = pos;
- let tagStart = pos;
- let tagNameStart = -1;
- let tagNameEnd = -1;
- let attributeNameStart = -1;
- let attributeNameEnd = -1;
- let attributeValueStart = -1;
- let attrQuoteType = QUOTE_NONE;
- let commentStart = pos;
- // Data range of the comment being scanned. The comment states append
- // characters that are not the ones just consumed (the `--` a `-->` inside a
- // comment turns into, the `--!` of a comment-end-bang), so the range is
- // tracked here rather than re-derived by trimming delimiters off the token.
- let commentDataStart = pos;
- let commentDataEnd = pos;
- // Sub-ranges of the DOCTYPE token being scanned (-1 = the spec's
- // "missing" value, which a consumer must tell apart from an empty one).
- let doctypeNameStart = -1;
- let doctypeNameEnd = -1;
- let doctypePublicStart = -1;
- let doctypePublicEnd = -1;
- let doctypeSystemStart = -1;
- let doctypeSystemEnd = -1;
- let doctypeForceQuirks = false;
- let lastOpenTagName = "";
- // Tag-name offsets of the last open tag; the lowercased `lastOpenTagName` is
- // derived from these lazily (only for special-content tags).
- let lastOpenTagStart = -1;
- let lastOpenTagEnd = -1;
- // Counter used by SCRIPT_DATA_DOUBLE_ESCAPE_{START,END} to detect whether
- // the ASCII-alpha run after `<` / `</` spells exactly `"script"`. Values
- // 0..6 = number of chars matched so far; 7 = no longer matches (sentinel).
- // Avoids growing a buffer for pathological inputs with long alpha runs.
- let scriptMatch = 0;
- let namedEntityConsumed = 0;
- // Offset of the opening `&` and the running numeric value (clamped past the
- // Unicode range so it can't overflow); used for numeric-reference errors.
- let charRefStart = -1;
- let charRefCode = 0;
- // Tracks whether the current tag has parsed any attributes — used to
- // fire the `end-tag-with-attributes` parse error when an end tag emits.
- let tagHasAttributes = false;
- // Memoized next occurrence of `<` / `&` / NUL for the text-run
- // fast-forwards: native `indexOf` beats a per-char JS loop, and `pos` only
- // moves forward, so each memo is refreshed at most once per occurrence
- // (`len` = no further occurrence).
- let nextLt = -1;
- let nextAmp = -1;
- let nextNul = -1;
- let nextHyphen = -1;
- let nextGt = -1;
- // Same memo scheme for the closing quote of quoted attribute values (long
- // values: data: URIs, srcset, inline style). Memoized so a value with many
- // `&` references doesn't re-run the quote scan per reference.
- let nextDQuote = -1;
- let nextSQuote = -1;
- /**
- * Reports a tokenizer parse error to the consumer. The offset range and
- * severity follow the WHATWG spec naming. Severity is `"error"` for
- * cases where the emitted token is incomplete (EOF inside a tag or
- * comment); everything else is a `"warning"`. Offsets are JS string
- * indices (UTF-16 code-unit offsets into `input`).
- * @param {string} code WHATWG parse-error code (kebab-case)
- * @param {number} start string offset where the error starts
- * @param {number} end string offset where the error ends
- * @param {ParseErrorSeverity} severity error severity
- */
- const reportError = (code, start, end, severity) => {
- if (callbacks.parseError !== undefined) {
- callbacks.parseError(input, code, start, end, severity);
- }
- };
- /**
- * Emits the WHATWG numeric-character-reference validation parse error for
- * the accumulated `charRefCode`, if any. Used both inline (when the
- * reference is terminated by a real next character) and at EOF (when the
- * reference runs to the end of input). The scanner only flags the error —
- * the spec's U+FFFD / Windows-1252 substitution is done by `decodeEntities`.
- * @param {number} endPos offset just past the reference
- */
- const validateNumericReference = (endPos) => {
- if (charRefCode === 0) {
- reportError("null-character-reference", charRefStart, endPos, "warning");
- } else if (charRefCode > 0x10ffff) {
- reportError(
- "character-reference-outside-unicode-range",
- charRefStart,
- endPos,
- "warning"
- );
- } else if (charRefCode >= 0xd800 && charRefCode <= 0xdfff) {
- reportError(
- "surrogate-character-reference",
- charRefStart,
- endPos,
- "warning"
- );
- } else if (isNoncharacter(charRefCode)) {
- reportError(
- "noncharacter-character-reference",
- charRefStart,
- endPos,
- "warning"
- );
- } else if (isControlReference(charRefCode)) {
- reportError(
- "control-character-reference",
- charRefStart,
- endPos,
- "warning"
- );
- }
- };
- // Content mode for the tag just opened (name at `lastOpenTagStart..End`). In
- // foreign content (SVG/MathML) the tree builder vetoes RAWTEXT/RCDATA/script
- // switching via `isForeign`, so e.g. an SVG `<title>`/`<style>` is parsed as
- // normal markup. `lastOpenTagName` (the lowercased name compared by the
- // special end-tag states) is materialized only when a special mode is
- // actually entered — ordinary tags never allocate it.
- const contentModeAfterOpenTag = () => {
- const m = getContentModeForRange(input, lastOpenTagStart, lastOpenTagEnd);
- // Ordinary tags stay in the data state whatever the tree builder did, so
- // they pay for neither the name nor the callback.
- if (m === STATE_DATA) return STATE_DATA;
- lastOpenTagName = input
- .slice(lastOpenTagStart, lastOpenTagEnd)
- .toLowerCase();
- if (
- callbacks.vetoesContentMode !== undefined &&
- callbacks.vetoesContentMode(lastOpenTagName)
- ) {
- return STATE_DATA;
- }
- return m;
- };
- // HTML fragment parsing: seed the tokenizer with the context element's
- // content mode (e.g. a `textarea`/`style`/`script` context starts in
- // RCDATA/RAWTEXT/script-data rather than data state).
- if (callbacks.fragmentContext !== undefined) {
- lastOpenTagName = callbacks.fragmentContext;
- state =
- callbacks.vetoesContentMode !== undefined &&
- callbacks.vetoesContentMode(lastOpenTagName)
- ? STATE_DATA
- : getContentModeForTag(lastOpenTagName);
- }
- /**
- * @param {number} endPos end position
- */
- const flushText = (endPos) => {
- if (textStart < endPos) {
- if (callbacks.text !== undefined) {
- callbacks.text(input, textStart, endPos);
- }
- // Advance `textStart` so a second `flushText` for the same span
- // (e.g. from the EOF handler after a tag-open transition already
- // flushed the pending text) is a no-op rather than a duplicate
- // emit. emitOpenTag / emitCloseTag overwrite `textStart` with
- // their own `nextPos` anyway, so this doesn't shift their start.
- textStart = endPos;
- }
- };
- /**
- * @param {number} endPos end position
- * @returns {number} next position
- */
- const emitAttribute = (endPos) => {
- // Default `nextPos` advances past the closing quote (if any) so the
- // state machine can continue when no `attribute` callback is provided.
- // When a callback IS provided, its return value overrides the default —
- // the callback is expected to do the same advance based on the
- // reported `quoteType`.
- let nextPos = attrQuoteType === QUOTE_NONE ? endPos : endPos + 1;
- if (callbacks.attribute !== undefined && attributeNameStart !== -1) {
- nextPos = callbacks.attribute(
- input,
- attributeNameStart,
- attributeNameEnd,
- attributeValueStart,
- attributeValueStart === -1 ? -1 : endPos,
- attrQuoteType
- );
- }
- if (attributeNameStart !== -1) tagHasAttributes = true;
- attributeNameStart = -1;
- attributeValueStart = -1;
- attrQuoteType = QUOTE_NONE;
- return nextPos;
- };
- /**
- * @param {number} endPos end position
- * @param {boolean} selfClosing is self closing
- * @returns {number} next position
- */
- const emitOpenTag = (endPos, selfClosing) => {
- let nextPos = endPos;
- if (callbacks.openTag !== undefined) {
- nextPos = callbacks.openTag(
- input,
- tagStart,
- endPos,
- tagNameStart,
- tagNameEnd,
- selfClosing
- );
- }
- if (!selfClosing) {
- // Record offsets only; `contentModeAfterOpenTag` lowercases lazily.
- lastOpenTagStart = tagNameStart;
- lastOpenTagEnd = tagNameEnd;
- }
- tagHasAttributes = false;
- textStart = nextPos;
- return nextPos;
- };
- /**
- * @param {number} endPos end position
- * @returns {number} next position
- */
- const emitCloseTag = (endPos) => {
- // Per WHATWG: an end tag emitted with attributes is a parse error.
- if (tagHasAttributes) {
- reportError("end-tag-with-attributes", tagStart, endPos, "warning");
- }
- let nextPos = endPos;
- if (callbacks.closeTag !== undefined) {
- nextPos = callbacks.closeTag(
- input,
- tagStart,
- endPos,
- tagNameStart,
- tagNameEnd
- );
- }
- tagHasAttributes = false;
- textStart = nextPos;
- return nextPos;
- };
- while (pos < len) {
- const cc = input.charCodeAt(pos);
- // All WHATWG tokenizer states handled. Deliberately omitted parse errors
- // (need state this offset scanner lacks): duplicate-attribute,
- // cdata-in-html-content, `*-in-input-stream`. Reference substitution is
- // left to `decodeEntities`.
- switch (state) {
- // https://html.spec.whatwg.org/multipage/parsing.html#data-state
- case STATE_DATA:
- // Consume the next input character:
- // U+003C LESS-THAN SIGN (<)
- // Set the return state to the data state. Switch to the tag open state.
- if (cc === CC_LESS_THAN) {
- tagStart = pos;
- state = STATE_TAG_OPEN;
- pos++;
- } else if (cc === CC_AMPERSAND) {
- // U+0026 AMPERSAND (&)
- // Set the return state to the data state. Switch to the
- // character reference state.
- returnState = STATE_DATA;
- state = STATE_CHARACTER_REFERENCE;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL: unexpected-null-character (the data state
- // emits the NULL as-is; the scanner only flags the error).
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward over the run of ordinary text without re-entering
- // the per-state switch — batches only the data state's "Anything
- // else — emit the current input character" arc; the stops are
- // exactly the code points with their own arcs (`<`, `&`, NUL).
- pos++;
- if (nextLt < pos) {
- nextLt = input.indexOf("<", pos);
- if (nextLt === -1) nextLt = len;
- }
- if (nextAmp < pos) {
- nextAmp = input.indexOf("&", pos);
- if (nextAmp === -1) nextAmp = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextLt, nextAmp, nextNul);
- // Fused: the memo invariant guarantees `input[nextLt] === "<"`, so
- // take this state's `<` arc without another dispatch.
- if (pos === nextLt && pos < len) {
- tagStart = pos;
- state = STATE_TAG_OPEN;
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#tag-open-state
- case STATE_TAG_OPEN:
- // Consume the next input character:
- // U+002F SOLIDUS (/)
- // Switch to the end tag open state.
- if (cc === CC_SOLIDUS) {
- state = STATE_END_TAG_OPEN;
- pos++;
- // Fused: when the next char is alpha, take the end-tag-open state's
- // alpha arc (and its tag-name run scan) without another dispatch.
- if (pos < len && isAsciiAlpha(input.charCodeAt(pos))) {
- flushText(tagStart);
- tagNameStart = pos;
- state = STATE_TAG_NAME;
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (c2 < 0x80 && (CHAR_CLASS[c2] & CCLS_TAG_NAME_TERM) !== 0) {
- break;
- }
- pos++;
- }
- }
- } else if (cc === CC_EXCLAMATION_MARK) {
- // U+0021 EXCLAMATION MARK (!)
- // Switch to the markup declaration open state.
- flushText(tagStart);
- commentStart = tagStart;
- state = STATE_MARKUP_DECLARATION_OPEN;
- pos++;
- } else if (isAsciiAlpha(cc)) {
- // ASCII alpha
- // Create a new start tag token, set its tag name to the empty string.
- // Reconsume in the tag name state.
- flushText(tagStart);
- tagNameStart = pos;
- state = STATE_TAG_NAME;
- // Fused reconsume: the first char is alpha, so the tag-name state
- // always lands in its run scan — run it here without a dispatch.
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (c2 < 0x80 && (CHAR_CLASS[c2] & CCLS_TAG_NAME_TERM) !== 0) {
- break;
- }
- pos++;
- }
- } else if (cc === CC_QUESTION_MARK) {
- // U+003F QUESTION MARK (?)
- // This is an unexpected-question-mark-instead-of-tag-name parse error.
- // Create a comment token whose data is the empty string. Reconsume in the
- // bogus comment state.
- reportError(
- "unexpected-question-mark-instead-of-tag-name",
- pos,
- pos + 1,
- "warning"
- );
- flushText(tagStart);
- commentStart = tagStart;
- commentDataStart = pos;
- commentDataEnd = pos;
- state = STATE_BOGUS_COMMENT;
- // Reconsume — let the bogus-comment state consume the `?`
- // itself, matching the spec.
- } else {
- // Anything else
- // This is an invalid-first-character-of-tag-name parse error. Emit a U+003C
- // LESS-THAN SIGN character token. Reconsume in the data state.
- reportError(
- "invalid-first-character-of-tag-name",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#end-tag-open-state
- case STATE_END_TAG_OPEN:
- // Consume the next input character. The spec's ASCII-alpha arc (create
- // an end tag token, reconsume in the tag name state) is taken inline by
- // the `<` `/` arc above, the only way into this state, so `cc` is never
- // ASCII alpha here.
- if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-end-tag-name parse error. Switch to the data state.
- // No token is emitted, so `</>` is dropped rather than left in the
- // text run: a browser renders nothing for it, and keeping it would
- // resurface as visible text once the text node is re-escaped.
- reportError("missing-end-tag-name", pos, pos + 1, "warning");
- flushText(tagStart);
- state = STATE_DATA;
- pos++;
- textStart = pos;
- } else {
- // Anything else
- // This is an invalid-first-character-of-tag-name parse error. Create a
- // comment token whose data is the empty string. Reconsume in the bogus
- // comment state.
- reportError(
- "invalid-first-character-of-tag-name",
- pos,
- pos + 1,
- "warning"
- );
- flushText(tagStart);
- commentStart = tagStart;
- commentDataStart = pos;
- commentDataEnd = pos;
- state = STATE_BOGUS_COMMENT;
- // Reconsume — let bogus-comment consume this char itself.
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#tag-name-state
- case STATE_TAG_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the before attribute name state.
- if (isSpace(cc)) {
- tagNameEnd = pos;
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- pos++;
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // Switch to the self-closing start tag state.
- tagNameEnd = pos;
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current tag token.
- tagNameEnd = pos;
- if (input.charCodeAt(tagStart + 1) === CC_SOLIDUS) {
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- const nextPos = emitOpenTag(pos + 1, false);
- state = nextPos > pos + 1 ? STATE_DATA : contentModeAfterOpenTag();
- pos = nextPos;
- }
- } else {
- // U+0000 NULL: unexpected-null-character (append U+FFFD).
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- // Fast-forward over the ordinary run of the tag name.
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (c2 < 0x80 && (CHAR_CLASS[c2] & CCLS_TAG_NAME_TERM) !== 0) {
- break;
- }
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#before-attribute-name-state
- case STATE_BEFORE_ATTRIBUTE_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- // Reconsume so space is handled in BEFORE_ATTRIBUTE_NAME
- if (isSpace(cc)) {
- pos++;
- } else if (cc === CC_SOLIDUS || cc === CC_GREATER_THAN) {
- // U+002F SOLIDUS (/)
- // U+003E GREATER-THAN SIGN (>)
- // EOF
- // Reconsume in the after attribute name state.
- state = STATE_AFTER_ATTRIBUTE_NAME;
- // Reconsume
- } else if (cc === CC_EQUALS) {
- // U+003D EQUALS SIGN (=)
- // This is an unexpected-equals-sign-before-attribute-name parse
- // error. Start a new attribute. Switch to the attribute name state.
- reportError(
- "unexpected-equals-sign-before-attribute-name",
- pos,
- pos + 1,
- "warning"
- );
- attributeNameStart = pos;
- state = STATE_ATTRIBUTE_NAME;
- pos++;
- } else {
- // Anything else
- // Start a new attribute in the current tag token. Set that attribute name
- // and value to the empty string. Reconsume in the attribute name state.
- attributeNameStart = pos;
- state = STATE_ATTRIBUTE_NAME;
- // Fused reconsume: `cc` can't be a terminator here (space / `/` /
- // `>` / `=` took earlier arcs), so run the attribute-name state's
- // else arc — its first-char errors and run scan — without a dispatch.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- } else if (
- cc === CC_QUOTATION_MARK ||
- cc === CC_APOSTROPHE ||
- cc === CC_LESS_THAN
- ) {
- reportError(
- "unexpected-character-in-attribute-name",
- pos,
- pos + 1,
- "warning"
- );
- }
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (c2 < 0x80 && (CHAR_CLASS[c2] & CCLS_ATTR_NAME_TERM) !== 0) {
- break;
- }
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#attribute-name-state
- case STATE_ATTRIBUTE_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // U+002F SOLIDUS (/)
- // U+003E GREATER-THAN SIGN (>)
- // EOF
- // Reconsume in the after attribute name state.
- if (isSpace(cc) || cc === CC_SOLIDUS || cc === CC_GREATER_THAN) {
- attributeNameEnd = pos;
- state = STATE_AFTER_ATTRIBUTE_NAME;
- // Reconsume
- } else if (cc === CC_EQUALS) {
- attributeNameEnd = pos;
- state = STATE_BEFORE_ATTRIBUTE_VALUE;
- pos++;
- } else {
- // NULL -> unexpected-null-character; `"` `'` `<` -> unexpected-character-in-attribute-name.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- } else if (
- cc === CC_QUOTATION_MARK ||
- cc === CC_APOSTROPHE ||
- cc === CC_LESS_THAN
- ) {
- reportError(
- "unexpected-character-in-attribute-name",
- pos,
- pos + 1,
- "warning"
- );
- }
- // Fast-forward over the ordinary run of the attribute name; stop on
- // any terminator (space / `/` / `>` / `=`) or a char that needs a
- // per-occurrence parse error (NULL / `"` / `'` / `<`), which the
- // outer switch then re-handles.
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (c2 < 0x80 && (CHAR_CLASS[c2] & CCLS_ATTR_NAME_TERM) !== 0) {
- break;
- }
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-attribute-name-state
- case STATE_AFTER_ATTRIBUTE_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- if (isSpace(cc)) {
- pos++;
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // Switch to the self-closing start tag state.
- emitAttribute(pos);
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else if (cc === CC_EQUALS) {
- // U+003D EQUALS SIGN (=)
- // Switch to the before attribute value state.
- state = STATE_BEFORE_ATTRIBUTE_VALUE;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current tag token.
- emitAttribute(pos);
- if (input.charCodeAt(tagStart + 1) === CC_SOLIDUS) {
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- const nextPos = emitOpenTag(pos + 1, false);
- state = nextPos > pos + 1 ? STATE_DATA : contentModeAfterOpenTag();
- pos = nextPos;
- }
- } else {
- // Anything else
- // Start a new attribute in the current tag token.
- emitAttribute(pos);
- attributeNameStart = pos;
- state = STATE_ATTRIBUTE_NAME;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#before-attribute-value-state
- case STATE_BEFORE_ATTRIBUTE_VALUE:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- if (isSpace(cc)) {
- pos++;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // Switch to the attribute value (double-quoted) state.
- attributeValueStart = pos + 1;
- attrQuoteType = QUOTE_DOUBLE;
- state = STATE_ATTRIBUTE_VALUE_DOUBLE_QUOTED;
- pos++;
- // Fused: run that state's memoized value scan now; a leading `"` /
- // `&` / NUL keeps the min at `pos` for the next dispatch.
- if (nextDQuote < pos) {
- nextDQuote = input.indexOf('"', pos);
- if (nextDQuote === -1) nextDQuote = len;
- }
- if (nextAmp < pos) {
- nextAmp = input.indexOf("&", pos);
- if (nextAmp === -1) nextAmp = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextDQuote, nextAmp, nextNul);
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // Switch to the attribute value (single-quoted) state.
- attributeValueStart = pos + 1;
- attrQuoteType = QUOTE_SINGLE;
- state = STATE_ATTRIBUTE_VALUE_SINGLE_QUOTED;
- pos++;
- // Fused: run that state's memoized value scan now; a leading `'` /
- // `&` / NUL keeps the min at `pos` for the next dispatch.
- if (nextSQuote < pos) {
- nextSQuote = input.indexOf("'", pos);
- if (nextSQuote === -1) nextSQuote = len;
- }
- if (nextAmp < pos) {
- nextAmp = input.indexOf("&", pos);
- if (nextAmp === -1) nextAmp = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextSQuote, nextAmp, nextNul);
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-attribute-value parse error. Switch to the data
- // state. Emit the current tag token. The attribute is reported with
- // an empty value range pointing at the `>` so the open-tag offset range
- // still includes the `>`.
- reportError("missing-attribute-value", pos, pos + 1, "warning");
- attributeValueStart = pos;
- attrQuoteType = QUOTE_NONE;
- pos = emitAttribute(pos);
- if (input.charCodeAt(tagStart + 1) === CC_SOLIDUS) {
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- const nextPos = emitOpenTag(pos + 1, false);
- state = nextPos > pos + 1 ? STATE_DATA : contentModeAfterOpenTag();
- pos = nextPos;
- }
- } else {
- // Anything else
- // Reconsume in the attribute value (unquoted) state.
- attributeValueStart = pos;
- attrQuoteType = QUOTE_NONE;
- state = STATE_ATTRIBUTE_VALUE_UNQUOTED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#attribute-value-(double-quoted)-state
- case STATE_ATTRIBUTE_VALUE_DOUBLE_QUOTED:
- // Consume the next input character:
- // U+0022 QUOTATION MARK (")
- // Switch to the after attribute value (quoted) state.
- if (cc === CC_QUOTATION_MARK) {
- pos = emitAttribute(pos);
- state = STATE_AFTER_ATTRIBUTE_VALUE_QUOTED;
- } else if (cc === CC_AMPERSAND) {
- // U+0026 AMPERSAND (&)
- // Set the return state to the attribute value (double-quoted)
- // state. Switch to the character reference state.
- returnState = STATE_ATTRIBUTE_VALUE_DOUBLE_QUOTED;
- state = STATE_CHARACTER_REFERENCE;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL: unexpected-null-character (append U+FFFD).
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward over the ordinary run of the quoted value with the
- // same memoized native-scan scheme as the data state.
- pos++;
- if (nextDQuote < pos) {
- nextDQuote = input.indexOf('"', pos);
- if (nextDQuote === -1) nextDQuote = len;
- }
- if (nextAmp < pos) {
- nextAmp = input.indexOf("&", pos);
- if (nextAmp === -1) nextAmp = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextDQuote, nextAmp, nextNul);
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#attribute-value-(single-quoted)-state
- case STATE_ATTRIBUTE_VALUE_SINGLE_QUOTED:
- // Consume the next input character:
- // U+0027 APOSTROPHE (')
- // Switch to the after attribute value (quoted) state.
- if (cc === CC_APOSTROPHE) {
- pos = emitAttribute(pos);
- state = STATE_AFTER_ATTRIBUTE_VALUE_QUOTED;
- } else if (cc === CC_AMPERSAND) {
- // U+0026 AMPERSAND (&)
- // Set the return state to the attribute value (single-quoted)
- // state. Switch to the character reference state.
- returnState = STATE_ATTRIBUTE_VALUE_SINGLE_QUOTED;
- state = STATE_CHARACTER_REFERENCE;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL: unexpected-null-character (append U+FFFD).
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward over the ordinary run of the quoted value with the
- // same memoized native-scan scheme as the data state.
- pos++;
- if (nextSQuote < pos) {
- nextSQuote = input.indexOf("'", pos);
- if (nextSQuote === -1) nextSQuote = len;
- }
- if (nextAmp < pos) {
- nextAmp = input.indexOf("&", pos);
- if (nextAmp === -1) nextAmp = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextSQuote, nextAmp, nextNul);
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#attribute-value-(unquoted)-state
- case STATE_ATTRIBUTE_VALUE_UNQUOTED:
- if (isSpace(cc)) {
- pos = emitAttribute(pos);
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- // Reconsume so space is handled in BEFORE_ATTRIBUTE_NAME
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-attribute-value parse error. Switch to the data state.
- // Emit the current tag token.
- pos = emitAttribute(pos);
- if (input.charCodeAt(tagStart + 1) === CC_SOLIDUS) {
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- const nextPos = emitOpenTag(pos + 1, false);
- state = nextPos > pos + 1 ? STATE_DATA : contentModeAfterOpenTag();
- pos = nextPos;
- }
- } else if (cc === CC_AMPERSAND) {
- // U+0026 AMPERSAND (&)
- // Set the return state to the attribute value (unquoted)
- // state. Switch to the character reference state.
- returnState = STATE_ATTRIBUTE_VALUE_UNQUOTED;
- state = STATE_CHARACTER_REFERENCE;
- pos++;
- } else {
- // NULL -> unexpected-null-character; `"` `'` `<` `=` `` ` `` -> unexpected-character-in-unquoted-attribute-value.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- } else if (
- cc === CC_QUOTATION_MARK ||
- cc === CC_APOSTROPHE ||
- cc === CC_LESS_THAN ||
- cc === CC_EQUALS ||
- cc === CC_GRAVE_ACCENT
- ) {
- reportError(
- "unexpected-character-in-unquoted-attribute-value",
- pos,
- pos + 1,
- "warning"
- );
- }
- pos++;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-attribute-value-(quoted)-state
- case STATE_AFTER_ATTRIBUTE_VALUE_QUOTED:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the before attribute name state.
- if (isSpace(cc)) {
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- pos++;
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // Switch to the self-closing start tag state.
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- if (input.charCodeAt(tagStart + 1) === CC_SOLIDUS) {
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- const nextPos = emitOpenTag(pos + 1, false);
- state = nextPos > pos + 1 ? STATE_DATA : contentModeAfterOpenTag();
- pos = nextPos;
- }
- } else {
- // Anything else
- // This is a missing-whitespace-between-attributes parse error. Reconsume in
- // the before attribute name state.
- reportError(
- "missing-whitespace-between-attributes",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#self-closing-start-tag-state
- case STATE_SELF_CLOSING_START_TAG:
- // Consume the next input character:
- // U+003E GREATER-THAN SIGN (>)
- // Set the self-closing flag of the current tag token. Switch to the data
- // state. Emit the current tag token.
- if (cc === CC_GREATER_THAN) {
- if (input.charCodeAt(tagStart + 1) === CC_SOLIDUS) {
- // An end tag emitted with the self-closing flag set is an
- // end-tag-with-trailing-solidus parse error.
- reportError(
- "end-tag-with-trailing-solidus",
- tagStart,
- pos + 1,
- "warning"
- );
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- pos = emitOpenTag(pos + 1, true);
- state = STATE_DATA;
- }
- } else {
- // Anything else
- // This is an unexpected-solidus-in-tag parse error. Reconsume in the before
- // attribute name state.
- reportError("unexpected-solidus-in-tag", pos, pos + 1, "warning");
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#markup-declaration-open-state
- case STATE_MARKUP_DECLARATION_OPEN:
- // If the next few characters are:
- // Two U+002D HYPHEN-MINUS characters (-)
- // Consume those two characters, create a comment token whose data
- // is the empty string, and switch to the comment start state.
- if (
- cc === CC_HYPHEN_MINUS &&
- input.charCodeAt(pos + 1) === CC_HYPHEN_MINUS
- ) {
- pos += 2;
- commentStart = tagStart;
- commentDataStart = pos;
- commentDataEnd = pos;
- state = STATE_COMMENT_START;
- } else if (
- // ASCII case-insensitive match for the word "DOCTYPE"
- // Consume those characters and switch to the DOCTYPE state.
- (cc === 0x44 || cc === 0x64) /* D or d */ &&
- (input.charCodeAt(pos + 1) | 0x20) === 0x6f /* o */ &&
- (input.charCodeAt(pos + 2) | 0x20) === 0x63 /* c */ &&
- (input.charCodeAt(pos + 3) | 0x20) === 0x74 /* t */ &&
- (input.charCodeAt(pos + 4) | 0x20) === 0x79 /* y */ &&
- (input.charCodeAt(pos + 5) | 0x20) === 0x70 /* p */ &&
- (input.charCodeAt(pos + 6) | 0x20) === 0x65 /* e */
- ) {
- pos += 7;
- commentStart = tagStart;
- doctypeNameStart = -1;
- doctypeNameEnd = -1;
- doctypePublicStart = -1;
- doctypePublicEnd = -1;
- doctypeSystemStart = -1;
- doctypeSystemEnd = -1;
- doctypeForceQuirks = false;
- state = STATE_DOCTYPE;
- } else if (
- // The string "[CDATA[" (the five uppercase letters "CDATA" with a
- // U+005B LEFT SQUARE BRACKET character before and after)
- // Consume those characters and switch to the CDATA section state.
- // Only when there is an adjusted current node and it is not an
- // element in the HTML namespace: everywhere else this is the
- // "anything else" branch below, a bogus comment ending at the first
- // `>`, which leaves the rest (typically `]]>`) as text.
- cc === CC_LEFT_SQUARE_BRACKET &&
- input.charCodeAt(pos + 1) === 0x43 /* C */ &&
- input.charCodeAt(pos + 2) === 0x44 /* D */ &&
- input.charCodeAt(pos + 3) === 0x41 /* A */ &&
- input.charCodeAt(pos + 4) === 0x54 /* T */ &&
- input.charCodeAt(pos + 5) === 0x41 /* A */ &&
- input.charCodeAt(pos + 6) === CC_LEFT_SQUARE_BRACKET &&
- callbacks.isForeign !== undefined &&
- callbacks.isForeign()
- ) {
- pos += 7;
- commentStart = tagStart;
- commentDataStart = pos;
- commentDataEnd = pos;
- state = STATE_CDATA_SECTION;
- } else {
- // Anything else
- // This is an incorrectly-opened-comment parse error. Create a comment token
- // whose data is the empty string. Switch to the bogus comment state (don't
- // consume anything in the current state).
- reportError("incorrectly-opened-comment", tagStart, pos, "warning");
- commentStart = tagStart;
- commentDataStart = pos;
- commentDataEnd = pos;
- state = STATE_BOGUS_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-start-state
- case STATE_COMMENT_START:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the comment start dash state.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_COMMENT_START_DASH;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an abrupt-closing-of-empty-comment parse error. Switch to the
- // data state. Emit the current comment token.
- reportError(
- "abrupt-closing-of-empty-comment",
- pos,
- pos + 1,
- "warning"
- );
- let nextPos = pos + 1;
- if (callbacks.comment !== undefined) {
- nextPos = callbacks.comment(
- input,
- commentStart,
- pos + 1,
- commentDataStart,
- commentDataEnd
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Reconsume in the comment state.
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-start-dash-state
- case STATE_COMMENT_START_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the comment end state.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_COMMENT_END;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an abrupt-closing-of-empty-comment parse error. Switch to the
- // data state. Emit the current comment token.
- reportError(
- "abrupt-closing-of-empty-comment",
- pos,
- pos + 1,
- "warning"
- );
- let nextPos = pos + 1;
- if (callbacks.comment !== undefined) {
- nextPos = callbacks.comment(
- input,
- commentStart,
- pos + 1,
- commentDataStart,
- commentDataEnd
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Append a U+002D HYPHEN-MINUS character (-) to the comment token's data.
- // Reconsume in the comment state.
- commentDataEnd = pos;
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-state
- case STATE_COMMENT:
- // Consume the next input character:
- // U+003C LESS-THAN SIGN (<)
- // Append a U+003C LESS-THAN SIGN character to the comment token's data. Switch to the comment less-than sign state.
- if (cc === CC_LESS_THAN) {
- state = STATE_COMMENT_LESS_THAN_SIGN;
- pos++;
- commentDataEnd = pos;
- } else if (cc === CC_HYPHEN_MINUS) {
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the comment end dash state.
- state = STATE_COMMENT_END_DASH;
- pos++;
- } else {
- // U+0000 NULL: unexpected-null-character (append U+FFFD).
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- // Fast-forward over ordinary comment text (same memoized
- // `indexOf` scheme as the data state); stop on `<` / `-` / NUL.
- pos++;
- if (nextLt < pos) {
- nextLt = input.indexOf("<", pos);
- if (nextLt === -1) nextLt = len;
- }
- if (nextHyphen < pos) {
- nextHyphen = input.indexOf("-", pos);
- if (nextHyphen === -1) nextHyphen = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextLt, nextHyphen, nextNul);
- commentDataEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-end-dash-state
- case STATE_COMMENT_END_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the comment end state.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_COMMENT_END;
- pos++;
- } else {
- // Anything else
- // Append a U+002D HYPHEN-MINUS character (-) to the comment token's data.
- // Reconsume in the comment state (so e.g. NULL and `<` are
- // handled there).
- commentDataEnd = pos;
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-end-state
- case STATE_COMMENT_END:
- // Consume the next input character:
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current comment token.
- if (cc === CC_GREATER_THAN) {
- let nextPos = pos + 1;
- if (callbacks.comment !== undefined) {
- nextPos = callbacks.comment(
- input,
- commentStart,
- pos + 1,
- commentDataStart,
- commentDataEnd
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else if (cc === CC_EXCLAMATION_MARK) {
- // U+0021 EXCLAMATION MARK (!)
- // Switch to the comment end bang state.
- state = STATE_COMMENT_END_BANG;
- pos++;
- } else if (cc === CC_HYPHEN_MINUS) {
- // One of the two pending dashes joins the data; the other stays
- // pending, so the range grows by one rather than to `pos`.
- commentDataEnd++;
- pos++;
- } else {
- // Anything else
- // Append two U+002D HYPHEN-MINUS characters (-) to the comment token's
- // data. Reconsume in the comment state (so NULL and `<` are
- // handled there).
- commentDataEnd = pos;
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-end-bang-state
- case STATE_COMMENT_END_BANG:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Append two U+002D HYPHEN-MINUS characters (-) and a U+0021 EXCLAMATION
- // MARK character (!) to the comment token's data. Switch to the comment end
- // dash state.
- if (cc === CC_HYPHEN_MINUS) {
- commentDataEnd = pos;
- state = STATE_COMMENT_END_DASH;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an incorrectly-closed-comment parse error. Switch to the data
- // state. Emit the current comment token.
- reportError("incorrectly-closed-comment", pos, pos + 1, "warning");
- let nextPos = pos + 1;
- if (callbacks.comment !== undefined) {
- nextPos = callbacks.comment(
- input,
- commentStart,
- pos + 1,
- commentDataStart,
- commentDataEnd
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Append two U+002D HYPHEN-MINUS characters (-) and a U+0021 EXCLAMATION
- // MARK character (!) to the comment token's data. Reconsume in the comment
- // state (so NULL and `<` are handled there).
- commentDataEnd = pos;
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#bogus-comment-state
- case STATE_BOGUS_COMMENT:
- // Consume the next input character:
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current comment token.
- if (cc === CC_GREATER_THAN) {
- let nextPos = pos + 1;
- if (callbacks.comment !== undefined) {
- nextPos = callbacks.comment(
- input,
- commentStart,
- pos + 1,
- commentDataStart,
- commentDataEnd
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // U+0000 NULL: unexpected-null-character (append U+FFFD).
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- // Fast-forward over ordinary bogus-comment text (memoized `indexOf`
- // scheme); stop on `>` / NUL.
- pos++;
- if (nextGt < pos) {
- nextGt = input.indexOf(">", pos);
- if (nextGt === -1) nextGt = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextGt, nextNul);
- commentDataEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-less-than-sign-state
- case STATE_COMMENT_LESS_THAN_SIGN:
- // Consume the next input character:
- // U+0021 EXCLAMATION MARK (!)
- // Append the current input character to the comment token's data. Switch to
- // the comment less-than sign bang state.
- if (cc === CC_EXCLAMATION_MARK) {
- state = STATE_COMMENT_LESS_THAN_SIGN_BANG;
- pos++;
- commentDataEnd = pos;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Append the current input character to the comment token's data.
- pos++;
- commentDataEnd = pos;
- } else {
- // Anything else
- // Reconsume in the comment state.
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-less-than-sign-bang-state
- case STATE_COMMENT_LESS_THAN_SIGN_BANG:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the comment less-than sign bang dash state.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH;
- pos++;
- } else {
- // Anything else
- // Reconsume in the comment state.
- state = STATE_COMMENT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-less-than-sign-bang-dash-state
- case STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the comment less-than sign bang dash dash state.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH_DASH;
- pos++;
- } else {
- // Anything else
- // Reconsume in the comment end dash state.
- state = STATE_COMMENT_END_DASH;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#comment-less-than-sign-bang-dash-dash-state
- case STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH_DASH:
- // Consume the next input character:
- // U+003E GREATER-THAN SIGN (>)
- // EOF
- // Reconsume in the comment end state.
- // Anything else
- // This is a nested-comment parse error. Reconsume in the comment end state.
- if (cc !== CC_GREATER_THAN) {
- reportError("nested-comment", pos, pos + 1, "warning");
- }
- state = STATE_COMMENT_END;
- // Reconsume
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#doctype-state
- case STATE_DOCTYPE:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the before DOCTYPE name state.
- if (isSpace(cc)) {
- state = STATE_BEFORE_DOCTYPE_NAME;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Reconsume in the before DOCTYPE name state.
- state = STATE_BEFORE_DOCTYPE_NAME;
- } else {
- // Anything else
- // This is a missing-whitespace-before-doctype-name parse error. Reconsume
- // in the before DOCTYPE name state.
- reportError(
- "missing-whitespace-before-doctype-name",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_BEFORE_DOCTYPE_NAME;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#before-doctype-name-state
- case STATE_BEFORE_DOCTYPE_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- if (isSpace(cc)) {
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Create a new DOCTYPE
- // token. Set the token's name to a U+FFFD REPLACEMENT CHARACTER character.
- // Switch to the DOCTYPE name state.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- state = STATE_DOCTYPE_NAME;
- doctypeNameStart = pos;
- pos++;
- doctypeNameEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-doctype-name parse error. Create a new DOCTYPE token.
- // Set its force-quirks flag to on. Switch to the data state. Emit the
- // current token.
- reportError("missing-doctype-name", pos, pos + 1, "warning");
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // ASCII upper alpha
- // Create a new DOCTYPE token. Set the token's name to the lowercase version
- // of the current input character (add 0x0020 to the character's code
- // point). Switch to the DOCTYPE name state.
- // Anything else
- // Create a new DOCTYPE token. Set the token's name to the current input
- // character. Switch to the DOCTYPE name state.
- state = STATE_DOCTYPE_NAME;
- doctypeNameStart = pos;
- pos++;
- doctypeNameEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#doctype-name-state
- case STATE_DOCTYPE_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the after DOCTYPE name state.
- if (isSpace(cc)) {
- state = STATE_AFTER_DOCTYPE_NAME;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current DOCTYPE token.
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Append a U+FFFD
- // REPLACEMENT CHARACTER character to the current DOCTYPE token's name.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- doctypeNameEnd = pos;
- } else {
- // ASCII upper alpha
- // Append the lowercase version of the current input character (add 0x0020
- // to the character's code point) to the current DOCTYPE token's name.
- // Anything else
- // Append the current input character to the current DOCTYPE token's name.
- pos++;
- doctypeNameEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-doctype-name-state
- case STATE_AFTER_DOCTYPE_NAME:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current DOCTYPE token.
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else if (
- pos + 5 < len &&
- (cc === 0x50 || cc === 0x70) /* P or p */ &&
- (input.charCodeAt(pos + 1) | 0x20) === 0x75 /* u */ &&
- (input.charCodeAt(pos + 2) | 0x20) === 0x62 /* b */ &&
- (input.charCodeAt(pos + 3) | 0x20) === 0x6c /* l */ &&
- (input.charCodeAt(pos + 4) | 0x20) === 0x69 /* i */ &&
- (input.charCodeAt(pos + 5) | 0x20) === 0x63 /* c */
- ) {
- // ASCII case-insensitive match for the word "PUBLIC"
- pos += 6;
- state = STATE_AFTER_DOCTYPE_PUBLIC_KEYWORD;
- } else if (
- pos + 5 < len &&
- (cc === 0x53 || cc === 0x73) /* S or s */ &&
- (input.charCodeAt(pos + 1) | 0x20) === 0x79 /* y */ &&
- (input.charCodeAt(pos + 2) | 0x20) === 0x73 /* s */ &&
- (input.charCodeAt(pos + 3) | 0x20) === 0x74 /* t */ &&
- (input.charCodeAt(pos + 4) | 0x20) === 0x65 /* e */ &&
- (input.charCodeAt(pos + 5) | 0x20) === 0x6d /* m */
- ) {
- // ASCII case-insensitive match for the word "SYSTEM"
- pos += 6;
- state = STATE_AFTER_DOCTYPE_SYSTEM_KEYWORD;
- } else {
- // Anything else
- // This is an invalid-character-sequence-after-doctype-name parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "invalid-character-sequence-after-doctype-name",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-doctype-public-keyword-state
- case STATE_AFTER_DOCTYPE_PUBLIC_KEYWORD:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the before DOCTYPE public identifier state.
- state = STATE_BEFORE_DOCTYPE_PUBLIC_IDENTIFIER;
- pos++;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // This is a missing-whitespace-after-doctype-public-keyword parse error.
- // Set the current DOCTYPE token's public identifier to the empty string
- // (not missing), then switch to the DOCTYPE public identifier
- // (double-quoted) state.
- reportError(
- "missing-whitespace-after-doctype-public-keyword",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED;
- pos++;
- doctypePublicStart = pos;
- doctypePublicEnd = pos;
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // This is a missing-whitespace-after-doctype-public-keyword parse error.
- // Set the current DOCTYPE token's public identifier to the empty string
- // (not missing), then switch to the DOCTYPE public identifier
- // (single-quoted) state.
- reportError(
- "missing-whitespace-after-doctype-public-keyword",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED;
- pos++;
- doctypePublicStart = pos;
- doctypePublicEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-doctype-public-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "missing-doctype-public-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // This is a missing-quote-before-doctype-public-identifier parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "missing-quote-before-doctype-public-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#before-doctype-public-identifier-state
- case STATE_BEFORE_DOCTYPE_PUBLIC_IDENTIFIER:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- pos++;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // Set the current DOCTYPE token's public identifier to the empty string
- // (not missing), then switch to the DOCTYPE public identifier
- // (double-quoted) state.
- state = STATE_DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED;
- pos++;
- doctypePublicStart = pos;
- doctypePublicEnd = pos;
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // Set the current DOCTYPE token's public identifier to the empty string
- // (not missing), then switch to the DOCTYPE public identifier
- // (single-quoted) state.
- state = STATE_DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED;
- pos++;
- doctypePublicStart = pos;
- doctypePublicEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-doctype-public-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "missing-doctype-public-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // This is a missing-quote-before-doctype-public-identifier parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "missing-quote-before-doctype-public-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#doctype-public-identifier-(double-quoted)-state
- case STATE_DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED:
- // Consume the next input character:
- if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // Switch to the after DOCTYPE public identifier state.
- state = STATE_AFTER_DOCTYPE_PUBLIC_IDENTIFIER;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Append a U+FFFD
- // REPLACEMENT CHARACTER character to the current DOCTYPE token's public
- // identifier.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- doctypePublicEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an abrupt-doctype-public-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "abrupt-doctype-public-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Append the current input character to the current DOCTYPE token's public
- // identifier.
- pos++;
- doctypePublicEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#doctype-public-identifier-(single-quoted)-state
- case STATE_DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED:
- // Consume the next input character:
- if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // Switch to the after DOCTYPE public identifier state.
- state = STATE_AFTER_DOCTYPE_PUBLIC_IDENTIFIER;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Append a U+FFFD
- // REPLACEMENT CHARACTER character to the current DOCTYPE token's public
- // identifier.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- doctypePublicEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an abrupt-doctype-public-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "abrupt-doctype-public-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Append the current input character to the current DOCTYPE token's public
- // identifier.
- pos++;
- doctypePublicEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-doctype-public-identifier-state
- case STATE_AFTER_DOCTYPE_PUBLIC_IDENTIFIER:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the between DOCTYPE public and system identifiers state.
- state = STATE_BETWEEN_DOCTYPE_PUBLIC_AND_SYSTEM_IDENTIFIERS;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current DOCTYPE token.
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // This is a missing-whitespace-between-doctype-public-and-system-identifiers
- // parse error. Set the current DOCTYPE token's system
- // identifier to the empty string (not missing), then switch
- // to the DOCTYPE system identifier (double-quoted) state.
- reportError(
- "missing-whitespace-between-doctype-public-and-system-identifiers",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // This is a missing-whitespace-between-doctype-public-and-system-identifiers
- // parse error. Set the current DOCTYPE token's system
- // identifier to the empty string (not missing), then switch
- // to the DOCTYPE system identifier (single-quoted) state.
- reportError(
- "missing-whitespace-between-doctype-public-and-system-identifiers",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else {
- // Anything else
- // This is a missing-quote-before-doctype-system-identifier parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "missing-quote-before-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#between-doctype-public-and-system-identifiers-state
- case STATE_BETWEEN_DOCTYPE_PUBLIC_AND_SYSTEM_IDENTIFIERS:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current DOCTYPE token.
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // Set the current DOCTYPE token's system identifier to the empty string
- // (not missing), then switch to the DOCTYPE system identifier
- // (double-quoted) state.
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // Set the current DOCTYPE token's system identifier to the empty string
- // (not missing), then switch to the DOCTYPE system identifier
- // (single-quoted) state.
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else {
- // Anything else
- // This is a missing-quote-before-doctype-system-identifier parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "missing-quote-before-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-doctype-system-keyword-state
- case STATE_AFTER_DOCTYPE_SYSTEM_KEYWORD:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Switch to the before DOCTYPE system identifier state.
- state = STATE_BEFORE_DOCTYPE_SYSTEM_IDENTIFIER;
- pos++;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // This is a missing-whitespace-after-doctype-system-keyword parse error.
- // Set the current DOCTYPE token's system identifier to the empty string
- // (not missing), then switch to the DOCTYPE system identifier
- // (double-quoted) state.
- reportError(
- "missing-whitespace-after-doctype-system-keyword",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // This is a missing-whitespace-after-doctype-system-keyword parse error.
- // Set the current DOCTYPE token's system identifier to the empty string
- // (not missing), then switch to the DOCTYPE system identifier
- // (single-quoted) state.
- reportError(
- "missing-whitespace-after-doctype-system-keyword",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-doctype-system-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "missing-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // This is a missing-quote-before-doctype-system-identifier parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "missing-quote-before-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#before-doctype-system-identifier-state
- case STATE_BEFORE_DOCTYPE_SYSTEM_IDENTIFIER:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- pos++;
- } else if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // Set the current DOCTYPE token's system identifier to the empty string
- // (not missing), then switch to the DOCTYPE system identifier
- // (double-quoted) state.
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // Set the current DOCTYPE token's system identifier to the empty string
- // (not missing), then switch to the DOCTYPE system identifier
- // (single-quoted) state.
- state = STATE_DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED;
- pos++;
- doctypeSystemStart = pos;
- doctypeSystemEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is a missing-doctype-system-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "missing-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // This is a missing-quote-before-doctype-system-identifier parse error. Set
- // the current DOCTYPE token's force-quirks flag to on. Reconsume in the
- // bogus DOCTYPE state.
- reportError(
- "missing-quote-before-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#doctype-system-identifier-(double-quoted)-state
- case STATE_DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED:
- // Consume the next input character:
- if (cc === CC_QUOTATION_MARK) {
- // U+0022 QUOTATION MARK (")
- // Switch to the after DOCTYPE system identifier state.
- state = STATE_AFTER_DOCTYPE_SYSTEM_IDENTIFIER;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Append a U+FFFD
- // REPLACEMENT CHARACTER character to the current DOCTYPE token's system
- // identifier.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- doctypeSystemEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an abrupt-doctype-system-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "abrupt-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Append the current input character to the current DOCTYPE token's system
- // identifier.
- pos++;
- doctypeSystemEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#doctype-system-identifier-(single-quoted)-state
- case STATE_DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED:
- // Consume the next input character:
- if (cc === CC_APOSTROPHE) {
- // U+0027 APOSTROPHE (')
- // Switch to the after DOCTYPE system identifier state.
- state = STATE_AFTER_DOCTYPE_SYSTEM_IDENTIFIER;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Append a U+FFFD
- // REPLACEMENT CHARACTER character to the current DOCTYPE token's system
- // identifier.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- doctypeSystemEnd = pos;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // This is an abrupt-doctype-system-identifier parse error. Set the current
- // DOCTYPE token's force-quirks flag to on. Switch to the data state. Emit
- // the current DOCTYPE token.
- reportError(
- "abrupt-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- doctypeForceQuirks = true;
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Append the current input character to the current DOCTYPE token's system
- // identifier.
- pos++;
- doctypeSystemEnd = pos;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#after-doctype-system-identifier-state
- case STATE_AFTER_DOCTYPE_SYSTEM_IDENTIFIER:
- // Consume the next input character:
- if (isSpace(cc)) {
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // Ignore the character.
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the current DOCTYPE token.
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // This is an unexpected-character-after-doctype-system-identifier parse
- // error. Reconsume in the bogus DOCTYPE state. (This does not set the
- // current DOCTYPE token's force-quirks flag to on.)
- reportError(
- "unexpected-character-after-doctype-system-identifier",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_BOGUS_DOCTYPE;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#bogus-doctype-state
- case STATE_BOGUS_DOCTYPE:
- // Consume the next input character:
- if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. Emit the DOCTYPE token.
- let nextPos = pos + 1;
- if (callbacks.doctype !== undefined) {
- nextPos = callbacks.doctype(
- input,
- commentStart,
- pos + 1,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else if (cc === CC_NULL) {
- // U+0000 NULL
- // This is an unexpected-null-character parse error. Ignore the character.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Anything else
- // Ignore the character.
- pos++;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#cdata-section-state
- case STATE_CDATA_SECTION:
- // Consume the next input character:
- // U+005D RIGHT SQUARE BRACKET (])
- // Switch to the CDATA section bracket state.
- if (cc === CC_RIGHT_SQUARE_BRACKET) {
- state = STATE_CDATA_SECTION_BRACKET;
- pos++;
- } else {
- // Anything else
- // Emit the current input character as a character token.
- // Fast-forward to the next `]` (the only code point with its own
- // arc) in one native scan.
- pos = input.indexOf("]", pos + 1);
- if (pos === -1) pos = len;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#cdata-section-bracket-state
- case STATE_CDATA_SECTION_BRACKET:
- // Consume the next input character:
- // U+005D RIGHT SQUARE BRACKET (])
- // Switch to the CDATA section end state.
- if (cc === CC_RIGHT_SQUARE_BRACKET) {
- state = STATE_CDATA_SECTION_END;
- pos++;
- } else {
- // Anything else
- // Emit a U+005D RIGHT SQUARE BRACKET character token. Reconsume in the
- // CDATA section state.
- state = STATE_CDATA_SECTION;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#cdata-section-end-state
- case STATE_CDATA_SECTION_END:
- // Consume the next input character:
- // U+005D RIGHT SQUARE BRACKET (])
- // Emit a U+005D RIGHT SQUARE BRACKET character token.
- if (cc === CC_RIGHT_SQUARE_BRACKET) {
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the data state. The `]]` just consumed closes the
- // section rather than joining its content.
- commentDataEnd = pos - 2;
- let nextPos = pos + 1;
- if (callbacks.comment !== undefined) {
- nextPos = callbacks.comment(
- input,
- commentStart,
- pos + 1,
- commentDataStart,
- commentDataEnd
- );
- }
- state = STATE_DATA;
- textStart = nextPos;
- pos = nextPos;
- } else {
- // Anything else
- // Emit two U+005D RIGHT SQUARE BRACKET character tokens. Reconsume in the
- // CDATA section state.
- state = STATE_CDATA_SECTION;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rcdata-state
- case STATE_RCDATA:
- // Consume the next input character:
- if (cc === CC_AMPERSAND) {
- // U+0026 AMPERSAND (&)
- // Set the return state to the RCDATA state. Switch to the
- // character reference state. (RCDATA processes references;
- // RAWTEXT/script/PLAINTEXT do not.)
- returnState = STATE_RCDATA;
- state = STATE_CHARACTER_REFERENCE;
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the RCDATA less-than sign state.
- tagStart = pos;
- state = STATE_RCDATA_LESS_THAN_SIGN;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL is an unexpected-null-character parse error.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward over ordinary RCDATA text (same memoized
- // `indexOf` scheme as the data state) — batches only this
- // state's "anything else" arc; `&` / `<` / NUL keep their own.
- pos++;
- if (nextLt < pos) {
- nextLt = input.indexOf("<", pos);
- if (nextLt === -1) nextLt = len;
- }
- if (nextAmp < pos) {
- nextAmp = input.indexOf("&", pos);
- if (nextAmp === -1) nextAmp = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextLt, nextAmp, nextNul);
- // Fused: take this state's `<` arc without another dispatch.
- if (pos === nextLt && pos < len) {
- tagStart = pos;
- state = STATE_RCDATA_LESS_THAN_SIGN;
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rcdata-less-than-sign-state
- case STATE_RCDATA_LESS_THAN_SIGN:
- // Consume the next input character:
- // U+002F SOLIDUS (/)
- // Switch to the RCDATA end tag open state. (Spec sets a
- // temporary buffer here; we track the would-be content via
- // offset ranges instead.)
- if (cc === CC_SOLIDUS) {
- state = STATE_RCDATA_END_TAG_OPEN;
- pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token. Reconsume in the RCDATA
- // state.
- state = STATE_RCDATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rcdata-end-tag-open-state
- case STATE_RCDATA_END_TAG_OPEN:
- // Consume the next input character:
- // ASCII alpha
- // Create a new end tag token, set its tag name to the empty string.
- // Reconsume in the RCDATA end tag name state.
- if (isAsciiAlpha(cc)) {
- tagNameStart = pos;
- state = STATE_RCDATA_END_TAG_NAME;
- // Reconsume
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token and a U+002F SOLIDUS
- // character token. Reconsume in the RCDATA state.
- state = STATE_RCDATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rcdata-end-tag-name-state
- case STATE_RCDATA_END_TAG_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // If the current end tag token is an appropriate end tag token, then switch
- // to the before attribute name state. Otherwise, treat it as per the
- // "anything else" entry below.
- if (isSpace(cc)) {
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- pos++;
- } else {
- state = STATE_RCDATA;
- // Reconsume
- }
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the self-closing start tag state. Otherwise, treat it as per the
- // "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else {
- state = STATE_RCDATA;
- // Reconsume
- }
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the data state and emit the current tag token. Otherwise, treat it as
- // per the "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- state = STATE_RCDATA;
- // Reconsume
- }
- } else if (isAsciiAlpha(cc)) {
- // ASCII upper alpha / ASCII lower alpha
- // Append the lowercase version of the current input character to the
- // current tag token's tag name. Append the current input character to
- // the temporary buffer.
- // Fused: consume the whole alpha run in one dispatch.
- pos++;
- while (pos < len && isAsciiAlpha(input.charCodeAt(pos))) pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token, a U+002F SOLIDUS character
- // token, and a character token for each of the characters in the temporary
- // buffer (in the order they were added to the buffer). Reconsume in the
- // RCDATA state.
- state = STATE_RCDATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rawtext-state
- case STATE_RAWTEXT:
- // Consume the next input character:
- // U+003C LESS-THAN SIGN (<)
- // Switch to the RAWTEXT less-than sign state.
- if (cc === CC_LESS_THAN) {
- tagStart = pos;
- state = STATE_RAWTEXT_LESS_THAN_SIGN;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL is an unexpected-null-character parse error.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward over ordinary RAWTEXT text (same memoized
- // `indexOf` scheme as the data state) — batches only this
- // state's "anything else" arc; `<` / NUL keep their own.
- pos++;
- if (nextLt < pos) {
- nextLt = input.indexOf("<", pos);
- if (nextLt === -1) nextLt = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextLt, nextNul);
- // Fused: take this state's `<` arc without another dispatch.
- if (pos === nextLt && pos < len) {
- tagStart = pos;
- state = STATE_RAWTEXT_LESS_THAN_SIGN;
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rawtext-less-than-sign-state
- case STATE_RAWTEXT_LESS_THAN_SIGN:
- // Consume the next input character:
- // U+002F SOLIDUS (/)
- // Switch to the RAWTEXT end tag open state. (Spec sets a
- // temporary buffer here; we track via offset ranges instead.)
- if (cc === CC_SOLIDUS) {
- state = STATE_RAWTEXT_END_TAG_OPEN;
- pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token. Reconsume in the RAWTEXT
- // state.
- state = STATE_RAWTEXT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rawtext-end-tag-open-state
- case STATE_RAWTEXT_END_TAG_OPEN:
- // Consume the next input character:
- // ASCII alpha
- // Create a new end tag token, set its tag name to the empty string.
- // Reconsume in the RAWTEXT end tag name state.
- if (isAsciiAlpha(cc)) {
- tagNameStart = pos;
- state = STATE_RAWTEXT_END_TAG_NAME;
- // Reconsume
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token and a U+002F SOLIDUS
- // character token. Reconsume in the RAWTEXT state.
- state = STATE_RAWTEXT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#rawtext-end-tag-name-state
- case STATE_RAWTEXT_END_TAG_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // If the current end tag token is an appropriate end tag token, then switch
- // to the before attribute name state. Otherwise, treat it as per the
- // "anything else" entry below.
- if (isSpace(cc)) {
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- pos++;
- } else {
- state = STATE_RAWTEXT;
- }
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the self-closing start tag state. Otherwise, treat it as per the
- // "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else {
- state = STATE_RAWTEXT;
- }
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the data state and emit the current tag token. Otherwise, treat it as
- // per the "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- state = STATE_RAWTEXT;
- }
- } else if (isAsciiAlpha(cc)) {
- // ASCII upper alpha / ASCII lower alpha
- // Append the lowercase version of the current input character to the
- // current tag token's tag name. Append the current input character to
- // the temporary buffer.
- // Fused: consume the whole alpha run in one dispatch.
- pos++;
- while (pos < len && isAsciiAlpha(input.charCodeAt(pos))) pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token, a U+002F SOLIDUS character
- // token, and a character token for each of the characters in the temporary
- // buffer (in the order they were added to the buffer). Reconsume in the
- // RAWTEXT state.
- state = STATE_RAWTEXT;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-state
- case STATE_SCRIPT_DATA:
- // Consume the next input character:
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data less-than sign state.
- if (cc === CC_LESS_THAN) {
- tagStart = pos;
- state = STATE_SCRIPT_DATA_LESS_THAN_SIGN;
- pos++;
- } else if (cc === CC_NULL) {
- // U+0000 NULL is an unexpected-null-character parse error.
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward over ordinary script-data text (same memoized
- // `indexOf` scheme as the data state) — batches only this
- // state's "anything else" arc; `<` / NUL keep their own.
- pos++;
- if (nextLt < pos) {
- nextLt = input.indexOf("<", pos);
- if (nextLt === -1) nextLt = len;
- }
- if (nextNul < pos) {
- nextNul = input.indexOf("\0", pos);
- if (nextNul === -1) nextNul = len;
- }
- pos = Math.min(nextLt, nextNul);
- // Fused: take this state's `<` arc without another dispatch.
- if (pos === nextLt && pos < len) {
- tagStart = pos;
- state = STATE_SCRIPT_DATA_LESS_THAN_SIGN;
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-less-than-sign-state
- case STATE_SCRIPT_DATA_LESS_THAN_SIGN:
- // Consume the next input character:
- // U+002F SOLIDUS (/)
- // Switch to the script data end tag open state. (Spec sets a
- // temporary buffer here; we track via offset ranges instead.)
- if (cc === CC_SOLIDUS) {
- state = STATE_SCRIPT_DATA_END_TAG_OPEN;
- pos++;
- } else if (cc === CC_EXCLAMATION_MARK) {
- // U+0021 EXCLAMATION MARK (!)
- // Switch to the script data escape start state. Emit a U+003C LESS-THAN
- // SIGN character token and a U+0021 EXCLAMATION MARK character token.
- state = STATE_SCRIPT_DATA_ESCAPE_START;
- pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token. Reconsume in the script
- // data state.
- state = STATE_SCRIPT_DATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-end-tag-open-state
- case STATE_SCRIPT_DATA_END_TAG_OPEN:
- // Consume the next input character:
- // ASCII alpha
- // Create a new end tag token, set its tag name to the empty string.
- // Reconsume in the script data end tag name state.
- if (isAsciiAlpha(cc)) {
- tagNameStart = pos;
- state = STATE_SCRIPT_DATA_END_TAG_NAME;
- // Reconsume
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token and a U+002F SOLIDUS
- // character token. Reconsume in the script data state.
- state = STATE_SCRIPT_DATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-end-tag-name-state
- case STATE_SCRIPT_DATA_END_TAG_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // If the current end tag token is an appropriate end tag token, then switch
- // to the before attribute name state. Otherwise, treat it as per the
- // "anything else" entry below.
- if (isSpace(cc)) {
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- pos++;
- } else {
- state = STATE_SCRIPT_DATA;
- }
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the self-closing start tag state. Otherwise, treat it as per the
- // "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else {
- state = STATE_SCRIPT_DATA;
- }
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the data state and emit the current tag token. Otherwise, treat it as
- // per the "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- state = STATE_SCRIPT_DATA;
- }
- } else if (isAsciiAlpha(cc)) {
- // ASCII upper alpha / ASCII lower alpha
- // Fused: consume the whole alpha run in one dispatch.
- pos++;
- while (pos < len && isAsciiAlpha(input.charCodeAt(pos))) pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token, a U+002F SOLIDUS character
- // token, and a character token for each of the characters in the temporary
- // buffer (in the order they were added to the buffer). Reconsume in the
- // script data state.
- state = STATE_SCRIPT_DATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escape-start-state
- case STATE_SCRIPT_DATA_ESCAPE_START:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the script data escape start dash state. Emit a U+002D
- // HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_SCRIPT_DATA_ESCAPE_START_DASH;
- pos++;
- } else {
- // Anything else
- // Reconsume in the script data state.
- state = STATE_SCRIPT_DATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escape-start-dash-state
- case STATE_SCRIPT_DATA_ESCAPE_START_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the script data escaped dash dash state. Emit a U+002D
- // HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_SCRIPT_DATA_ESCAPED_DASH_DASH;
- pos++;
- } else {
- // Anything else
- // Reconsume in the script data state.
- state = STATE_SCRIPT_DATA;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escaped-state
- case STATE_SCRIPT_DATA_ESCAPED:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the script data escaped dash state. Emit a U+002D HYPHEN-MINUS
- // character token.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_SCRIPT_DATA_ESCAPED_DASH;
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data escaped less-than sign state.
- tagStart = pos;
- state = STATE_SCRIPT_DATA_ESCAPED_LESS_THAN_SIGN;
- pos++;
- } else {
- // Anything else
- // Emit the current input character as a character token.
- // U+0000 NULL is an unexpected-null-character parse error.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- // Fast-forward ordinary escaped script text without re-entering
- // the state switch; stop on the significant code points above.
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (
- c2 === CC_HYPHEN_MINUS ||
- c2 === CC_LESS_THAN ||
- c2 === CC_NULL
- ) {
- break;
- }
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escaped-dash-state
- case STATE_SCRIPT_DATA_ESCAPED_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the script data escaped dash dash state. Emit a U+002D
- // HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_SCRIPT_DATA_ESCAPED_DASH_DASH;
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data escaped less-than sign state.
- tagStart = pos;
- state = STATE_SCRIPT_DATA_ESCAPED_LESS_THAN_SIGN;
- pos++;
- } else {
- // Anything else
- // Switch to the script data escaped state. Emit the current input character
- // as a character token. U+0000 NULL is an unexpected-null-character error.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- state = STATE_SCRIPT_DATA_ESCAPED;
- pos++;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escaped-dash-dash-state
- case STATE_SCRIPT_DATA_ESCAPED_DASH_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Emit a U+002D HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data escaped less-than sign state.
- tagStart = pos;
- state = STATE_SCRIPT_DATA_ESCAPED_LESS_THAN_SIGN;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the script data state. Emit a U+003E GREATER-THAN SIGN
- // character token.
- state = STATE_SCRIPT_DATA;
- pos++;
- } else {
- // Anything else
- // Switch to the script data escaped state. Emit the current input character
- // as a character token. U+0000 NULL is an unexpected-null-character error.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- state = STATE_SCRIPT_DATA_ESCAPED;
- pos++;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escaped-less-than-sign-state
- case STATE_SCRIPT_DATA_ESCAPED_LESS_THAN_SIGN:
- // Consume the next input character:
- // U+002F SOLIDUS (/)
- // Switch to the script data escaped end tag open state.
- // (Spec sets a temporary buffer; we track via offset ranges.)
- if (cc === CC_SOLIDUS) {
- state = STATE_SCRIPT_DATA_ESCAPED_END_TAG_OPEN;
- pos++;
- } else if (isAsciiAlpha(cc)) {
- // ASCII alpha
- // Set the temporary buffer to the empty string. Emit a U+003C LESS-THAN
- // SIGN character token. Reconsume in the script data double escape start
- // state.
- scriptMatch = 0;
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPE_START;
- // Reconsume
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token. Reconsume in the script
- // data escaped state.
- state = STATE_SCRIPT_DATA_ESCAPED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escaped-end-tag-open-state
- case STATE_SCRIPT_DATA_ESCAPED_END_TAG_OPEN:
- // Consume the next input character:
- // ASCII alpha
- // Create a new end tag token, set its tag name to the empty string.
- // Reconsume in the script data escaped end tag name state.
- if (isAsciiAlpha(cc)) {
- tagNameStart = pos;
- state = STATE_SCRIPT_DATA_ESCAPED_END_TAG_NAME;
- // Reconsume
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token and a U+002F SOLIDUS
- // character token. Reconsume in the script data escaped state.
- state = STATE_SCRIPT_DATA_ESCAPED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-escaped-end-tag-name-state
- case STATE_SCRIPT_DATA_ESCAPED_END_TAG_NAME:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // If the current end tag token is an appropriate end tag token, then switch
- // to the before attribute name state. Otherwise, treat it as per the
- // "anything else" entry below.
- if (isSpace(cc)) {
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_BEFORE_ATTRIBUTE_NAME;
- pos++;
- } else {
- state = STATE_SCRIPT_DATA_ESCAPED;
- }
- } else if (cc === CC_SOLIDUS) {
- // U+002F SOLIDUS (/)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the self-closing start tag state. Otherwise, treat it as per the
- // "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_SELF_CLOSING_START_TAG;
- pos++;
- } else {
- state = STATE_SCRIPT_DATA_ESCAPED;
- }
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // If the current end tag token is an appropriate end tag token, then switch
- // to the data state and emit the current tag token. Otherwise, treat it as
- // per the "anything else" entry below.
- tagNameEnd = pos;
- if (
- rangeEqualsLowerCase(
- input,
- tagNameStart,
- tagNameEnd,
- lastOpenTagName
- )
- ) {
- flushText(tagStart);
- state = STATE_DATA;
- pos = emitCloseTag(pos + 1);
- } else {
- state = STATE_SCRIPT_DATA_ESCAPED;
- }
- } else if (isAsciiAlpha(cc)) {
- // ASCII upper alpha / ASCII lower alpha
- // Fused: consume the whole alpha run in one dispatch.
- pos++;
- while (pos < len && isAsciiAlpha(input.charCodeAt(pos))) pos++;
- } else {
- // Anything else
- // Emit a U+003C LESS-THAN SIGN character token, a U+002F SOLIDUS character
- // token, and a character token for each of the characters in the temporary
- // buffer (in the order they were added to the buffer). Reconsume in the
- // script data escaped state.
- state = STATE_SCRIPT_DATA_ESCAPED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-double-escape-start-state
- case STATE_SCRIPT_DATA_DOUBLE_ESCAPE_START:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // U+002F SOLIDUS (/)
- // U+003E GREATER-THAN SIGN (>)
- // If the temporary buffer is the string "script", then switch to the script
- // data double escaped state. Otherwise, switch to the script data escaped
- // state. Emit the current input character as a character token.
- if (isSpace(cc) || cc === CC_SOLIDUS || cc === CC_GREATER_THAN) {
- state =
- scriptMatch === 6
- ? STATE_SCRIPT_DATA_DOUBLE_ESCAPED
- : STATE_SCRIPT_DATA_ESCAPED;
- pos++;
- } else if (isAsciiUpperAlpha(cc) || isAsciiLowerAlpha(cc)) {
- // ASCII alpha — advance the `"script"` match counter if the
- // lowercase form matches the next expected char, otherwise
- // snap to the sentinel so further chars can't revive a
- // match. No buffer allocation.
- const lower = isAsciiUpperAlpha(cc) ? cc + 0x20 : cc;
- if (scriptMatch < 6 && lower === "script".charCodeAt(scriptMatch)) {
- scriptMatch++;
- } else {
- scriptMatch = 7;
- }
- pos++;
- } else {
- // Anything else
- // Reconsume in the script data escaped state.
- state = STATE_SCRIPT_DATA_ESCAPED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-double-escaped-state
- case STATE_SCRIPT_DATA_DOUBLE_ESCAPED:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the script data double escaped dash state. Emit a U+002D
- // HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH;
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data double escaped less-than sign state. Emit a
- // U+003C LESS-THAN SIGN character token.
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED_LESS_THAN_SIGN;
- pos++;
- } else {
- // Anything else
- // Emit the current input character as a character token.
- // U+0000 NULL is an unexpected-null-character parse error.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- // Fast-forward ordinary double-escaped script text without re-entering
- // the state switch; stop on the significant code points above.
- pos++;
- while (pos < len) {
- const c2 = input.charCodeAt(pos);
- if (
- c2 === CC_HYPHEN_MINUS ||
- c2 === CC_LESS_THAN ||
- c2 === CC_NULL
- ) {
- break;
- }
- pos++;
- }
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-double-escaped-dash-state
- case STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Switch to the script data double escaped dash dash state. Emit a U+002D
- // HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH_DASH;
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data double escaped less-than sign state. Emit a
- // U+003C LESS-THAN SIGN character token.
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED_LESS_THAN_SIGN;
- pos++;
- } else {
- // Anything else
- // Switch to the script data double escaped state. Emit the current input
- // character as a character token. NULL is unexpected-null-character.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED;
- pos++;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-double-escaped-dash-dash-state
- case STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH_DASH:
- // Consume the next input character:
- // U+002D HYPHEN-MINUS (-)
- // Emit a U+002D HYPHEN-MINUS character token.
- if (cc === CC_HYPHEN_MINUS) {
- pos++;
- } else if (cc === CC_LESS_THAN) {
- // U+003C LESS-THAN SIGN (<)
- // Switch to the script data double escaped less-than sign state. Emit a
- // U+003C LESS-THAN SIGN character token.
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED_LESS_THAN_SIGN;
- pos++;
- } else if (cc === CC_GREATER_THAN) {
- // U+003E GREATER-THAN SIGN (>)
- // Switch to the script data state. Emit a U+003E GREATER-THAN SIGN
- // character token.
- state = STATE_SCRIPT_DATA;
- pos++;
- } else {
- // Anything else
- // Switch to the script data double escaped state. Emit the current input
- // character as a character token. NULL is unexpected-null-character.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- }
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED;
- pos++;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-double-escaped-less-than-sign-state
- case STATE_SCRIPT_DATA_DOUBLE_ESCAPED_LESS_THAN_SIGN:
- // Consume the next input character:
- // U+002F SOLIDUS (/)
- // Set the temporary buffer to the empty string. Switch to the script data
- // double escape end state. Emit a U+002F SOLIDUS character token.
- if (cc === CC_SOLIDUS) {
- scriptMatch = 0;
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPE_END;
- pos++;
- } else {
- // Anything else
- // Reconsume in the script data double escaped state.
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#script-data-double-escape-end-state
- case STATE_SCRIPT_DATA_DOUBLE_ESCAPE_END:
- // Consume the next input character:
- // U+0009 CHARACTER TABULATION (tab)
- // U+000A LINE FEED (LF)
- // U+000C FORM FEED (FF)
- // U+0020 SPACE
- // U+002F SOLIDUS (/)
- // U+003E GREATER-THAN SIGN (>)
- // If the temporary buffer is the string "script", then switch to the script
- // data escaped state. Otherwise, switch to the script data double escaped
- // state. Emit the current input character as a character token.
- if (isSpace(cc) || cc === CC_SOLIDUS || cc === CC_GREATER_THAN) {
- state =
- scriptMatch === 6
- ? STATE_SCRIPT_DATA_ESCAPED
- : STATE_SCRIPT_DATA_DOUBLE_ESCAPED;
- pos++;
- } else if (isAsciiUpperAlpha(cc) || isAsciiLowerAlpha(cc)) {
- // ASCII alpha — advance the `"script"` match counter if the
- // lowercase form matches the next expected char, otherwise
- // snap to the sentinel so further chars can't revive a
- // match. No buffer allocation.
- const lower = isAsciiUpperAlpha(cc) ? cc + 0x20 : cc;
- if (scriptMatch < 6 && lower === "script".charCodeAt(scriptMatch)) {
- scriptMatch++;
- } else {
- scriptMatch = 7;
- }
- pos++;
- } else {
- // Anything else
- // Reconsume in the script data double escaped state.
- state = STATE_SCRIPT_DATA_DOUBLE_ESCAPED;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#plaintext-state
- case STATE_PLAINTEXT:
- // Consume the next input character:
- // U+0000 NULL is an unexpected-null-character parse error.
- // Anything else: emit the current input character.
- if (cc === CC_NULL) {
- reportError("unexpected-null-character", pos, pos + 1, "warning");
- pos++;
- } else {
- // Fast-forward to the next NULL (or EOF) in one native scan.
- pos++;
- const nul = input.indexOf("\0", pos);
- pos = nul === -1 ? len : nul;
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#character-reference-state
- case STATE_CHARACTER_REFERENCE:
- // Set the temporary buffer to the empty string. Append a U+0026
- // AMPERSAND (&) character to the temporary buffer.
- // `charRefStart` points at that `&` (one before the current pos).
- charRefStart = pos - 1;
- // Consume the next input character:
- if (isAsciiAlphanumeric(cc)) {
- // ASCII alphanumeric
- // Reconsume in the named character reference state.
- state = STATE_NAMED_CHARACTER_REFERENCE;
- // Reconsume
- } else if (cc === CC_NUMBER_SIGN) {
- // U+0023 NUMBER SIGN (#)
- // Append the current input character to the temporary buffer.
- // Set the character reference code to zero. Switch to the
- // numeric character reference state.
- charRefCode = 0;
- state = STATE_NUMERIC_CHARACTER_REFERENCE;
- pos++;
- } else {
- // Anything else
- // Flush code points consumed as a character reference.
- // Reconsume in the return state.
- state = returnState;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state
- case STATE_NAMED_CHARACTER_REFERENCE: {
- // Consume the maximum number of characters possible where the
- // consumed characters are one of the identifiers in the first
- // column of the named character references table.
- //
- // We measure the longest run of ASCII alphanumeric characters
- // (capped at MAX_ENTITY_NAME_LEN - 1 since the optional `;` is
- // handled separately), then walk that run from longest to
- // shortest looking for the first prefix that exists in the
- // entity table (with a trailing `;` if present, otherwise the
- // legacy bare form).
- let runLen = 0;
- while (
- pos + runLen < len &&
- isAsciiAlphanumeric(input.charCodeAt(pos + runLen)) &&
- runLen < MAX_ENTITY_NAME_LEN - 1
- ) {
- runLen++;
- }
- const hasSemicolon =
- pos + runLen < len && input.charCodeAt(pos + runLen) === CC_SEMICOLON;
- namedEntityConsumed = 0;
- let matchedWithSemicolon = false;
- // Try the full run with its trailing `;` first — the overwhelmingly
- // common case (`&`, ` `, …) then needs exactly one slice.
- if (hasSemicolon && runLen > 0) {
- const withSemicolon = input.slice(pos, pos + runLen + 1);
- if (HTML_ENTITIES[withSemicolon] !== undefined) {
- namedEntityConsumed = runLen + 1;
- matchedWithSemicolon = true;
- }
- }
- if (namedEntityConsumed === 0) {
- // Slice the candidate run once; prefixes come from this short
- // string instead of re-slicing the input per length.
- const run = input.slice(pos, pos + runLen);
- for (let n = runLen; n > 0; n--) {
- const bare = n === runLen ? run : run.slice(0, n);
- if (HTML_ENTITIES[bare] !== undefined) {
- namedEntityConsumed = n;
- break;
- }
- }
- }
- if (namedEntityConsumed > 0) {
- // A legacy match without a trailing `;` is a
- // missing-semicolon-after-character-reference parse error,
- // except for the spec's historical attribute rule: when
- // consumed in an attribute value and the next char is `=` or
- // ASCII alphanumeric, the reference is left undecoded silently.
- if (!matchedWithSemicolon) {
- const next = input.charCodeAt(pos + namedEntityConsumed);
- const inAttribute =
- returnState === STATE_ATTRIBUTE_VALUE_DOUBLE_QUOTED ||
- returnState === STATE_ATTRIBUTE_VALUE_SINGLE_QUOTED ||
- returnState === STATE_ATTRIBUTE_VALUE_UNQUOTED;
- if (!(
- inAttribute &&
- (next === CC_EQUALS || isAsciiAlphanumeric(next))
- )) {
- reportError(
- "missing-semicolon-after-character-reference",
- pos + namedEntityConsumed,
- pos + namedEntityConsumed + 1,
- "warning"
- );
- }
- }
- pos += namedEntityConsumed;
- state = returnState;
- } else {
- // No match — flush code points consumed as a character
- // reference. Switch to the ambiguous ampersand state.
- state = STATE_AMBIGUOUS_AMPERSAND;
- }
- break;
- }
- // https://html.spec.whatwg.org/multipage/parsing.html#ambiguous-ampersand-state
- case STATE_AMBIGUOUS_AMPERSAND:
- // Consume the next input character:
- if (isAsciiAlphanumeric(cc)) {
- // ASCII alphanumeric
- // If the character reference was consumed as part of an
- // attribute, then append the current input character to the
- // current attribute's value. Otherwise, emit the current
- // input character as a character token.
- pos++;
- } else if (cc === CC_SEMICOLON) {
- // U+003B SEMICOLON (;)
- // This is an unknown-named-character-reference parse error.
- // Reconsume in the return state.
- reportError(
- "unknown-named-character-reference",
- pos,
- pos + 1,
- "warning"
- );
- state = returnState;
- // Reconsume
- } else {
- // Anything else
- // Reconsume in the return state.
- state = returnState;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#numeric-character-reference-state
- case STATE_NUMERIC_CHARACTER_REFERENCE:
- // Set the character reference code to zero (0).
- // Consume the next input character:
- if (cc === 0x78 || cc === 0x58) {
- // U+0078 LATIN SMALL LETTER X
- // U+0058 LATIN CAPITAL LETTER X
- // Append the current input character to the temporary
- // buffer. Switch to the hexadecimal character reference
- // start state.
- state = STATE_HEXADECIMAL_CHARACTER_REFERENCE_START;
- pos++;
- } else {
- // Anything else
- // Reconsume in the decimal character reference start state.
- state = STATE_DECIMAL_CHARACTER_REFERENCE_START;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#hexadecimal-character-reference-start-state
- case STATE_HEXADECIMAL_CHARACTER_REFERENCE_START:
- // Consume the next input character:
- // ASCII hex digit: reconsume in the hexadecimal character reference state.
- // Anything else: absence-of-digits-in-numeric-character-reference parse
- // error. Flush code points consumed as a character reference. Reconsume
- // in the return state.
- if (isAsciiHexDigit(cc)) {
- state = STATE_HEXADECIMAL_CHARACTER_REFERENCE;
- } else {
- reportError(
- "absence-of-digits-in-numeric-character-reference",
- pos,
- pos + 1,
- "warning"
- );
- state = returnState;
- }
- // Reconsume
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#decimal-character-reference-start-state
- case STATE_DECIMAL_CHARACTER_REFERENCE_START:
- // Consume the next input character:
- // ASCII digit: reconsume in the decimal character reference state.
- // Anything else: absence-of-digits-in-numeric-character-reference parse
- // error. Flush code points consumed as a character reference. Reconsume
- // in the return state.
- if (isAsciiDigit(cc)) {
- state = STATE_DECIMAL_CHARACTER_REFERENCE;
- } else {
- reportError(
- "absence-of-digits-in-numeric-character-reference",
- pos,
- pos + 1,
- "warning"
- );
- state = returnState;
- }
- // Reconsume
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#hexadecimal-character-reference-state
- case STATE_HEXADECIMAL_CHARACTER_REFERENCE:
- // Consume the next input character:
- if (isAsciiHexDigit(cc)) {
- // ASCII digit / upper hex / lower hex
- // Multiply the character reference code by 16. Add a numeric
- // version of the current input character to the character
- // reference code. Stop accumulating once past the Unicode
- // range so the value can't overflow (still flags as
- // outside-range at the end).
- if (charRefCode < 0x110000) {
- const v = cc <= 0x39 ? cc - 0x30 : (cc | 0x20) - 0x61 + 10;
- charRefCode = charRefCode * 16 + v;
- }
- pos++;
- } else if (cc === CC_SEMICOLON) {
- // U+003B SEMICOLON
- // Switch to the numeric character reference end state.
- state = STATE_NUMERIC_CHARACTER_REFERENCE_END;
- pos++;
- } else {
- // Anything else
- // This is a missing-semicolon-after-character-reference
- // parse error. Reconsume in the numeric character reference
- // end state.
- reportError(
- "missing-semicolon-after-character-reference",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_NUMERIC_CHARACTER_REFERENCE_END;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#decimal-character-reference-state
- case STATE_DECIMAL_CHARACTER_REFERENCE:
- // Consume the next input character:
- if (isAsciiDigit(cc)) {
- // ASCII digit
- // Multiply the character reference code by 10. Add a numeric
- // version of the current input character (subtract 0x0030
- // from the character's code point) to the character reference
- // code. Stop accumulating once past the Unicode range so the
- // value can't overflow (still flags as outside-range).
- if (charRefCode < 0x110000) {
- charRefCode = charRefCode * 10 + (cc - 0x30);
- }
- pos++;
- } else if (cc === CC_SEMICOLON) {
- // U+003B SEMICOLON
- // Switch to the numeric character reference end state.
- state = STATE_NUMERIC_CHARACTER_REFERENCE_END;
- pos++;
- } else {
- // Anything else
- // This is a missing-semicolon-after-character-reference
- // parse error. Reconsume in the numeric character reference
- // end state.
- reportError(
- "missing-semicolon-after-character-reference",
- pos,
- pos + 1,
- "warning"
- );
- state = STATE_NUMERIC_CHARACTER_REFERENCE_END;
- // Reconsume
- }
- break;
- // https://html.spec.whatwg.org/multipage/parsing.html#numeric-character-reference-end-state
- case STATE_NUMERIC_CHARACTER_REFERENCE_END:
- // Check the character reference code and report the matching
- // WHATWG validation parse error.
- validateNumericReference(pos);
- // Flush code points consumed as a character reference.
- // Switch to the return state.
- state = returnState;
- // Reconsume
- break;
- /* istanbul ignore next -- @preserve: defensive fallback, all states are explicit above */
- default:
- pos++;
- }
- }
- // Handle EOF in non-data states per the WHATWG spec.
- //
- // Each in-progress comment / doctype / cdata / tag emits its partial
- // token range plus a corresponding `eof-in-X` parse error. Severity is
- // `"error"` because the emitted token offset range is incomplete (missing
- // trailing `-->`, `>`, `]]>`, etc.). For data / `<` / `</` / `<!`-only
- // inputs we emit `eof-before-tag-name` and fall through to flush the
- // pending text span (which still contains the lone `<`).
- // EOF inside a character-reference state: run the end-of-reference
- // processing (numeric end states never run when the reference ends at
- // EOF), then resume in the return state for the branches below.
- if (
- state >= STATE_CHARACTER_REFERENCE &&
- state <= STATE_NUMERIC_CHARACTER_REFERENCE_END
- ) {
- if (
- state === STATE_NUMERIC_CHARACTER_REFERENCE ||
- state === STATE_HEXADECIMAL_CHARACTER_REFERENCE_START ||
- state === STATE_DECIMAL_CHARACTER_REFERENCE_START
- ) {
- // No digits before EOF.
- reportError(
- "absence-of-digits-in-numeric-character-reference",
- len,
- len,
- "warning"
- );
- } else if (
- state === STATE_HEXADECIMAL_CHARACTER_REFERENCE ||
- state === STATE_DECIMAL_CHARACTER_REFERENCE
- ) {
- // Digits but no closing `;` before EOF.
- reportError(
- "missing-semicolon-after-character-reference",
- len,
- len,
- "warning"
- );
- validateNumericReference(len);
- } else if (state === STATE_NUMERIC_CHARACTER_REFERENCE_END) {
- validateNumericReference(len);
- }
- state = returnState;
- }
- if (
- (state >= STATE_TAG_NAME && state <= STATE_SELF_CLOSING_START_TAG) ||
- state === STATE_RCDATA_END_TAG_NAME ||
- state === STATE_RAWTEXT_END_TAG_NAME ||
- state === STATE_SCRIPT_DATA_END_TAG_NAME ||
- state === STATE_SCRIPT_DATA_ESCAPED_END_TAG_NAME
- ) {
- // EOF mid-tag — emit the partial open/close tag at EOF so the
- // consumer still sees the tag. This is a deliberate deviation
- // from the spec's per-character emission model: rather than
- // dropping the in-progress tag, we emit its offset range up to EOF.
- reportError("eof-in-tag", len, len, "error");
- // If we hit EOF mid-attribute-name, the name runs to EOF. Set
- // attributeNameEnd here so the emitted attribute range is valid.
- if (state === STATE_ATTRIBUTE_NAME && attributeNameStart !== -1) {
- attributeNameEnd = len;
- }
- if (attributeNameStart !== -1) emitAttribute(len);
- // If we hit EOF before the tag-name end was recorded, the name runs
- // to EOF. `tagNameEnd` may carry over from a previously emitted tag,
- // so reset it whenever it's missing or stale (less than `tagNameStart`)
- // — covers `<div` open-tag EOFs as well as `<title>x</tit` and other
- // content-mode end-tag-name EOFs.
- if (tagNameStart !== -1 && tagNameEnd < tagNameStart) {
- tagNameEnd = len;
- }
- flushText(tagStart);
- pos =
- input.charCodeAt(tagStart + 1) === CC_SOLIDUS
- ? emitCloseTag(len)
- : emitOpenTag(len, false);
- } else if (
- (state >= STATE_COMMENT_START && state <= STATE_BOGUS_COMMENT) ||
- (state >= STATE_COMMENT_LESS_THAN_SIGN &&
- state <= STATE_COMMENT_LESS_THAN_SIGN_BANG_DASH_DASH) ||
- state === STATE_MARKUP_DECLARATION_OPEN
- ) {
- // EOF in markup-declaration-open takes the spec's "anything else"
- // branch: incorrectly-opened-comment, then a bogus comment (which has
- // no EOF error of its own). Bogus comments at EOF are likewise normal.
- if (state === STATE_MARKUP_DECLARATION_OPEN) {
- reportError("incorrectly-opened-comment", commentStart, len, "warning");
- // The bogus comment this becomes has nothing left to consume.
- commentDataStart = len;
- commentDataEnd = len;
- } else if (state !== STATE_BOGUS_COMMENT) {
- reportError("eof-in-comment", len, len, "error");
- }
- if (callbacks.comment !== undefined) {
- pos = callbacks.comment(
- input,
- commentStart,
- len,
- commentDataStart,
- commentDataEnd
- );
- }
- } else if (state >= STATE_CDATA_SECTION && state <= STATE_CDATA_SECTION_END) {
- reportError("eof-in-cdata", len, len, "error");
- if (callbacks.comment !== undefined) {
- // No `]]>` closed it, so everything after `<![CDATA[` is content.
- pos = callbacks.comment(input, commentStart, len, commentDataStart, len);
- }
- } else if (state >= STATE_DOCTYPE && state <= STATE_BOGUS_DOCTYPE) {
- // EOF in bogus DOCTYPE emits the token with no parse error (spec).
- if (state !== STATE_BOGUS_DOCTYPE) {
- reportError("eof-in-doctype", len, len, "error");
- doctypeForceQuirks = true;
- }
- if (callbacks.doctype !== undefined) {
- pos = callbacks.doctype(
- input,
- commentStart,
- len,
- doctypeNameStart,
- doctypeNameEnd,
- doctypePublicStart,
- doctypePublicEnd,
- doctypeSystemStart,
- doctypeSystemEnd,
- doctypeForceQuirks
- );
- }
- } else {
- if (
- state === STATE_SCRIPT_DATA_ESCAPED ||
- state === STATE_SCRIPT_DATA_ESCAPED_DASH ||
- state === STATE_SCRIPT_DATA_ESCAPED_DASH_DASH ||
- state === STATE_SCRIPT_DATA_ESCAPED_LESS_THAN_SIGN ||
- state === STATE_SCRIPT_DATA_DOUBLE_ESCAPED ||
- state === STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH ||
- state === STATE_SCRIPT_DATA_DOUBLE_ESCAPED_DASH_DASH ||
- state === STATE_SCRIPT_DATA_DOUBLE_ESCAPED_LESS_THAN_SIGN ||
- state === STATE_SCRIPT_DATA_DOUBLE_ESCAPE_END
- ) {
- // Inside `<script><!-- … ` at EOF — spec calls this an
- // eof-in-script-html-comment-like-text parse error. The
- // less-than-sign and double-escape-end states reconsume back
- // into the (double-)escaped state on EOF per spec, which then
- // hits this same error.
- reportError("eof-in-script-html-comment-like-text", len, len, "error");
- } else if (state === STATE_TAG_OPEN || state === STATE_END_TAG_OPEN) {
- // `<` or `</` with nothing after; spec calls this
- // eof-before-tag-name. The lone `<` / `</` is preserved in the
- // pending text span which is flushed below.
- reportError("eof-before-tag-name", len, len, "warning");
- }
- if (textStart < len && callbacks.text !== undefined) {
- callbacks.text(input, textStart, len);
- }
- }
- return pos;
- };
- // WHATWG numeric-character-reference-end Windows-1252 remap table for the
- // 0x80-0x9F range. Per spec these C1 control code points decode to the
- // corresponding Windows-1252 glyph (with a parse error) rather than to the
- // raw C1 control character.
- const NUMERIC_C1_REMAP = {
- 0x80: "€",
- 0x82: "‚",
- 0x83: "ƒ",
- 0x84: "„",
- 0x85: "…",
- 0x86: "†",
- 0x87: "‡",
- 0x88: "ˆ",
- 0x89: "‰",
- 0x8a: "Š",
- 0x8b: "‹",
- 0x8c: "Œ",
- 0x8e: "Ž",
- 0x91: "‘",
- 0x92: "’",
- 0x93: "“",
- 0x94: "”",
- 0x95: "•",
- 0x96: "–",
- 0x97: "—",
- 0x98: "˜",
- 0x99: "™",
- 0x9a: "š",
- 0x9b: "›",
- 0x9c: "œ",
- 0x9e: "ž",
- 0x9f: "Ÿ"
- };
- /**
- * @param {number} code numeric character reference code point
- * @returns {string} decoded character per WHATWG remap rules
- */
- const decodeNumericReference = (code) => {
- // Per WHATWG numeric-character-reference-end-state:
- // - 0x00, > 0x10FFFF, or surrogate (0xD800-0xDFFF) -> U+FFFD.
- // - 0x80-0x9F -> Windows-1252 remap (above).
- // - Anything else (including noncharacters and C0 controls) -> the
- // code point itself; we don't surface the spec's parse-error
- // classes here since decoding is happening after the scanner ran.
- if (code === 0 || code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) {
- return "�";
- }
- if (code >= 0x80 && code <= 0x9f) {
- const remapped = /** @type {Record<number, string>} */ (NUMERIC_C1_REMAP)[
- code
- ];
- if (remapped !== undefined) return remapped;
- }
- return String.fromCodePoint(code);
- };
- /**
- * Decode a single matched character reference.
- * @param {string} match the matched reference text
- * @param {number} nextCharCode char code following the match in the source (NaN at the end)
- * @param {boolean=} isAttribute true when the match came from an attribute value
- * @returns {string} decoded text, or `match` itself when it stays literal
- */
- /**
- * Decode a named character reference spanning `[start, end)` of `str` (the
- * `&` at `start`, the optional `;` included). Slices the name once; WHATWG
- * longest-prefix and consumed-as-part-of-an-attribute semantics as before.
- * @param {string} str the raw string
- * @param {number} start offset of the reference's `&`
- * @param {number} end offset just past the reference (past the `;` if present)
- * @param {number} nextCharCode char code following the reference (NaN at the end)
- * @param {boolean=} isAttribute true when the reference came from an attribute value
- * @returns {string | undefined} decoded text, or `undefined` when it stays literal
- */
- const decodeNamedReference = (str, start, end, nextCharCode, isAttribute) => {
- // Mirrors the old `match.slice(1)`: the name keeps its trailing `;`.
- const name = str.slice(start + 1, end);
- const matchEndsWithSemi = name.charCodeAt(name.length - 1) === 0x3b;
- // Attribute-context guard: if the entity match didn't end with `;`
- // and the next character in the source is `=` or ASCII
- // alphanumeric, the WHATWG spec says to flush the literal text
- // rather than decode. The greedy scan already absorbed any
- // trailing alphanumerics, so the only candidate "next char" here
- // is `=` (or any non-alphanumeric).
- if (isAttribute && !matchEndsWithSemi && nextCharCode === 0x3d /* = */) {
- return undefined;
- }
- // Fast path: the scan usually captures exactly one entity (`&`,
- // `<`, ` `, ...), so the whole `name` is the match.
- if (name.length <= MAX_ENTITY_NAME_LEN) {
- const full = HTML_ENTITIES[name];
- if (full !== undefined) return full;
- }
- // Cap the longest-prefix search at MAX_ENTITY_NAME_LEN so pathological
- // inputs like `&` + thousands of alphanumerics stay linear-time.
- // Anything past that cap can't possibly match and is appended
- // verbatim as part of `name.slice(i)`. The full-length case was just
- // handled above, so start one shorter when it's the cap.
- const searchLen =
- name.length > MAX_ENTITY_NAME_LEN ? MAX_ENTITY_NAME_LEN : name.length - 1;
- for (let i = searchLen; i > 0; i--) {
- const prefix = name.slice(0, i);
- if (HTML_ENTITIES[prefix] !== undefined) {
- // Attribute-context longest-prefix guard: if the matched
- // prefix doesn't end with `;` and the leftover starts with
- // an alphanumeric character, leave literal per WHATWG.
- if (
- isAttribute &&
- i < name.length &&
- prefix.charCodeAt(prefix.length - 1) !== 0x3b
- ) {
- return undefined;
- }
- return HTML_ENTITIES[prefix] + name.slice(i);
- }
- }
- return undefined;
- };
- /** @typedef {{ text: string, map: number[] | undefined }} DecodedEntitiesWithMap */
- /**
- * Decode HTML character references in a string. Handles all numeric
- * references (with WHATWG remap of 0x00, surrogates, out-of-range, and the
- * C1 Windows-1252 table) and the full WHATWG named character references
- * table. Unknown or malformed references are left as literal text.
- *
- * When `isAttribute` is `true`, applies the WHATWG
- * "consumed-as-part-of-an-attribute" rule: a named reference without a
- * trailing `;` whose next character is `=` or ASCII alphanumeric is left
- * undecoded, so e.g. `&=foo` stays literal in an attribute value but
- * decodes to `&=foo` in text.
- *
- * With `withMap` the result also carries a boundary map from the decoded
- * string back to raw offsets, so spans computed on the decoded text (e.g.
- * srcset candidate URLs) can be translated to source ranges. `map[i]` is the
- * raw offset of decoded boundary `i` (`0..text.length`); boundaries inside a
- * reference's decoded text map to the reference start. `map` is `undefined`
- * when nothing was decoded (then `text === str`).
- * @overload
- * @param {string} str the raw string from the token slice
- * @param {boolean=} isAttribute true if `str` came from an attribute value
- * @returns {string} decoded string
- */
- /**
- * @overload
- * @param {string} str the raw string from the token slice
- * @param {boolean | undefined} isAttribute true if `str` came from an attribute value
- * @param {true} withMap also build the decoded-to-raw boundary map
- * @returns {DecodedEntitiesWithMap} decoded text and offset map
- */
- /**
- * @param {string} str the raw string from the token slice
- * @param {boolean=} isAttribute true if `str` came from an attribute value
- * @param {boolean=} withMap also build the decoded-to-raw boundary map
- * @returns {string | DecodedEntitiesWithMap} decoded string (with map when `withMap`)
- */
- const decodeEntities = (str, isAttribute, withMap) => {
- // Hand-rolled scan of one of three reference forms, each with an optional
- // trailing `;` — `&#x<hex>+` / `&#<dec>+` / `&<alpha><alnum>*` (kept separate
- // so `Ab` doesn't eat the `b` as hex): entity-dense text pays no regex
- // machinery or per-match object, and a string whose references all stay
- // literal is returned unchanged without rebuilding. Map bookkeeping is
- // gated on `withMap` at the rare per-decoded-reference points, so the
- // dominant no-map mode pays nothing for the unified implementation.
- let amp = str.indexOf("&");
- if (amp === -1) return withMap ? { text: str, map: undefined } : str;
- let out = "";
- let last = 0;
- /** @type {number[] | undefined} */
- let map;
- do {
- // `end` stays 0 when the `&` doesn't start a well-formed reference; a
- // numeric reference's code point accumulates during the digit scan
- // (saturating above the code-point range — every overflow decodes the
- // same replacement) so it needs no slice at all.
- let end = 0;
- let code = -1;
- let cc = str.charCodeAt(amp + 1);
- if (cc === 0x23 /* # */) {
- cc = str.charCodeAt(amp + 2);
- let p;
- if (cc === 0x78 /* x */ || cc === 0x58 /* X */) {
- code = 0;
- p = amp + 3;
- while (isAsciiHexDigit((cc = str.charCodeAt(p)))) {
- code = code * 16 + (cc <= 0x39 ? cc - 0x30 : (cc | 0x20) - 0x57);
- if (code > 0x10ffff) code = 0x110000;
- p++;
- }
- if (p > amp + 3) end = p;
- } else {
- code = 0;
- p = amp + 2;
- while (isAsciiDigit((cc = str.charCodeAt(p)))) {
- code = code * 10 + (cc - 0x30);
- if (code > 0x10ffff) code = 0x110000;
- p++;
- }
- if (p > amp + 2) end = p;
- }
- } else if (isAsciiAlpha(cc)) {
- code = -1;
- let p = amp + 2;
- while (isAsciiAlphanumeric(str.charCodeAt(p))) p++;
- end = p;
- }
- if (end !== 0) {
- if (str.charCodeAt(end) === 0x3b /* ; */) end++;
- // Numeric references always decode; named ones may stay literal.
- const decoded =
- code >= 0
- ? decodeNumericReference(code)
- : decodeNamedReference(
- str,
- amp,
- end,
- str.charCodeAt(end),
- isAttribute
- );
- if (decoded !== undefined) {
- if (withMap) map = _mapReference(map, last, amp, decoded.length);
- out += str.slice(last, amp);
- out += decoded;
- last = end;
- }
- amp = str.indexOf("&", end);
- } else {
- amp = str.indexOf("&", amp + 1);
- }
- } while (amp !== -1);
- if (last === 0) return withMap ? { text: str, map: undefined } : str;
- const text = out + str.slice(last);
- return withMap
- ? _finishMap(text, /** @type {number[]} */ (map), last, str.length)
- : text;
- };
- // Map-mode bookkeeping for `decodeEntities`, kept out of the scanner body so
- // the dominant no-map mode stays small enough to inline.
- /**
- * @param {number[] | undefined} map boundary map so far
- * @param {number} last raw offset the literal run starts at
- * @param {number} amp raw offset of the decoded reference
- * @param {number} decodedLength length of the decoded text
- * @returns {number[]} the map, with the literal run's and reference's boundaries appended
- */
- const _mapReference = (map, last, amp, decodedLength) => {
- if (map === undefined) map = [];
- for (let r = last; r < amp; r++) map.push(r);
- for (let i = 0; i < decodedLength; i++) map.push(amp);
- return map;
- };
- /**
- * @param {string} text decoded text
- * @param {number[]} map boundary map so far
- * @param {number} last raw offset of the trailing literal run
- * @param {number} rawLength raw string length
- * @returns {DecodedEntitiesWithMap} the finished result
- */
- const _finishMap = (text, map, last, rawLength) => {
- for (let r = last; r < rawLength; r++) map.push(r);
- map.push(rawLength);
- return { text, map };
- };
- /**
- * Escape a string per the WHATWG "escape a string" algorithm
- * (https://html.spec.whatwg.org/multipage/parsing.html#escapingString) in
- * attribute mode: `&`, U+00A0 and whichever quote delimits the value \u2014 the
- * other one stands literal, which is what makes `alt='say "hi"'` legal. CR/LF
- * are additionally encoded as numeric references so the result stays
- * single-line (usable in a `data:` URI). Single pass, returns the input
- * unchanged when nothing needs escaping.
- * @param {string} s string to escape
- * @param {number=} delimiter the quote the value is written in, `"` by default
- * @param {boolean=} minimal leave an `&` no reference can start after it bare
- * @returns {string} HTML attribute-safe string
- */
- const escapeAttribute = (s, delimiter, minimal) => {
- const quote = delimiter === CC_APOSTROPHE ? CC_APOSTROPHE : CC_QUOTATION_MARK;
- // fast path: native indexOf scans beat any per-character loop
- if (
- !s.includes("&") &&
- !s.includes(quote === CC_APOSTROPHE ? "'" : '"') &&
- !s.includes("\u00A0") &&
- !s.includes("\n") &&
- !s.includes("\r")
- ) {
- return s;
- }
- let out = "";
- let last = 0;
- for (let i = 0; i < s.length; i++) {
- let rep;
- switch (s.charCodeAt(i)) {
- case CC_AMPERSAND: {
- // §13.2.5.41: a `&` the parser cannot continue into a reference is
- // data. The minifier writes it bare; generated markup escapes it.
- const next = s.charCodeAt(i + 1);
- if (
- minimal === true &&
- next !== CC_NUMBER_SIGN &&
- !(
- (next >= 0x30 && next <= 0x39) ||
- (next >= 0x41 && next <= 0x5a) ||
- (next >= 0x61 && next <= 0x7a)
- )
- ) {
- continue;
- }
- rep = "&";
- break;
- }
- // Only the delimiter is escaped, so the compare costs a quote, not a
- // character: everything ordinary still falls straight through.
- case CC_QUOTATION_MARK:
- if (quote !== CC_QUOTATION_MARK) continue;
- rep = """;
- break;
- case CC_APOSTROPHE:
- if (quote !== CC_APOSTROPHE) continue;
- rep = "'";
- break;
- case CC_NO_BREAK_SPACE:
- rep = " ";
- break;
- case CC_LF:
- rep = " ";
- break;
- case CC_CR:
- rep = " ";
- break;
- default:
- continue;
- }
- out += s.slice(last, i) + rep;
- last = i + 1;
- }
- return out + s.slice(last);
- };
- /**
- * Escape a string per the WHATWG "escape a string" algorithm in text mode:
- * `&`, U+00A0, `<` and `>`. CR/LF are additionally encoded as numeric
- * references so the result stays single-line (usable in a `data:` URI).
- * Single pass, returns the input unchanged when nothing needs escaping.
- * @param {string} s string to escape
- * @returns {string} HTML text-content-safe string
- */
- const escapeText = (s) => {
- // fast path: native indexOf scans beat any per-character loop
- if (
- !s.includes("&") &&
- !s.includes("<") &&
- !s.includes(">") &&
- !s.includes("\u00A0") &&
- !s.includes("\n") &&
- !s.includes("\r")
- ) {
- return s;
- }
- let out = "";
- let last = 0;
- for (let i = 0; i < s.length; i++) {
- let rep;
- switch (s.charCodeAt(i)) {
- case CC_AMPERSAND:
- rep = "&";
- break;
- case CC_LESS_THAN:
- rep = "<";
- break;
- case CC_GREATER_THAN:
- rep = ">";
- break;
- case CC_NO_BREAK_SPACE:
- rep = " ";
- break;
- case CC_LF:
- rep = " ";
- break;
- case CC_CR:
- rep = " ";
- break;
- default:
- continue;
- }
- out += s.slice(last, i) + rep;
- last = i + 1;
- }
- return out + s.slice(last);
- };
- /**
- * @param {string} name meta name
- * @param {string} content meta content
- * @returns {string} meta tag
- */
- const metaTag = (name, content) => {
- // og: uses property=; all others (including twitter:) use name=
- const attr = name.startsWith("og:")
- ? `property="${escapeAttribute(name)}"`
- : `name="${escapeAttribute(name)}"`;
- return `<meta ${attr} content="${escapeAttribute(content)}">`;
- };
- /**
- * @param {string | { href: string, target?: string }} base base option
- * @returns {string} base tag
- */
- const baseTag = (base) => {
- const href = typeof base === "string" ? base : base.href;
- const targetAttr =
- typeof base === "object" && base.target
- ? ` target="${escapeAttribute(base.target)}"`
- : "";
- return `<base href="${escapeAttribute(href)}"${targetAttr}>`;
- };
- /**
- * Serializes head tags in spec order — charset, base, meta, title — for a page
- * built from scratch; an authored page merges the same options in `parse`.
- * @param {OutputHtmlOptions} opts html options
- * @returns {string} head tags string
- */
- const buildHeadTags = (opts) => {
- let out = "";
- const meta = opts.meta;
- if (meta && meta.charset) {
- out += `<meta charset="${escapeAttribute(meta.charset)}">`;
- }
- if (opts.base) out += baseTag(opts.base);
- if (meta) {
- for (const [name, content] of Object.entries(meta)) {
- if (name === "charset") continue;
- out += metaTag(name, content);
- }
- }
- if (opts.title) {
- out += `<title>${escapeText(opts.title)}</title>`;
- }
- return out;
- };
- // cspell:ignore definitionurl malignmark mglyph selectedcontent megamorphic attributeless rowspan imagesizes novalidate maxlength
- // WHATWG HTML tree construction (https://html.spec.whatwg.org/multipage/parsing.html#tree-construction)
- // on top of tokenize. Scripting is always disabled (webpack is a build tool).
- // Namespaces (mirrors swc_html_ast::Namespace)
- const NS_HTML = 0;
- const NS_MATHML = 1;
- const NS_SVG = 2;
- /**
- * AST node `type` discriminators. Numeric for the same reason as the CSS
- * `NodeType`: compact integer `===` dispatch on the tree-construction and
- * visitor-walk hot paths.
- * @type {{ Document: 1, DocumentFragment: 2, Element: 3, Text: 4, Comment: 5, Doctype: 6, ProcessingInstruction: 7 }}
- */
- const NodeType = {
- Document: 1,
- DocumentFragment: 2,
- Element: 3,
- Text: 4,
- Comment: 5,
- Doctype: 6,
- ProcessingInstruction: 7
- };
- /**
- * A contiguous run of attribute ids in the attribute columns — how a start-tag
- * token and an element refer to their attributes. `start` is the first id
- * (`count` 0 = none).
- * @typedef {{ start: number, count: number }} AttributeRun
- */
- // Shared frozen empty run for attributeless elements and synthesized tags.
- const EMPTY_ATTRS = /** @type {AttributeRun} */ (
- Object.freeze({ start: 0, count: 0 })
- );
- // === Struct-of-arrays AST backend ===
- // One AST node = one integer id (`HtmlNodeRef`) indexing the parallel columns
- // below — no per-node object and no per-parent children array. Tree shape
- // lives in the four link columns (parent / firstChild / lastChild /
- // nextSibling); the only heap references are the string payload / attribute
- // name-and-value side arrays. Columns are module-level and reused
- // across parses (grown, shrunk on release only past `_COLUMN_SHRINK_CAPACITY`,
- // mirroring the CSS parser's column backend), so a steady-state parse
- // allocates almost nothing per node; consumers must fully read a tree before
- // the next `parseHtml` call. Id 0 is reserved as "no node" so the link columns
- // can use 0 as null.
- let _nodeCapacity = 0;
- let _nodeCount = 0;
- // Highest id `_nodeCount` ever reached this parse. Streaming rewinds the
- // allocator, so `_nodeCount` alone no longer bounds the ids that were written —
- // the release pass has to clear up to here or the recycled slots keep their
- // strings alive until a later parse happens to overwrite them.
- let _nodeHighWater = 0;
- /** `NodeType` per node */
- let _nodeTypes = new Uint8Array(0);
- /** bits 0-1 namespace (`NS_*`), bit 2 self-closing (void), bit 3 template content */
- let _nodeFlags = new Uint8Array(0);
- let _nodeStarts = new Int32Array(0);
- let _nodeEnds = new Int32Array(0);
- /** end offset of an element's opening tag (after `>`) */
- let _nodeTagEnds = new Int32Array(0);
- /** end offset of an element's tag name */
- let _nodeNameEnds = new Int32Array(0);
- /**
- * under `skip.text`, a raw-text element's body-end offset (`HtmlAstSkip`); for
- * a `<template>` (`FLAG_HAS_TEMPLATE`) it instead holds the content-fragment ref
- */
- let _nodeContentEnds = new Int32Array(0);
- let _nodeParents = new Int32Array(0);
- let _nodeFirstChildren = new Int32Array(0);
- let _nodeLastChildren = new Int32Array(0);
- let _nodeNextSiblings = new Int32Array(0);
- /** @type {string[]} tag name / text data / comment data / doctype name */
- const _nodeStrings = [];
- /** first attribute id of an element's contiguous run */
- let _nodeAttrStarts = new Int32Array(0);
- /** attribute count of an element's run */
- let _nodeAttrCounts = new Int32Array(0);
- // The single doctype node's public/system ids (a document inserts at most one
- // doctype node — later doctype tokens are ignored — so no column is needed).
- /** @type {string | null} */
- let _doctypePublicId = null;
- /** @type {string | null} */
- let _doctypeSystemId = null;
- const NS_MASK = 3;
- const FLAG_SELF_CLOSING = 4;
- // A `<template>` element: `_nodeContentEnds` holds its content-fragment ref
- // instead of a raw-text body-end offset (the two never coexist on one node).
- const FLAG_HAS_TEMPLATE = 8;
- // A `<template shadowrootmode>` kept where it was written, and its host: the
- // engine takes it out of the tree, so nothing may move it away from that host.
- const FLAG_SHADOW_ROOT = 16;
- const FLAG_SHADOW_HOST = 32;
- // The element whose content prints from source, and everything still open around
- // it. A flag, not a lookup: the walk asks this of every node.
- const FLAG_FOSTER_REGION = 128;
- /**
- * @param {HtmlNodeRef} el an element
- * @returns {boolean} whether a shadow root can be attached to it
- */
- const _isShadowHost = (el) => {
- if (_namespaceOf(el) !== NS_HTML) return false;
- const tagName = _tagNameOf(el);
- if (SHADOW_HOSTS.has(tagName)) return true;
- return (
- tagName.includes("-") &&
- tagName.charCodeAt(0) >= 97 &&
- tagName.charCodeAt(0) <= 122 &&
- !RESERVED_ELEMENT_NAMES.has(tagName)
- );
- };
- /** @type {(el: HtmlNodeRef) => string} */
- const _tagNameOf = (el) => _nodeStrings[el];
- // A `<template>`'s content-fragment ref (in the shared `_nodeContentEnds` slot),
- // or 0 when `el` is not a template — the tree builder's spec "template contents".
- /** @type {(el: HtmlNodeRef) => HtmlDocumentFragment} */
- const _templateContentOf = (el) =>
- (_nodeFlags[el] & FLAG_HAS_TEMPLATE) !== 0 ? _nodeContentEnds[el] : 0;
- /** @type {(el: HtmlNodeRef) => number} */
- const _namespaceOf = (el) => _nodeFlags[el] & NS_MASK;
- // === Attribute columns ===
- // One attribute = one integer id into these columns; an element (and a
- // start-tag token) holds a contiguous run. The value string is derived from
- // the source by offset on read — `_attrValues` carries an override only for
- // valueless attributes (`""`) and offset-less adoption-agency clones — and the
- // html5lib serializer name is derived from the adjusted name plus one flag
- // bit, so per attribute only the interned name pointer is retained.
- let _attrCapacity = 0;
- let _attrCount = 0;
- // Elements a repeated `<html>` / `<body>` start tag merged attributes onto
- // (§13.2.6.4.4, §13.2.6.4.7). The merged names are not in the element's source
- // open tag, so the printer rebuilds that tag instead of echoing it.
- const _mergedAttrNodes = new Set();
- // A processing instruction's target; its data rides in `_nodeStrings` like a
- // comment's. A map, since a second string column per node would cost more.
- /** @type {Map<HtmlNodeRef, string>} */
- const _piTargets = new Map();
- let _attrNameStarts = new Int32Array(0);
- let _attrNameEnds = new Int32Array(0);
- let _attrValueStarts = new Int32Array(0);
- let _attrValueEnds = new Int32Array(0);
- /** bit 0: name has a `FOREIGN_ATTR_NS` serializer name (set on foreign adjust) */
- let _attrFlags = new Uint8Array(0);
- /** @type {string[]} lowercased (foreign-content: adjusted) attribute name */
- const _attrNames = [];
- /** @type {(string | null)[]} value override (null = slice the source by offset) */
- const _attrValues = [];
- /** source of the current parse, for by-offset attribute values */
- let _htmlSource = "";
- /**
- * @param {number} need minimum capacity
- * @param {boolean=} exact allocate `need` directly (initial pre-size) instead of doubling
- */
- const _growAttrColumns = (need, exact) => {
- let cap = _attrCapacity || 4096;
- if (exact) {
- if (need > cap) cap = need;
- } else {
- while (cap < need) cap *= 2;
- // A capacity over the shrink threshold is released after the parse, so
- // doubling just past the pre-size costs the next one its columns.
- if (cap > _COLUMN_SHRINK_CAPACITY && need <= _COLUMN_SHRINK_CAPACITY) {
- cap = _COLUMN_SHRINK_CAPACITY;
- }
- }
- const nameStart = new Int32Array(cap);
- nameStart.set(_attrNameStarts);
- _attrNameStarts = nameStart;
- const nameEnd = new Int32Array(cap);
- nameEnd.set(_attrNameEnds);
- _attrNameEnds = nameEnd;
- const valStart = new Int32Array(cap);
- valStart.set(_attrValueStarts);
- _attrValueStarts = valStart;
- const valEnd = new Int32Array(cap);
- valEnd.set(_attrValueEnds);
- _attrValueEnds = valEnd;
- const fl = new Uint8Array(cap);
- fl.set(_attrFlags);
- _attrFlags = fl;
- // Keep the string columns filled to capacity (see `_growNodeColumns`).
- for (let i = _attrNames.length; i < cap; i++) {
- _attrNames.push("");
- _attrValues.push(null);
- }
- _attrCapacity = cap;
- };
- /** @type {(name: string, value: string | null, nameStart: number, nameEnd: number, valueStart: number, valueEnd: number) => number} */
- const _allocAttr = (name, value, nameStart, nameEnd, valueStart, valueEnd) => {
- const i = ++_attrCount;
- if (i >= _attrCapacity) _growAttrColumns(i + 1);
- _attrNameStarts[i] = nameStart;
- _attrNameEnds[i] = nameEnd;
- _attrValueStarts[i] = valueStart;
- _attrValueEnds[i] = valueEnd;
- _attrFlags[i] = 0;
- // Ids are sequential, so these indexed writes append (arrays stay packed).
- _attrNames[i] = name;
- _attrValues[i] = value;
- return i;
- };
- /** @type {(i: number) => string} */
- const _attrValueOf = (i) => {
- const v = _attrValues[i];
- return v !== null
- ? v
- : _htmlSource.slice(_attrValueStarts[i], _attrValueEnds[i]);
- };
- // Linear name lookup in a run — attribute lists are short, a loop beats a Map.
- /** @type {(start: number, count: number, name: string) => number} */
- const _findAttr = (start, count, name) => {
- for (let i = start; i < start + count; i++) {
- if (_attrNames[i] === name) return i;
- }
- return 0;
- };
- /** @type {(i: number) => number} exact copy of an attribute into a new id */
- const _copyAttr = (i) => {
- const c = _allocAttr(
- _attrNames[i],
- _attrValues[i],
- _attrNameStarts[i],
- _attrNameEnds[i],
- _attrValueStarts[i],
- _attrValueEnds[i]
- );
- _attrFlags[c] = _attrFlags[i];
- return c;
- };
- // html5lib serializer name, derived: a `FOREIGN_ATTR_NS`-adjusted attribute is
- // flagged (its name is the table key), and a camelCase-adjusted name contains
- // an uppercase letter (unadjusted names are always lowercased), serializing as
- // itself. Everything else serializes as the plain name (undefined here).
- /** @type {(i: number) => string | undefined} */
- const _attrSerializedName = (i) => {
- if ((_attrFlags[i] & 1) !== 0) return FOREIGN_ATTR_NS[_attrNames[i]];
- const name = _attrNames[i];
- return /[A-Z]/.test(name) ? name : undefined;
- };
- /**
- * @param {number} need minimum capacity
- * @param {boolean=} exact allocate `need` directly (initial pre-size) instead of doubling
- */
- const _growNodeColumns = (need, exact) => {
- let cap = _nodeCapacity || 4096;
- // `exact` (the initial pre-size): allocate the estimate directly. Doubling
- // from 4096 would overshoot it to the next power of two (~2x) and waste that
- // capacity for the process lifetime. Incremental growth still doubles.
- if (exact) {
- if (need > cap) cap = need;
- } else {
- while (cap < need) cap *= 2;
- // A capacity over the shrink threshold is released after the parse, so
- // doubling just past the pre-size costs the next one its columns.
- if (cap > _COLUMN_SHRINK_CAPACITY && need <= _COLUMN_SHRINK_CAPACITY) {
- cap = _COLUMN_SHRINK_CAPACITY;
- }
- }
- const ty = new Uint8Array(cap);
- ty.set(_nodeTypes);
- _nodeTypes = ty;
- const fl = new Uint8Array(cap);
- fl.set(_nodeFlags);
- _nodeFlags = fl;
- const st = new Int32Array(cap);
- st.set(_nodeStarts);
- _nodeStarts = st;
- const en = new Int32Array(cap);
- en.set(_nodeEnds);
- _nodeEnds = en;
- const tagEnd = new Int32Array(cap);
- tagEnd.set(_nodeTagEnds);
- _nodeTagEnds = tagEnd;
- const nameEnd = new Int32Array(cap);
- nameEnd.set(_nodeNameEnds);
- _nodeNameEnds = nameEnd;
- const cEnd = new Int32Array(cap);
- cEnd.set(_nodeContentEnds);
- _nodeContentEnds = cEnd;
- const parent = new Int32Array(cap);
- parent.set(_nodeParents);
- _nodeParents = parent;
- const first = new Int32Array(cap);
- first.set(_nodeFirstChildren);
- _nodeFirstChildren = first;
- const last = new Int32Array(cap);
- last.set(_nodeLastChildren);
- _nodeLastChildren = last;
- const next = new Int32Array(cap);
- next.set(_nodeNextSiblings);
- _nodeNextSiblings = next;
- const aStart = new Int32Array(cap);
- aStart.set(_nodeAttrStarts);
- _nodeAttrStarts = aStart;
- const aCount = new Int32Array(cap);
- aCount.set(_nodeAttrCounts);
- _nodeAttrCounts = aCount;
- // Keep the string column filled to capacity so per-node writes are in-place
- // packed stores, not length-growing appends re-paying the regrowth cascade.
- for (let i = _nodeStrings.length; i < cap; i++) _nodeStrings.push("");
- _nodeCapacity = cap;
- };
- /** Start a new parse: invalidate all prior refs, release prior heap refs. */
- const _resetAstColumns = () => {
- // Overwrite the used prefix (releases the prior parse's strings) instead of
- // truncating: a `length = 0` would right-size the backing store and make
- // every write of the next parse a length-growing append again.
- const written = _nodeCount > _nodeHighWater ? _nodeCount : _nodeHighWater;
- const usedNodes = Math.min(written, _nodeStrings.length - 1);
- for (let i = 0; i <= usedNodes; i++) _nodeStrings[i] = "";
- const usedAttrs = Math.min(_attrCount, _attrNames.length - 1);
- for (let i = 0; i <= usedAttrs; i++) {
- _attrNames[i] = "";
- _attrValues[i] = null;
- }
- _nodeCount = 0;
- _nodeHighWater = 0;
- _attrCount = 0;
- _mergedAttrNodes.clear();
- _piTargets.clear();
- _doctypePublicId = null;
- _doctypeSystemId = null;
- };
- // The typed-array columns grow to the largest document ever parsed; above this
- // capacity they are re-shrunk on release so one pathological file can't pin
- // tens of MB at module level for the process lifetime (~46 bytes/node,
- // ~17 bytes/attribute of capacity across the columns).
- const _COLUMN_SHRINK_CAPACITY = 65536;
- // Release the side arrays' heap references (strings, attribute names/values)
- // once a walk has consumed the tree, so the retained columns don't pin the
- // parsed source until the next parse.
- const _releaseAstColumns = () => {
- // Kept regime: overwrite the used prefix in place (see `_resetAstColumns`);
- // oversized arrays are dropped entirely, mirroring the typed columns below.
- if (_nodeStrings.length > _COLUMN_SHRINK_CAPACITY) {
- _nodeStrings.length = 0;
- } else {
- const written = _nodeCount > _nodeHighWater ? _nodeCount : _nodeHighWater;
- const usedNodes = Math.min(written, _nodeStrings.length - 1);
- for (let i = 0; i <= usedNodes; i++) _nodeStrings[i] = "";
- }
- if (_attrNames.length > _COLUMN_SHRINK_CAPACITY) {
- _attrNames.length = 0;
- _attrValues.length = 0;
- } else {
- const usedAttrs = Math.min(_attrCount, _attrNames.length - 1);
- for (let i = 0; i <= usedAttrs; i++) {
- _attrNames[i] = "";
- _attrValues[i] = null;
- }
- }
- // The printer's sort buffers grew to the widest tag and longest token list
- // this document held; dropping them keeps one page off the module forever.
- _sortOrderScratch.length = 0;
- _sortKeyScratch.length = 0;
- _tokenSpanStarts.length = 0;
- _tokenSpanEnds.length = 0;
- _nodeCount = 0;
- _nodeHighWater = 0;
- _attrCount = 0;
- _mergedAttrNodes.clear();
- _piTargets.clear();
- _htmlSource = "";
- _doctypePublicId = null;
- _doctypeSystemId = null;
- if (_nodeCapacity > _COLUMN_SHRINK_CAPACITY) {
- _nodeCapacity = 0;
- _nodeTypes = new Uint8Array(0);
- _nodeFlags = new Uint8Array(0);
- _nodeStarts = new Int32Array(0);
- _nodeEnds = new Int32Array(0);
- _nodeTagEnds = new Int32Array(0);
- _nodeNameEnds = new Int32Array(0);
- _nodeContentEnds = new Int32Array(0);
- _nodeParents = new Int32Array(0);
- _nodeFirstChildren = new Int32Array(0);
- _nodeLastChildren = new Int32Array(0);
- _nodeNextSiblings = new Int32Array(0);
- _nodeAttrStarts = new Int32Array(0);
- _nodeAttrCounts = new Int32Array(0);
- }
- if (_attrCapacity > _COLUMN_SHRINK_CAPACITY) {
- _attrCapacity = 0;
- _attrNameStarts = new Int32Array(0);
- _attrNameEnds = new Int32Array(0);
- _attrValueStarts = new Int32Array(0);
- _attrValueEnds = new Int32Array(0);
- _attrFlags = new Uint8Array(0);
- }
- };
- /** @type {(type: number, start: number, end: number) => HtmlNodeRef} */
- // The element-only columns (tagEnds/nameEnds/contentEnds/attrStarts/
- // attrCounts) are left stale here — they are only ever read for Element
- // nodes, and every Element writes them in mkEl/cloneSubtree.
- const _allocNode = (type, start, end) => {
- const i = ++_nodeCount;
- if (i >= _nodeCapacity) _growNodeColumns(i + 1);
- _nodeTypes[i] = type;
- _nodeFlags[i] = 0;
- _nodeStarts[i] = start;
- _nodeEnds[i] = end;
- _nodeParents[i] = 0;
- _nodeFirstChildren[i] = 0;
- _nodeLastChildren[i] = 0;
- _nodeNextSiblings[i] = 0;
- // `_nodeStrings[i]` is written by every caller right after this returns
- // (tag name / text / comment data; "" at the few string-less sites), as a
- // packed in-place store — no placeholder double-write here.
- return i;
- };
- // Raw child append — no text merging, no `<template>` content redirect (the
- // tree builder layers those on top). `node` must be detached (`next` = 0).
- /** @type {(parent: HtmlNodeRef, node: HtmlNodeRef) => void} */
- const _appendChild = (parent, node) => {
- _nodeParents[node] = parent;
- const last = _nodeLastChildren[parent];
- if (last === 0) _nodeFirstChildren[parent] = node;
- else _nodeNextSiblings[last] = node;
- _nodeLastChildren[parent] = node;
- };
- /** @type {(data: string, start: number, end: number) => HtmlNodeRef} */
- const _makeTextNode = (data, start, end) => {
- const i = _allocNode(NodeType.Text, start, end);
- _nodeStrings[i] = data;
- return i;
- };
- /** @type {(data: string, start: number, end: number) => HtmlNodeRef} */
- const _makeCommentNode = (data, start, end) => {
- const i = _allocNode(NodeType.Comment, start, end);
- _nodeStrings[i] = data;
- return i;
- };
- /** @type {(target: string, data: string, start: number, end: number) => HtmlNodeRef} */
- const _makeProcessingInstructionNode = (target, data, start, end) => {
- const i = _allocNode(NodeType.ProcessingInstruction, start, end);
- _nodeStrings[i] = data;
- _piTargets.set(i, target);
- return i;
- };
- // Marker entry in the active-formatting-elements list (never a valid ref).
- const AFE_MARKER = -1;
- // Clone of an element's attribute run: keep name/value (and the serializer
- // name flag) but drop source offsets so the consumer doesn't emit a duplicate
- // dependency for the reopened element's spans. Values are materialized since
- // the offsets are gone.
- const cloneAttrs = (/** @type {HtmlElement} */ el) => {
- const start = _nodeAttrStarts[el];
- const count = _nodeAttrCounts[el];
- const newStart = _attrCount + 1;
- for (let i = start; i < start + count; i++) {
- const c = _allocAttr(_attrNames[i], _attrValueOf(i), -1, -1, -1, -1);
- _attrFlags[c] = _attrFlags[i];
- }
- return { start: newStart, count };
- };
- // Merge a repeated `<html>`/`<body>` tag's attributes into the element: only
- // names not already present are added (in source order after the existing
- // ones). Runs are contiguous, so any addition re-allocates the whole run; the
- // old slots are orphaned (at most once per repeated tag, rare).
- const mergeAttrs = (
- /** @type {HtmlElement} */ el,
- /** @type {AttributeRun} */ run
- ) => {
- const start = _nodeAttrStarts[el];
- const count = _nodeAttrCounts[el];
- let extra = 0;
- for (let i = run.start; i < run.start + run.count; i++) {
- if (_findAttr(start, count, _attrNames[i]) === 0) extra++;
- }
- if (extra === 0) return;
- const newStart = _attrCount + 1;
- for (let i = start; i < start + count; i++) _copyAttr(i);
- for (let i = run.start; i < run.start + run.count; i++) {
- if (_findAttr(start, count, _attrNames[i]) === 0) _copyAttr(i);
- }
- _nodeAttrStarts[el] = newStart;
- _nodeAttrCounts[el] = count + extra;
- _mergedAttrNodes.add(el);
- };
- /**
- * A materialized attribute as returned by `A.attributes` (tests/tooling) —
- * the parser-facing representation is an id into the attribute columns, read
- * through the scalar `A.attr*` accessors.
- * @typedef {object} HtmlAttribute
- * @property {string} name lowercased (and, in foreign content, adjusted) attribute name
- * @property {string} value
- * @property {string=} serializedName name used by the html5lib tree serializer (foreign-namespaced)
- * @property {number} nameStart source offset, or -1 on adoption-agency clones
- * @property {number} nameEnd
- * @property {number} valueStart source offset, or -1 when valueless / on clones
- * @property {number} valueEnd
- */
- /**
- * A node reference into the struct-of-arrays AST: an integer id indexing the
- * parallel `_h*` columns. Read fields through the exported accessor `A`. Refs
- * are only valid until the next `parseHtml` call — the columns are reused
- * across parses — so consume a tree fully before parsing again.
- * @typedef {number} HtmlNodeRef
- */
- /** @typedef {HtmlNodeRef} HtmlElement ref to an Element node */
- /** @typedef {HtmlNodeRef} HtmlText ref to a Text node */
- /** @typedef {HtmlNodeRef} HtmlComment ref to a Comment node */
- /** @typedef {HtmlNodeRef} HtmlDoctype ref to a Doctype node */
- /** @typedef {HtmlNodeRef} HtmlProcessingInstruction ref to a ProcessingInstruction node */
- /** @typedef {HtmlNodeRef} HtmlDocument ref to the Document node */
- /** @typedef {HtmlNodeRef} HtmlDocumentFragment ref to a DocumentFragment node */
- /** @typedef {HtmlNodeRef} HtmlNode */
- /**
- * An attribute reference: an integer id into the attribute columns, read
- * through the `A.attr*` accessors. Same validity contract as `HtmlNodeRef`.
- * @typedef {number} HtmlAttributeRef
- */
- /** @typedef {{ start: number, end: number, tagEnd: number, nameEnd: number }} TagPos */
- // Tree-construction token `type` discriminators. Numeric for the same reason
- // as `NodeType` / the CSS `TT_*` constants: the insertion modes dispatch on
- // `t.type` per token, and integer `===` beats string comparison there.
- const TOKEN_CHAR = 1;
- const TOKEN_COMMENT = 2;
- const TOKEN_DOCTYPE = 3;
- const TOKEN_START_TAG = 4;
- const TOKEN_END_TAG = 5;
- const TOKEN_EOF = 6;
- const TOKEN_PROCESSING_INSTRUCTION = 7;
- /** @typedef {{ type: typeof TOKEN_CHAR, data: string, start: number, end: number }} CharToken */
- /** @typedef {{ type: typeof TOKEN_COMMENT, data: string, start: number, end: number }} CommentToken */
- /** @typedef {{ type: typeof TOKEN_DOCTYPE, name: string, publicId: (string | null), systemId: (string | null), forceQuirks: boolean, start: number, end: number }} DoctypeToken */
- /** @typedef {{ type: typeof TOKEN_START_TAG, name: string, attrs: AttributeRun, selfClosing: boolean, pos: TagPos, swallowNewline?: boolean }} StartTagToken */
- /** @typedef {{ type: typeof TOKEN_END_TAG, name: string, pos: TagPos }} EndTagToken */
- /** @typedef {{ type: typeof TOKEN_EOF }} EofToken */
- /** @typedef {{ type: typeof TOKEN_PROCESSING_INSTRUCTION, name: string, data: string, start: number, end: number }} ProcessingInstructionToken the target rides in `name`, as the doctype's does */
- // Every insertion mode handles a processing instruction token on the same arc
- // as a comment token, so the arcs test both.
- const isCommentOrProcessingInstruction = (/** @type {Token} */ t) =>
- t.type === TOKEN_COMMENT || t.type === TOKEN_PROCESSING_INSTRUCTION;
- /**
- * Internal token passed through the tree-construction insertion modes.
- * @typedef {CharToken | CommentToken | DoctypeToken | StartTagToken | EndTagToken | EofToken | ProcessingInstructionToken} Token
- */
- /**
- * The tree builder reuses a single mutable token (with a reused `pos`) instead
- * of allocating one object per tokenizer callback. All fields are always
- * present so the shape never changes — keeping the `process`/insertion-mode
- * `t.*` reads monomorphic — and fields irrelevant to the current `type` carry
- * stale values that those handlers never read. Tokens that must outlive the
- * current callback (buffered table characters, synthesized re-dispatches) are
- * copied into fresh plain objects instead.
- * @typedef {{ type: number, name: string, data: string, attrs: AttributeRun, selfClosing: boolean, start: number, end: number, publicId: (string | null), systemId: (string | null), forceQuirks: boolean, swallowNewline: boolean, pos: TagPos }} MutableToken
- */
- /** @typedef {{ parent: HtmlNodeRef, beforeNode: HtmlNodeRef }} InsertionPlace `beforeNode` 0 = plain append */
- // Insertion modes (§13.2.4.1). Numeric for the same reason as the token and
- // `NodeType` enums: `runMode` dispatches on `mode` once per token.
- const MODE_INITIAL = 1;
- const MODE_BEFORE_HTML = 2;
- const MODE_BEFORE_HEAD = 3;
- const MODE_IN_HEAD = 4;
- const MODE_IN_HEAD_NOSCRIPT = 5;
- const MODE_AFTER_HEAD = 6;
- const MODE_IN_BODY = 7;
- const MODE_TEXT = 8;
- const MODE_IN_TABLE = 9;
- const MODE_IN_TABLE_TEXT = 10;
- const MODE_IN_CAPTION = 11;
- const MODE_IN_COLUMN_GROUP = 12;
- const MODE_IN_TABLE_BODY = 13;
- const MODE_IN_ROW = 14;
- const MODE_IN_CELL = 15;
- const MODE_IN_TEMPLATE = 16;
- const MODE_AFTER_BODY = 17;
- const MODE_IN_FRAMESET = 18;
- const MODE_AFTER_FRAMESET = 19;
- const MODE_AFTER_AFTER_BODY = 20;
- const MODE_AFTER_AFTER_FRAMESET = 21;
- // The §13.2.6.4.18 re-dispatch table, bound to this file's numbering. A name the
- // binding does not cover would silently dispatch to `undefined`, so it throws.
- /** @type {Record<string, number>} */
- const MODE_BY_NAME = {
- IN_TABLE: MODE_IN_TABLE,
- IN_COLUMN_GROUP: MODE_IN_COLUMN_GROUP,
- IN_TABLE_BODY: MODE_IN_TABLE_BODY,
- IN_ROW: MODE_IN_ROW
- };
- const TEMPLATE_START_TAG_MODES = new Map(
- Object.entries(TEMPLATE_START_TAG_MODE_NAMES).map(([tag, name]) => {
- const mode = MODE_BY_NAME[name];
- if (mode === undefined) throw new Error(`Unknown insertion mode ${name}`);
- return [tag, mode];
- })
- );
- // MathML/SVG specials handled via namespace checks below.
- const isSpecial = (/** @type {HtmlElement} */ el) => {
- const ns = _namespaceOf(el);
- const tag = _tagNameOf(el);
- if (ns === NS_HTML) return SPECIAL.has(tag);
- if (ns === NS_MATHML) return MATHML_SPECIAL.has(tag);
- if (ns === NS_SVG) return SVG_SPECIAL.has(tag.toLowerCase());
- return false;
- };
- // `<font color|face|size>` breaks out of foreign content (§13.2.6.5).
- const hasFontBreakoutAttr = (/** @type {AttributeRun} */ run) => {
- for (let i = run.start; i < run.start + run.count; i++) {
- if (FONT_BREAKOUT_ATTRS.has(_attrNames[i])) return true;
- }
- return false;
- };
- /**
- * Hash of the ASCII-lowercased `name` for the intern tables below; must stay
- * in sync with the range hash in `internLowerName`.
- * @param {string} name lowercase name
- * @returns {number} hash
- */
- const hashLowerName = (name) => {
- let h = name.length;
- for (let i = 0; i < name.length; i++) {
- let c = name.charCodeAt(i);
- if (c >= 0x41 && c <= 0x5a) c += 0x20;
- h = (Math.imul(h, 31) + c) | 0;
- }
- return h;
- };
- // Text-run scan classes for the `skip.text` fast path: 2 = stop the fast
- // path (& / NUL / CR), 1 = ASCII whitespace, 0 = ordinary text.
- const _TEXT_SCAN_CLASS = new Uint8Array(128);
- _TEXT_SCAN_CLASS[0x09] = 1;
- _TEXT_SCAN_CLASS[0x0a] = 1;
- _TEXT_SCAN_CLASS[0x0c] = 1;
- _TEXT_SCAN_CLASS[0x20] = 1;
- _TEXT_SCAN_CLASS[0x26] = 2;
- _TEXT_SCAN_CLASS[0x00] = 2;
- _TEXT_SCAN_CLASS[0x0d] = 2;
- /**
- * @param {Iterable<string>} names lowercase names to intern
- * @returns {{ mask: number, hashes: Int32Array, values: (string | undefined)[], memo: string[], sourceBacked: number[], noted: Uint8Array }} open-addressed intern table
- */
- const buildNameInternTable = (names) => {
- // Open-addressed table (~25% load): the per-name probe is one or two array
- // reads instead of a `Map#get`, and the tables are built once at startup.
- const unique = [...new Set(names)];
- let size = 8;
- while (size < unique.length * 4) size <<= 1;
- const mask = size - 1;
- const hashes = new Int32Array(size);
- /** @type {(string | undefined)[]} */
- const values = Array.from({ length: size });
- for (const name of unique) {
- const h = hashLowerName(name);
- let slot = h & mask;
- // Two names sharing a hash take two slots rather than one bucket, so a
- // slot holds one string and the lookup's load stays monomorphic.
- while (values[slot] !== undefined) slot = (slot + 1) & mask;
- hashes[slot] = h;
- values[slot] = name;
- }
- // `memo` short-circuits the hash walk for a name seen a moment ago (see
- // `internLowerName`); "" marks an empty slot so the load stays monomorphic.
- // `sourceBacked` lists the slots holding a slice of the parsed input — the
- // only ones the next parse drops — and `noted` keeps each in it once.
- return {
- mask,
- hashes,
- values,
- memo: Array.from({ length: 512 }).fill(""),
- sourceBacked: [],
- noted: new Uint8Array(512)
- };
- };
- /** @typedef {ReturnType<typeof buildNameInternTable>} NameInternTable */
- /**
- * Drop the memo entries holding a slice of the last parsed input, so no parse
- * retains another's source. Past the memo's own size a full clear is cheaper.
- * @param {NameInternTable} table intern table
- * @returns {void}
- */
- const _dropSourceBackedNames = (table) => {
- const slots = table.sourceBacked;
- if (slots.length === 0) return;
- const { memo, noted } = table;
- // A slot a known name has since claimed is dropped with the rest: the entry is
- // safe to keep, but noting that costs a write on the path every name takes.
- if (slots.length >= memo.length) {
- memo.fill("");
- noted.fill(0);
- } else {
- for (const slot of slots) {
- memo[slot] = "";
- noted[slot] = 0;
- }
- }
- slots.length = 0;
- };
- /**
- * The lowercased name for `input[start..end)`, returning the shared interned
- * string for known names — skipping the per-tag `slice().toLowerCase()`
- * allocation, and making the tree builder's many Set/Map lookups and `===`
- * comparisons on the name hit one string instance with a cached hash.
- * @param {NameInternTable} table intern table
- * @param {string} input source text
- * @param {number} start name start
- * @param {number} end name end
- * @returns {string} lowercased name
- */
- const internLowerName = (table, input, start, end) => {
- // A document spells the same few names over and over, so a direct-mapped
- // memo keyed on folded first char + length answers most calls with one
- // verified compare instead of the full hash walk. Every hit re-verifies
- // against the current range, so a collision only ever misses, never lies.
- let c0 = input.charCodeAt(start);
- if (c0 >= 0x41 && c0 <= 0x5a) c0 += 0x20;
- const memo = table.memo;
- const memoSlot = (Math.imul(c0, 31) + (end - start)) & 511;
- const cached = memo[memoSlot];
- if (cached.length !== 0 && rangeEqualsLowerCase(input, start, end, cached)) {
- return cached;
- }
- let h = end - start;
- for (let i = start; i < end; i++) {
- let c = input.charCodeAt(i);
- if (c >= 0x41 && c <= 0x5a) c += 0x20;
- h = (Math.imul(h, 31) + c) | 0;
- }
- const { mask, hashes, values } = table;
- let slot = h & mask;
- for (;;) {
- const hit = values[slot];
- // Only an empty slot ends the walk: a name that hashes the same as another
- // sits one slot along, so an equal hash that does not match is not a miss.
- if (hit === undefined) break;
- if (hashes[slot] === h && rangeEqualsLowerCase(input, start, end, hit)) {
- return (memo[memoSlot] = hit);
- }
- slot = (slot + 1) & mask;
- }
- // Unknown name (custom element, data-* attribute, non-ASCII, …). Only ASCII
- // folds — a Unicode `toLowerCase` maps U+212A onto `k` and U+0130 onto two
- // characters, neither of which the tokenizer does.
- if (table.noted[memoSlot] === 0) {
- table.noted[memoSlot] = 1;
- table.sourceBacked.push(memoSlot);
- }
- return (memo[memoSlot] = _asciiLowerCase(input.slice(start, end)));
- };
- // Every tag name the tree builder compares against (the sets above already
- // cover most of the spec), plus the remaining standard/foreign names so
- // ordinary documents intern every tag.
- const TAG_NAME_INTERN = buildNameInternTable([
- ...VOID,
- ...SPECIAL,
- ...FORMATTING,
- ...HEADING,
- ...MATHML_TEXT_INTEGRATION,
- ...MATHML_SPECIAL,
- ...SVG_SPECIAL,
- ...HTML_SCOPE,
- ...TABLE_CONTEXT,
- ...VOID_FORMATTING,
- ...NO_DECODE_TEXT,
- ...FOREIGN_BREAKOUT,
- ...HEAD_ELEMENTS,
- ...Object.keys(SVG_TAG_ADJUST),
- ..."a abbr audio bdi bdo canvas cite data datalist del dfn dialog ins kbd label legend map mark math menuitem meter optgroup option output picture progress q rb rp rt rtc ruby samp selectedcontent slot span sub sup svg time u var video".split(
- " "
- )
- ]);
- // Common attribute names (unknown ones — data-*, ARIA, events — fall back).
- const ATTR_NAME_INTERN = buildNameInternTable(
- "href src srcset sizes alt title class id style name type value content charset rel media target action method placeholder disabled checked selected multiple readonly required hidden tabindex role lang dir width height loading decoding async defer integrity crossorigin referrerpolicy nonce as for colspan rowspan span label max min step pattern autocomplete autofocus autoplay controls loop muted poster preload download ping imagesrcset imagesizes slot part is property http-equiv accept enctype novalidate maxlength minlength size cols rows wrap open scope headers datetime cite usemap ismap shape coords start reversed face color encoding xmlns".split(
- " "
- )
- );
- // Hoisted so the many `open.some(...)` "is there an open HTML <template>?"
- // checks reuse one predicate instead of allocating an arrow per call.
- const isHtmlTemplateEl = (/** @type {HtmlElement} */ e) =>
- _tagNameOf(e) === "template" && _namespaceOf(e) === NS_HTML;
- /**
- * Shared empty skip set so the common (no-skip) call allocates nothing.
- * @type {HtmlAstSkip}
- */
- const EMPTY_SKIP = Object.freeze({});
- /**
- * Optional node kinds a consumer can drop from the AST for speed/memory. Each
- * is a pure output reduction — tree construction (and quirks detection) runs
- * unchanged, so element structure and offsets are identical either way.
- * @typedef {object} HtmlAstSkip
- * @property {boolean=} text drop every `Text` node. Raw-text element bodies (`<script>`/`<style>`/…) aren't emitted either — their content span is recorded as the element's `contentEnd` (see `RAW_TEXT_ELEMENTS`) so a consumer can read `[tagEnd, contentEnd]` by offset. For consumers that read text by offset (e.g. `HtmlParser`), never the html5lib serializer.
- * @property {boolean=} comments drop comment nodes entirely. Not for consumers that read comments (e.g. webpack magic comments).
- * @property {boolean=} doctype drop the doctype node; quirks-mode detection is unaffected.
- */
- /**
- * @typedef {object} HtmlParseOptions
- * @property {string=} fragmentContext context element name for fragment parsing (e.g. `td`, `svg path`); omit for a full document
- * @property {HtmlAstSkip=} skip node kinds to omit from the AST (see `HtmlAstSkip`); omit to build the full tree
- */
- /**
- * @param {string} name raw SVG tag name
- * @returns {string} case-adjusted SVG tag name
- */
- const adjustSvgTag = (name) =>
- /** @type {Record<string, string>} */ (SVG_TAG_ADJUST)[name] || name;
- /**
- * @param {string} name doctype name
- * @param {string | null} pub public id
- * @param {string | null} sys system id
- * @returns {boolean} whether the doctype forces quirks mode
- */
- const isQuirky = (name, pub, sys) => {
- if (name !== "html") return true;
- const p = pub ? pub.toLowerCase() : null;
- const sl = sys ? sys.toLowerCase() : null;
- if (p !== null) {
- if (QUIRKY_EXACT.has(p)) return true;
- for (const pre of QUIRKY_PREFIXES) if (p.startsWith(pre)) return true;
- if (
- sl === null &&
- (p.startsWith("-//w3c//dtd html 4.01 frameset//") ||
- p.startsWith("-//w3c//dtd html 4.01 transitional//"))
- ) {
- return true;
- }
- }
- if (sl === "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
- return true;
- }
- return false;
- };
- const mkEl = (
- /** @type {string} */ tagName,
- /** @type {number} */ ns,
- /** @type {AttributeRun} */ attrs,
- /** @type {TagPos | null | undefined} */ pos
- ) => {
- // Inlined element allocation: every column is written exactly once (the
- // generic _allocNode would zero five columns mkEl immediately overwrites).
- const el = ++_nodeCount;
- if (el >= _nodeCapacity) _growNodeColumns(el + 1);
- _nodeTypes[el] = NodeType.Element;
- // Void HTML elements are marked self-closing and never receive children.
- _nodeFlags[el] =
- ns === NS_HTML && VOID.has(tagName) ? ns | FLAG_SELF_CLOSING : ns;
- if (pos) {
- _nodeStarts[el] = pos.start;
- _nodeEnds[el] = pos.end;
- _nodeTagEnds[el] = pos.tagEnd;
- _nodeNameEnds[el] = pos.nameEnd;
- // End of a raw-text element's body under `skip.text` (defaults to the
- // body start, i.e. empty); lets consumers read `<script>`/`<style>`
- // content as [`tagEnd`, `contentEnd`] without a `Text` node.
- _nodeContentEnds[el] = pos.tagEnd;
- } else {
- _nodeStarts[el] = 0;
- _nodeEnds[el] = 0;
- _nodeTagEnds[el] = 0;
- _nodeNameEnds[el] = 0;
- _nodeContentEnds[el] = 0;
- }
- _nodeParents[el] = 0;
- _nodeFirstChildren[el] = 0;
- _nodeLastChildren[el] = 0;
- _nodeNextSiblings[el] = 0;
- _nodeAttrStarts[el] = attrs.start;
- _nodeAttrCounts[el] = attrs.count;
- _nodeStrings[el] = tagName;
- return el;
- };
- /**
- * A `<template>`'s children live in its content fragment; inserting into a
- * template really inserts there (the spec's "template contents" redirect).
- * @param {HtmlNodeRef} parent container
- * @returns {HtmlNodeRef} effective container to link children into
- */
- const effParent = (parent) => {
- const tc = _templateContentOf(parent);
- return tc !== 0 ? tc : parent;
- };
- const isScopeBoundary = (/** @type {HtmlElement} */ el) => {
- if (_namespaceOf(el) === NS_HTML) return HTML_SCOPE.has(_tagNameOf(el));
- if (_namespaceOf(el) === NS_MATHML) {
- return MATHML_SPECIAL.has(_tagNameOf(el));
- }
- if (_namespaceOf(el) === NS_SVG) {
- return SVG_SPECIAL.has(_tagNameOf(el).toLowerCase());
- }
- return false;
- };
- // Attribute values are stored raw (offsets into the source), so two unresolved
- // values compare by range — `_attrValueOf` would slice a string per side just
- // to throw it away.
- /** @type {(i: number, j: number) => boolean} */
- const _attrValueEquals = (i, j) => {
- if (_attrValues[i] === null && _attrValues[j] === null) {
- const start = _attrValueStarts[i];
- const len = _attrValueEnds[i] - start;
- const otherStart = _attrValueStarts[j];
- if (len !== _attrValueEnds[j] - otherStart) return false;
- for (let k = 0; k < len; k++) {
- if (
- _htmlSource.charCodeAt(start + k) !==
- _htmlSource.charCodeAt(otherStart + k)
- ) {
- return false;
- }
- }
- return true;
- }
- return _attrValueOf(i) === _attrValueOf(j);
- };
- const sameAttrs = (
- /** @type {HtmlElement} */ a,
- /** @type {HtmlElement} */ b
- ) => {
- const aStart = _nodeAttrStarts[a];
- const aCount = _nodeAttrCounts[a];
- const bStart = _nodeAttrStarts[b];
- const bCount = _nodeAttrCounts[b];
- if (aCount !== bCount) return false;
- // Names are unique (deduped), counts tiny — a nested scan beats a Map
- // here on this formatting-element hot path.
- for (let i = bStart; i < bStart + bCount; i++) {
- const j = _findAttr(aStart, aCount, _attrNames[i]);
- if (j === 0 || !_attrValueEquals(j, i)) return false;
- }
- return true;
- };
- const startsWithWs = (/** @type {string} */ s) => {
- const c = s.charCodeAt(0);
- return c === 0x09 || c === 0x0a || c === 0x0c || c === 0x0d || c === 0x20;
- };
- const isAllWs = (/** @type {string} */ s) => {
- for (let i = 0; i < s.length; i++) {
- const c = s.charCodeAt(i);
- // HTML whitespace: tab / LF / FF / CR / space. A charCodeAt loop avoids
- // the `for…of` code-point iterator + per-char string + Set lookup.
- if (c !== 0x09 && c !== 0x0a && c !== 0x0c && c !== 0x0d && c !== 0x20) {
- return false;
- }
- }
- return true;
- };
- // A whitespace run the printer can shorten: two of them in a row, or a lone one
- // that is not already a space.
- const _COLLAPSIBLE_WHITESPACE_REGEXP = /[\t\n\f\r]|[ ][ ]/;
- /**
- * Collapse every run of HTML whitespace to one space — never remove one:
- * `a <b>c</b>` and `a<b>c</b>` render differently.
- * @param {string} s text data
- * @returns {string} the collapsed text
- */
- const collapseWhitespaceRuns = (s) => {
- // Most text collapses to itself, and one scan answers that before the walk
- // below reaches its first run.
- if (!_COLLAPSIBLE_WHITESPACE_REGEXP.test(s)) return s;
- let out = "";
- let last = 0;
- let run = -1;
- for (let i = 0; i <= s.length; i++) {
- const c = i === s.length ? 0 : s.charCodeAt(i);
- const isWhitespace =
- c === 0x09 || c === 0x0a || c === 0x0c || c === 0x0d || c === 0x20;
- if (isWhitespace) {
- if (run === -1) run = i;
- continue;
- }
- // A one-character run is already a single space unless it is a tab or a
- // newline, which still print shorter as one.
- if (run !== -1 && (i - run > 1 || s.charCodeAt(run) !== 0x20)) {
- out += `${s.slice(last, run)} `;
- last = i;
- }
- run = -1;
- }
- return last === 0 ? s : out + s.slice(last);
- };
- const mathmlTextIntegrationPoint = (/** @type {HtmlElement} */ el) =>
- _namespaceOf(el) === NS_MATHML && MATHML_TEXT_INTEGRATION.has(_tagNameOf(el));
- const htmlIntegrationPoint = (/** @type {HtmlElement} */ el) => {
- if (_namespaceOf(el) === NS_MATHML && _tagNameOf(el) === "annotation-xml") {
- const enc = _findAttr(_nodeAttrStarts[el], _nodeAttrCounts[el], "encoding");
- if (enc !== 0) {
- const value = _attrValueOf(enc).toLowerCase();
- if (value === "text/html" || value === "application/xhtml+xml") {
- return true;
- }
- }
- return false;
- }
- if (
- _namespaceOf(el) === NS_SVG &&
- SVG_SPECIAL.has(_tagNameOf(el).toLowerCase())
- ) {
- return true;
- }
- return false;
- };
- // Adjust a foreign start tag's attribute run in place (the run is consumed
- // only by this tag's element): SVG camelCase names are rewritten, and
- // namespaced names get the serializer-name flag (see `_attrSerializedName`).
- const adjustForeignAttrs = (
- /** @type {AttributeRun} */ run,
- /** @type {number} */ ns
- ) => {
- for (let i = run.start; i < run.start + run.count; i++) {
- const name = _attrNames[i];
- if (
- ns === NS_SVG &&
- /** @type {Record<string, string>} */ (SVG_ATTR_ADJUST)[name]
- ) {
- _attrNames[i] = /** @type {Record<string, string>} */ (SVG_ATTR_ADJUST)[
- name
- ];
- }
- if (/** @type {Record<string, string>} */ (FOREIGN_ATTR_NS)[name]) {
- _attrFlags[i] |= 1;
- }
- }
- return run;
- };
- // Only `definitionurl` is rewritten (in place — the run is consumed only by
- // this tag's element); the camelCase name serializes as itself.
- const adjustMathmlAttrs = (/** @type {AttributeRun} */ run) => {
- for (let i = run.start; i < run.start + run.count; i++) {
- if (_attrNames[i] === "definitionurl") _attrNames[i] = "definitionURL";
- }
- return run;
- };
- // ---- HTML tree-construction parser state ----
- // One active parse at a time (results are valid only until the next
- // `parseHtml`); kept at module scope so the helpers below are defined once
- // instead of re-created as closures on every parse.
- let source = "";
- // Whole-document properties, so a substring cannot contain what the input does
- // not. Decoding cannot introduce either: a numeric reference to 0 yields U+FFFD,
- // and CR normalization runs on the raw slice before decoding. Default to `true`
- // so an unset flag only ever costs the old scan.
- let inputHasCr = true;
- let inputHasNul = true;
- let skipText = false;
- let skipComments = false;
- let skipDoctype = false;
- /** @type {HtmlNodeRef} */
- let doc = 0;
- let mode = 0;
- let originalMode = 0;
- /** @type {HtmlElement[]} stack of open elements (bottom .. top) */
- const open = [];
- /** @type {HtmlNodeRef[]} active formatting elements (AFE_MARKER = marker) */
- const afe = [];
- /** @type {HtmlElement} 0 = none */
- let head = 0;
- /** @type {HtmlElement} 0 = none */
- let form = 0;
- let framesetOk = true;
- let pendAttrStart = 0;
- let fosterParenting = false;
- /** @type {number[]} */
- const templateModes = [];
- let quirks = false;
- /** @type {HtmlElement} fragment context element (0 = document parse) */
- let fragment = 0;
- let tokenEnd = 0;
- let swallowNextNewline = false;
- let eofInTag = false;
- let sawSelectedContent = false;
- const decode = decodeEntities;
- /** @type {MutableToken} */
- const tok = {
- type: TOKEN_EOF,
- name: "",
- data: "",
- attrs: { start: 0, count: 0 },
- selfClosing: false,
- start: 0,
- end: 0,
- publicId: null,
- systemId: null,
- forceQuirks: false,
- swallowNewline: false,
- pos: { start: 0, end: 0, tagEnd: 0, nameEnd: 0 }
- };
- const cur = () =>
- /** @type {HtmlElement} */ (open.length > 0 ? open[open.length - 1] : 0);
- /**
- * Remove an element from the middle of the open element stack. A removal that
- * leaves still-open descendants above it breaks the streamed walk whether or
- * not the walk entered the element: it drops off the stack without its subtree
- * ending, so those descendants would be entered under a parent the walk has no
- * open-stack evidence for. Removing the top is just a pop and keeps streaming —
- * which is the common `</form>` after implied end tags.
- * @param {number} index position in `open` to remove
- * @returns {void}
- */
- const removeOpenAt = (index) => {
- open.splice(index, 1);
- _openSpliced = true;
- _openStackChanged = true;
- if (_streaming && index < open.length) _streamHalted = true;
- };
- /**
- * Pop the open element stack. Every pop routes through here so the streamed
- * walk has a single completion point — the HTML analog of the CSS parser's
- * `onRule` sink — instead of being polled once per token.
- * @returns {HtmlElement} the popped element
- */
- const popOpen = () => {
- const el = /** @type {HtmlElement} */ (open.pop());
- // Only mark: an element's end offset is assigned after it is popped, so the
- // walk has to wait for the token to finish.
- _openStackChanged = true;
- return el;
- };
- const adjustedCurrent = () => {
- if (open.length === 1 && fragment) return fragment;
- return cur();
- };
- const appendTo = (
- /** @type {HtmlNodeRef} */ parent,
- /** @type {HtmlNodeRef} */ node
- ) => {
- const p = effParent(parent);
- const last = _nodeLastChildren[p];
- if (
- _nodeTypes[node] === NodeType.Text &&
- last !== 0 &&
- _nodeTypes[last] === NodeType.Text
- ) {
- _nodeStrings[last] += _nodeStrings[node];
- _nodeEnds[last] = _nodeEnds[node];
- return;
- }
- _appendChild(p, node);
- };
- // Reused result of `appropriatePlace` — consumed synchronously by
- // `insertAtPlace` and never retained, so one shared object avoids an
- // allocation per inserted node.
- /** @type {InsertionPlace} */
- const sharedPlace = { parent: doc, beforeNode: 0 };
- const placeAt = (
- /** @type {HtmlNodeRef} */ parent,
- /** @type {HtmlNodeRef} */ beforeNode
- ) => {
- sharedPlace.parent = parent;
- sharedPlace.beforeNode = beforeNode;
- return sharedPlace;
- };
- /**
- * §13.2.6.1 "appropriate place for inserting a node", the foster-parenting
- * branch: where a node goes when the target is a table context — inside the
- * innermost `<template>` when one stands above the innermost `<table>`, else
- * before that table, and at the root when neither is open.
- * @returns {[HtmlNodeRef, HtmlNodeRef]} the parent and the node to insert before (`0` = append)
- */
- const _fosterPlace = () => {
- // The innermost of each still open, found in one walk down the stack.
- let lastTemplate = -1;
- let lastTable = -1;
- for (let i = open.length - 1; i >= 0; i--) {
- if (_namespaceOf(open[i]) !== NS_HTML) continue;
- const name = _tagNameOf(open[i]);
- if (lastTemplate === -1 && name === "template") lastTemplate = i;
- else if (lastTable === -1 && name === "table") lastTable = i;
- }
- if (lastTemplate !== -1 && (lastTable === -1 || lastTemplate > lastTable)) {
- return [open[lastTemplate], 0];
- }
- if (lastTable === -1) return [open[0], 0];
- const table = open[lastTable];
- const parent = _nodeParents[table];
- // A table already taken out of the tree fosters before the element that
- // stood under it instead.
- return parent !== 0 ? [parent, table] : [open[lastTable - 1], 0];
- };
- // "appropriate place for inserting a node"
- const appropriatePlace = () => {
- const target = cur();
- if (
- fosterParenting &&
- TABLE_CONTEXT.has(_tagNameOf(target)) &&
- _namespaceOf(target) === NS_HTML
- ) {
- const [parent, beforeNode] = _fosterPlace();
- return placeAt(parent, beforeNode);
- }
- return placeAt(target, 0);
- };
- const insertAtPlace = (
- /** @type {InsertionPlace} */ place,
- /** @type {HtmlNodeRef} */ node
- ) => {
- const before = place.beforeNode;
- if (before !== 0) {
- const p = effParent(place.parent);
- // Find `before`'s previous sibling (insert-before is a rare foster/
- // adoption path, so the sibling scan stays off the hot path).
- let prev = 0;
- let c = _nodeFirstChildren[p];
- while (c !== 0 && c !== before) {
- prev = c;
- c = _nodeNextSiblings[c];
- }
- if (c === 0) {
- // `before` not under `parent` (not reachable from the spec paths).
- appendTo(place.parent, node);
- return;
- }
- if (
- _nodeTypes[node] === NodeType.Text &&
- prev !== 0 &&
- _nodeTypes[prev] === NodeType.Text
- ) {
- // Before-node merge deliberately does not bump the sibling's `end`.
- _nodeStrings[prev] += _nodeStrings[node];
- return;
- }
- _nodeParents[node] = p;
- _nodeNextSiblings[node] = before;
- if (prev === 0) _nodeFirstChildren[p] = node;
- else _nodeNextSiblings[prev] = node;
- if (_namespaceOf(before) === NS_HTML && _tagNameOf(before) === "table") {
- _recordFostered(before, node);
- }
- } else {
- appendTo(place.parent, node);
- }
- };
- /**
- * Remember that `node` was foster parented out of `table`. The parent it landed
- * in is not recorded: the adoption agency can move it elsewhere afterwards, so
- * only the tree answers that.
- * @param {HtmlNodeRef} table the table it was written inside
- * @param {HtmlNodeRef} node the fostered node
- * @returns {void}
- */
- const _recordFostered = (table, node) => {
- if (_fosteredLog === null) _fosteredLog = [];
- _fosteredLog.push(table, node);
- };
- const insertCharacters = (
- /** @type {string} */ data,
- /** @type {number} */ start,
- /** @type {number} */ end
- ) => {
- const place = appropriatePlace();
- if (_nodeTypes[place.parent] === NodeType.Document) return; // never insert text into document
- // `skip.text`: drop every `Text` node — construction already used the
- // decoded token, so removing the node never affects element structure.
- // For a raw-text element record the body end so a consumer reads the span
- // [`tagEnd`, `contentEnd`] without a `Text` node (see `HtmlParser`).
- // Namespace-agnostic: `HtmlParser` extracts `<script>`/`<style>` bodies in
- // foreign content (e.g. SVG `<style>`) too.
- if (skipText) {
- const p = place.parent;
- if (
- _nodeTypes[p] === NodeType.Element &&
- RAW_TEXT_ELEMENTS.has(_tagNameOf(p))
- ) {
- _nodeContentEnds[p] = end;
- }
- return;
- }
- // Inlined text insert: when the run merges into the adjacent text sibling
- // (common with inline formatting) only the string is appended — no
- // throwaway text node is allocated. Mirrors `insertAtPlace`/`appendTo`,
- // including that the before-node merge does not bump `end`.
- const p = effParent(place.parent);
- const before = place.beforeNode;
- if (before !== 0) {
- let prev = 0;
- let c = _nodeFirstChildren[p];
- while (c !== 0 && c !== before) {
- prev = c;
- c = _nodeNextSiblings[c];
- }
- if (c === 0) {
- // `before` not under `parent` (not reachable from the spec paths).
- _appendChild(p, _makeTextNode(data, start, end));
- return;
- }
- if (prev !== 0 && _nodeTypes[prev] === NodeType.Text) {
- _nodeStrings[prev] += data;
- return;
- }
- const node = _makeTextNode(data, start, end);
- _nodeParents[node] = p;
- _nodeNextSiblings[node] = before;
- if (prev === 0) _nodeFirstChildren[p] = node;
- else _nodeNextSiblings[prev] = node;
- if (_namespaceOf(before) === NS_HTML && _tagNameOf(before) === "table") {
- _recordFostered(before, node);
- }
- } else {
- const last = _nodeLastChildren[p];
- if (last !== 0 && _nodeTypes[last] === NodeType.Text) {
- _nodeStrings[last] += data;
- _nodeEnds[last] = end;
- return;
- }
- _appendChild(p, _makeTextNode(data, start, end));
- }
- };
- /**
- * @param {CommentToken | ProcessingInstructionToken} t comment or processing instruction token
- * @param {InsertionPlace=} place explicit insertion place
- */
- // Every insertion mode inserts a processing instruction where it inserts a
- // comment (§13.2.6.4). `skip.comments` drops comments alone.
- const insertCommentOrProcessingInstruction = (t, place) => {
- const isPi = t.type === TOKEN_PROCESSING_INSTRUCTION;
- if (skipComments && !isPi) return;
- const p = place || appropriatePlace();
- insertAtPlace(
- p,
- isPi
- ? _makeProcessingInstructionNode(t.name, t.data, t.start, t.end)
- : _makeCommentNode(t.data, t.start, t.end)
- );
- };
- const insertHtmlElement = (
- /** @type {string} */ tagName,
- /** @type {AttributeRun} */ attrs,
- /** @type {TagPos | null} */ pos
- ) => {
- const el = mkEl(tagName, NS_HTML, attrs, pos);
- const place = appropriatePlace();
- insertAtPlace(place, el);
- open.push(el);
- return el;
- };
- const insertForeignElement = (
- /** @type {string} */ tagName,
- /** @type {number} */ ns,
- /** @type {AttributeRun} */ attrs,
- /** @type {TagPos | null} */ pos
- ) => {
- const el = mkEl(tagName, ns, attrs, pos);
- const place = appropriatePlace();
- insertAtPlace(place, el);
- open.push(el);
- return el;
- };
- // ---- scopes ----
- // Scope "kind" selects which extra elements act as boundaries. Passed as a
- // small int so the scope checks below allocate no per-call predicate closure
- // (these run several times per body tag).
- const SCOPE_DEFAULT = 0;
- const SCOPE_BUTTON = 1;
- const SCOPE_LIST_ITEM = 2;
- const isBoundaryForKind = (
- /** @type {HtmlElement} */ el,
- /** @type {number} */ kind
- ) => {
- if (isScopeBoundary(el)) return true;
- if (_namespaceOf(el) !== NS_HTML) return false;
- if (kind === SCOPE_BUTTON) return _tagNameOf(el) === "button";
- if (kind === SCOPE_LIST_ITEM) {
- return _tagNameOf(el) === "ol" || _tagNameOf(el) === "ul";
- }
- return false;
- };
- // "have an element in scope": walk the open stack from the top until the
- // named HTML element is found (true) or a scope boundary is hit (false).
- const hasNameInScope = (
- /** @type {string} */ tagName,
- /** @type {number} */ kind
- ) => {
- for (let i = open.length - 1; i >= 0; i--) {
- const el = open[i];
- if (_namespaceOf(el) === NS_HTML && _tagNameOf(el) === tagName) {
- return true;
- }
- if (isBoundaryForKind(el, kind)) return false;
- }
- return false;
- };
- const inScope = (/** @type {string} */ tagName) =>
- hasNameInScope(tagName, SCOPE_DEFAULT);
- const inButtonScope = (/** @type {string} */ tagName) =>
- hasNameInScope(tagName, SCOPE_BUTTON);
- const inListItemScope = (/** @type {string} */ tagName) =>
- hasNameInScope(tagName, SCOPE_LIST_ITEM);
- const inScopeEl = (/** @type {HtmlElement} */ target) => {
- for (let i = open.length - 1; i >= 0; i--) {
- const el = open[i];
- if (el === target) return true;
- if (isScopeBoundary(el)) return false;
- }
- return false;
- };
- // `target` is a single tag name (the common case) or a Set of names.
- const inTableScope = (/** @type {string | Set<string>} */ target) => {
- const set = typeof target === "string" ? null : target;
- for (let i = open.length - 1; i >= 0; i--) {
- const el = open[i];
- if (_namespaceOf(el) === NS_HTML) {
- if (set ? set.has(_tagNameOf(el)) : _tagNameOf(el) === target) {
- return true;
- }
- if (TABLE_SCOPE_STOP.has(_tagNameOf(el))) return false;
- }
- }
- return false;
- };
- const generateImpliedEndTags = (except = "") => {
- while (open.length) {
- const el = cur();
- if (
- _namespaceOf(el) === NS_HTML &&
- IMPLIED.has(_tagNameOf(el)) &&
- _tagNameOf(el) !== except
- ) {
- popOpen();
- } else {
- break;
- }
- }
- };
- const generateImpliedEndTagsThorough = () => {
- while (open.length) {
- const el = cur();
- if (_namespaceOf(el) === NS_HTML && IMPLIED_THOROUGH.has(_tagNameOf(el))) {
- popOpen();
- } else {
- break;
- }
- }
- };
- // ---- active formatting elements ----
- // `splice` allocates an array for the removed element on every call; the list
- // is short, so shifting in place is both cheaper and allocation-free.
- /** @type {(i: number) => void} */
- const afeRemoveAt = (i) => {
- const last = afe.length - 1;
- for (let k = i; k < last; k++) afe[k] = afe[k + 1];
- afe.pop();
- };
- const pushAfe = (/** @type {HtmlElement} */ el) => {
- let count = 0;
- for (let i = afe.length - 1; i >= 0; i--) {
- const e = afe[i];
- if (e === AFE_MARKER) break;
- if (
- _tagNameOf(e) === _tagNameOf(el) &&
- _namespaceOf(e) === _namespaceOf(el) &&
- sameAttrs(e, el)
- ) {
- count++;
- if (count === 3) {
- afeRemoveAt(i);
- break;
- }
- }
- }
- afe.push(el);
- };
- const insertMarker = () => afe.push(AFE_MARKER);
- const clearAfeToMarker = () => {
- while (afe.length) {
- if (afe.pop() === AFE_MARKER) break;
- }
- };
- const reconstructAfe = () => {
- if (afe.length === 0) return;
- let i = afe.length - 1;
- if (afe[i] === AFE_MARKER || open.includes(afe[i])) return;
- while (i > 0) {
- i--;
- if (afe[i] === AFE_MARKER || open.includes(afe[i])) {
- i++;
- break;
- }
- }
- for (; i < afe.length; i++) {
- const e = afe[i];
- // §13.2.6.4.7 gives `a` and `nobr` a start-tag rule of their own; every other
- // formatting element re-nests as written.
- _printFromSource = true;
- const el = mkEl(_tagNameOf(e), _namespaceOf(e), cloneAttrs(e), null);
- const place = appropriatePlace();
- insertAtPlace(place, el);
- open.push(el);
- afe[i] = el;
- }
- };
- // ---- close p ----
- const closePElement = () => {
- generateImpliedEndTags("p");
- // pop until a p has been popped
- while (open.length) {
- const el = /** @type {HtmlElement} */ (popOpen());
- _nodeEnds[el] = tokenEnd;
- if (_namespaceOf(el) === NS_HTML && _tagNameOf(el) === "p") {
- break;
- }
- }
- };
- const popUntil = (/** @type {string} */ tagName) => {
- while (open.length) {
- const el = /** @type {HtmlElement} */ (popOpen());
- _nodeEnds[el] = tokenEnd;
- if (_namespaceOf(el) === NS_HTML && _tagNameOf(el) === tagName) {
- break;
- }
- }
- };
- const popUntilOneOf = (/** @type {Set<string>} */ set) => {
- while (open.length) {
- const el = /** @type {HtmlElement} */ (popOpen());
- _nodeEnds[el] = tokenEnd;
- if (_namespaceOf(el) === NS_HTML && set.has(_tagNameOf(el))) {
- break;
- }
- }
- };
- // ---- reset insertion mode appropriately ----
- const resetInsertionMode = () => {
- let last = false;
- for (let i = open.length - 1; i >= 0; i--) {
- let node = open[i];
- if (i === 0) {
- last = true;
- if (fragment) node = fragment;
- }
- const tn = _tagNameOf(node);
- if (_namespaceOf(node) === NS_HTML) {
- if ((tn === "td" || tn === "th") && !last) {
- mode = MODE_IN_CELL;
- return;
- }
- if (tn === "tr") {
- mode = MODE_IN_ROW;
- return;
- }
- if (TBODY_GROUP.has(tn)) {
- mode = MODE_IN_TABLE_BODY;
- return;
- }
- if (tn === "caption") {
- mode = MODE_IN_CAPTION;
- return;
- }
- if (tn === "colgroup") {
- mode = MODE_IN_COLUMN_GROUP;
- return;
- }
- if (tn === "table") {
- mode = MODE_IN_TABLE;
- return;
- }
- if (tn === "template") {
- mode = templateModes[templateModes.length - 1];
- return;
- }
- if (tn === "head" && !last) {
- mode = MODE_IN_HEAD;
- return;
- }
- if (tn === "body") {
- mode = MODE_IN_BODY;
- return;
- }
- if (tn === "frameset") {
- mode = MODE_IN_FRAMESET;
- return;
- }
- if (tn === "html") {
- mode = head ? MODE_AFTER_HEAD : MODE_BEFORE_HEAD;
- return;
- }
- }
- if (last) {
- mode = MODE_IN_BODY;
- return;
- }
- }
- };
- // ---------- token processing ----------
- // Split a character token's leading whitespace; per the spec each character
- // is its own token, so a mixed run can straddle a mode change. Inserts the
- // leading whitespace when `insert`, returns the non-whitespace remainder
- // token (or null when the token was entirely whitespace).
- const leadingWs = (
- /** @type {CharToken} */ t,
- /** @type {boolean} */ insert
- ) => {
- const m = /^[\t\n\f\r ]+/.exec(t.data);
- const ws = m ? m[0] : "";
- // `t.data` is decoded, the offsets are raw, so counting one in the other
- // only lines up when decoding changed no length (no CRLF, no reference).
- // Otherwise keep the whole span: too wide is harmless, too narrow names
- // source that is not this text (see the `Text` printer).
- const aligned = t.end - t.start === t.data.length;
- if (ws && insert) {
- insertCharacters(ws, t.start, aligned ? t.start + ws.length : t.end);
- }
- if (ws.length === t.data.length) return null;
- return {
- ...t,
- data: t.data.slice(ws.length),
- start: aligned ? t.start + ws.length : t.start
- };
- };
- const process = (/** @type {Token} */ t) => {
- // Track the current token's end so explicit closes can set element `.end`.
- // Dispatch on `type` instead of the `in` operator, which goes megamorphic
- // across the token union and shows up on the per-token hot path.
- const ty = t.type;
- if (ty === TOKEN_START_TAG || ty === TOKEN_END_TAG) tokenEnd = t.pos.end;
- else if (ty !== TOKEN_EOF) tokenEnd = t.end;
- // foreign content dispatch
- const ac = adjustedCurrent();
- const useForeign =
- open.length > 0 &&
- ac &&
- _namespaceOf(ac) !== NS_HTML &&
- ty !== TOKEN_EOF &&
- shouldUseForeignRules(ac, t);
- if (useForeign) {
- foreignContent(t);
- return;
- }
- runMode(t);
- };
- const shouldUseForeignRules = (
- /** @type {HtmlElement} */ ac,
- /** @type {Token} */ t
- ) => {
- if (_namespaceOf(ac) === NS_HTML) return false;
- if (t.type === TOKEN_START_TAG) {
- if (
- mathmlTextIntegrationPoint(ac) &&
- t.name !== "mglyph" &&
- t.name !== "malignmark"
- ) {
- return false;
- }
- if (
- _namespaceOf(ac) === NS_MATHML &&
- _tagNameOf(ac) === "annotation-xml" &&
- t.name === "svg"
- ) {
- return false;
- }
- if (htmlIntegrationPoint(ac)) return false;
- return true;
- }
- if (t.type === TOKEN_CHAR) {
- if (mathmlTextIntegrationPoint(ac)) return false;
- if (htmlIntegrationPoint(ac)) return false;
- return true;
- }
- if (t.type === TOKEN_END_TAG) return true;
- if (isCommentOrProcessingInstruction(t)) return true;
- return false;
- };
- const foreignContent = (/** @type {Token} */ t) => {
- if (t.type === TOKEN_CHAR) {
- const data = inputHasNul ? t.data.replace(/\0/g, "�") : t.data;
- insertCharacters(data, t.start, t.end);
- // eslint-disable-next-line no-control-regex
- if (/[^\t\n\f\r \u0000]/.test(t.data)) framesetOk = false;
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG) {
- const acn = _namespaceOf(adjustedCurrent());
- if (
- FOREIGN_BREAKOUT.has(t.name) ||
- (t.name === "font" && hasFontBreakoutAttr(t.attrs))
- ) {
- // parse error; pop until integration point / html / mathml-text-integration
- while (open.length > 1) {
- const c = cur();
- if (
- _namespaceOf(c) === NS_HTML ||
- mathmlTextIntegrationPoint(c) ||
- htmlIntegrationPoint(c)
- ) {
- break;
- }
- popOpen();
- }
- runMode(t);
- return;
- }
- const ns = acn;
- let name = t.name;
- let attrs = t.attrs;
- if (ns === NS_SVG) {
- name = adjustSvgTag(name);
- }
- if (ns === NS_MATHML) {
- attrs = adjustMathmlAttrs(attrs);
- }
- attrs = adjustForeignAttrs(attrs, ns);
- insertForeignElement(name, ns, attrs, t.pos);
- if (t.selfClosing) {
- popOpen();
- }
- return;
- }
- if (t.type === TOKEN_END_TAG) {
- if (
- t.name === "script" &&
- _tagNameOf(cur()) === "script" &&
- _namespaceOf(cur()) === NS_SVG
- ) {
- popOpen();
- return;
- }
- // `</p>` and `</br>` break out: pop foreign elements up to the
- // nearest HTML element or integration point, then process in HTML.
- if (t.name === "p" || t.name === "br") {
- while (
- open.length > 1 &&
- _namespaceOf(cur()) !== NS_HTML &&
- !mathmlTextIntegrationPoint(cur()) &&
- !htmlIntegrationPoint(cur())
- ) {
- popOpen();
- }
- runMode(t);
- return;
- }
- // any other end tag
- let i = open.length - 1;
- let node = open[i];
- if (_tagNameOf(node).toLowerCase() !== t.name) {
- /* parse error */
- }
- while (i >= 0) {
- node = open[i];
- if (i === 0) return;
- if (
- _namespaceOf(node) !== NS_HTML &&
- _tagNameOf(node).toLowerCase() === t.name
- ) {
- while (open.length > i) popOpen();
- return;
- }
- i--;
- if (open[i] && _namespaceOf(open[i]) === NS_HTML) {
- runMode(t);
- return;
- }
- }
- }
- };
- // ---------- adoption agency algorithm ----------
- const adoptionAgency = (
- /** @type {string} */ subject,
- /** @type {TagPos | null} */ pos
- ) => {
- // step 1
- const c = cur();
- if (
- _namespaceOf(c) === NS_HTML &&
- _tagNameOf(c) === subject &&
- !afe.includes(c)
- ) {
- popOpen();
- return true;
- }
- let outer = 0;
- while (outer < 8) {
- outer++;
- // find formatting element
- let fmtIdx = -1;
- for (let i = afe.length - 1; i >= 0; i--) {
- if (afe[i] === AFE_MARKER) {
- if (_afeNameBelowMarker(i, subject)) _printFromSource = true;
- break;
- }
- if (_tagNameOf(afe[i]) === subject && _namespaceOf(afe[i]) === NS_HTML) {
- fmtIdx = i;
- break;
- }
- }
- if (fmtIdx === -1) return false; // act as any other end tag
- const fmt = afe[fmtIdx];
- const openIdx = open.indexOf(fmt);
- if (openIdx === -1) {
- afeRemoveAt(fmtIdx);
- return true;
- }
- if (!inScopeEl(fmt)) return true; // parse error, ignore
- // step: furthest block
- let furthestIdx = -1;
- for (let i = openIdx + 1; i < open.length; i++) {
- if (isSpecial(open[i])) {
- furthestIdx = i;
- break;
- }
- }
- if (furthestIdx === -1) {
- while (open.length > openIdx) {
- _nodeEnds[/** @type {HtmlElement} */ (open[open.length - 1])] =
- tokenEnd;
- popOpen();
- }
- afeRemoveAt(fmtIdx);
- return true;
- }
- const furthest = open[furthestIdx];
- const commonAncestor = open[openIdx - 1];
- let bookmark = fmtIdx;
- let node = furthest;
- let lastNode = furthest;
- let nodeIdx = furthestIdx;
- let inner = 0;
- while (true) {
- inner++;
- nodeIdx--;
- node = open[nodeIdx];
- if (node === fmt) break;
- let nodeAfeIdx = afe.indexOf(node);
- if (inner > 3 && nodeAfeIdx !== -1) {
- afeRemoveAt(nodeAfeIdx);
- if (nodeAfeIdx < bookmark) bookmark--;
- nodeAfeIdx = -1;
- }
- if (nodeAfeIdx === -1) {
- removeOpenAt(nodeIdx);
- continue;
- }
- // create clone
- const clone = mkEl(
- _tagNameOf(node),
- _namespaceOf(node),
- cloneAttrs(node),
- null
- );
- afe[nodeAfeIdx] = clone;
- open[nodeIdx] = clone;
- node = clone;
- if (lastNode === furthest) bookmark = nodeAfeIdx + 1;
- // append lastNode to node
- detach(lastNode);
- appendTo(node, lastNode);
- lastNode = node;
- }
- // insert lastNode into common ancestor (with foster parenting)
- detach(lastNode);
- const place = placeForCommonAncestor(commonAncestor);
- insertAtPlace(place, lastNode);
- // create element for fmt token, take children of furthest
- _printFromSource = true;
- const cloneFmt = mkEl(
- _tagNameOf(fmt),
- _namespaceOf(fmt),
- cloneAttrs(fmt),
- null
- );
- // Take all direct children of `furthest` (a template's content fragment
- // deliberately stays put — mirrors childrenOf-less spec behavior here). A
- // declarative shadow root is not one of them: the engine took it out of
- // the tree when it attached, so it stays with the host it hangs off.
- let k = _nodeFirstChildren[furthest];
- _nodeFirstChildren[furthest] = 0;
- _nodeLastChildren[furthest] = 0;
- while (k !== 0) {
- const next = _nodeNextSiblings[k];
- _nodeNextSiblings[k] = 0;
- _nodeParents[k] = 0;
- appendTo(
- (_nodeFlags[k] & FLAG_SHADOW_ROOT) === 0 ? cloneFmt : furthest,
- k
- );
- k = next;
- }
- appendTo(furthest, cloneFmt);
- // remove fmt from afe, insert clone at bookmark
- const curFmtIdx = afe.indexOf(fmt);
- if (curFmtIdx !== -1) {
- afeRemoveAt(curFmtIdx);
- if (curFmtIdx < bookmark) bookmark--;
- }
- afe.splice(bookmark, 0, cloneFmt);
- // remove fmt from open, insert clone below furthest
- const ofi = open.indexOf(fmt);
- if (ofi !== -1) removeOpenAt(ofi);
- const newFurthestIdx = open.indexOf(furthest);
- open.splice(newFurthestIdx + 1, 0, cloneFmt);
- _openSpliced = true;
- _openStackChanged = true;
- }
- return true;
- };
- const placeForCommonAncestor = (/** @type {HtmlElement} */ commonAncestor) => {
- if (
- TABLE_CONTEXT.has(_tagNameOf(commonAncestor)) &&
- _namespaceOf(commonAncestor) === NS_HTML
- ) {
- // The same place `appropriatePlace` fosters to, spelled as its own object:
- // the adoption agency holds this one while it moves nodes about, so the
- // shared one would be overwritten under it.
- const [parent, beforeNode] = _fosterPlace();
- return { parent, beforeNode };
- }
- return { parent: commonAncestor, beforeNode: 0 };
- };
- const detach = (/** @type {HtmlNodeRef} */ node) => {
- const p = _nodeParents[node];
- if (p === 0) return;
- let prev = 0;
- let c = _nodeFirstChildren[p];
- while (c !== 0 && c !== node) {
- prev = c;
- c = _nodeNextSiblings[c];
- }
- if (c === 0) return;
- if (prev === 0) _nodeFirstChildren[p] = _nodeNextSiblings[node];
- else _nodeNextSiblings[prev] = _nodeNextSiblings[node];
- if (_nodeLastChildren[p] === node) _nodeLastChildren[p] = prev;
- _nodeNextSiblings[node] = 0;
- _nodeParents[node] = 0;
- };
- // ---------- insertion modes ----------
- /** @type {Record<string, (t: Token) => void>} */
- const modes = {};
- // Dispatch the current insertion mode. An integer switch (cases ordered by
- // frequency) keeps each `modes.x(t)` call site monomorphic, where a keyed
- // `modes[mode]` load + indirect call would defeat inlining on the per-token
- // hot path.
- const runMode = (/** @type {Token} */ t) => {
- switch (mode) {
- case MODE_IN_BODY:
- return modes.inBody(t);
- case MODE_TEXT:
- return modes.text(t);
- case MODE_IN_CELL:
- return modes.inCell(t);
- case MODE_IN_ROW:
- return modes.inRow(t);
- case MODE_IN_TABLE_BODY:
- return modes.inTableBody(t);
- case MODE_IN_TABLE:
- return modes.inTable(t);
- case MODE_IN_TABLE_TEXT:
- return modes.inTableText(t);
- case MODE_IN_CAPTION:
- return modes.inCaption(t);
- case MODE_IN_COLUMN_GROUP:
- return modes.inColumnGroup(t);
- case MODE_IN_TEMPLATE:
- return modes.inTemplate(t);
- case MODE_IN_HEAD:
- return modes.inHead(t);
- case MODE_IN_HEAD_NOSCRIPT:
- return modes.inHeadNoscript(t);
- case MODE_AFTER_HEAD:
- return modes.afterHead(t);
- case MODE_BEFORE_HEAD:
- return modes.beforeHead(t);
- case MODE_BEFORE_HTML:
- return modes.beforeHtml(t);
- case MODE_INITIAL:
- return modes.initial(t);
- case MODE_AFTER_BODY:
- return modes.afterBody(t);
- case MODE_AFTER_AFTER_BODY:
- return modes.afterAfterBody(t);
- case MODE_IN_FRAMESET:
- return modes.inFrameset(t);
- case MODE_AFTER_FRAMESET:
- return modes.afterFrameset(t);
- // MODE_AFTER_AFTER_FRAMESET — every mode is enumerated, so the last
- // one is the `default` (also satisfies exhaustiveness linting).
- default:
- return modes.afterAfterFrameset(t);
- }
- };
- modes.initial = (t) => {
- if (t.type === TOKEN_CHAR) {
- const r = leadingWs(t, false);
- if (!r) return;
- quirks = true;
- mode = MODE_BEFORE_HTML;
- process(r);
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t, { parent: doc, beforeNode: 0 });
- return;
- }
- if (t.type === TOKEN_DOCTYPE) {
- // `skip.doctype` drops the node only; quirks detection below is unaffected.
- if (!skipDoctype) {
- const dt = _allocNode(NodeType.Doctype, t.start, t.end);
- _nodeStrings[dt] = t.name;
- // At most one doctype node is ever inserted (later doctype tokens
- // are ignored), so its ids live in two per-parse scalars.
- _doctypePublicId = t.publicId;
- _doctypeSystemId = t.systemId;
- _appendChild(doc, dt);
- }
- quirks = t.forceQuirks || isQuirky(t.name, t.publicId, t.systemId);
- mode = MODE_BEFORE_HTML;
- return;
- }
- quirks = true;
- mode = MODE_BEFORE_HTML;
- process(t);
- };
- modes.beforeHtml = (t) => {
- if (t.type === TOKEN_DOCTYPE) return;
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t, { parent: doc, beforeNode: 0 });
- return;
- }
- if (t.type === TOKEN_CHAR) {
- const r = leadingWs(t, false);
- if (!r) return;
- t = r;
- }
- if (t.type === TOKEN_START_TAG && t.name === "html") {
- const el = mkEl("html", NS_HTML, t.attrs, t.pos);
- _appendChild(doc, el);
- open.push(el);
- mode = MODE_BEFORE_HEAD;
- return;
- }
- if (t.type === TOKEN_END_TAG && !HEAD_BODY_HTML_BR.has(t.name)) {
- return;
- }
- const el = mkEl("html", NS_HTML, EMPTY_ATTRS, null);
- _appendChild(doc, el);
- open.push(el);
- mode = MODE_BEFORE_HEAD;
- process(t);
- };
- modes.beforeHead = (t) => {
- if (t.type === TOKEN_CHAR) {
- const r = leadingWs(t, false);
- if (!r) return;
- t = r;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_START_TAG && t.name === "head") {
- head = insertHtmlElement("head", t.attrs, t.pos);
- mode = MODE_IN_HEAD;
- return;
- }
- if (t.type === TOKEN_END_TAG && !HEAD_BODY_HTML_BR.has(t.name)) {
- return;
- }
- head = insertHtmlElement("head", EMPTY_ATTRS, null);
- mode = MODE_IN_HEAD;
- process(t);
- };
- modes.inHead = (t) => {
- if (t.type === TOKEN_CHAR) {
- const r = leadingWs(t, true);
- if (!r) return;
- popOpen();
- mode = MODE_AFTER_HEAD;
- process(r);
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG) {
- if (t.name === "html") return modes.inBody(t);
- if (HEAD_VOID_ELEMENTS.has(t.name)) {
- insertHtmlElement(t.name, t.attrs, t.pos);
- popOpen();
- return;
- }
- if (t.name === "title") {
- genericRcdata(t);
- return;
- }
- if (NOFRAMES_STYLE_NOSCRIPT.has(t.name)) {
- if (t.name === "noscript") {
- insertHtmlElement("noscript", t.attrs, t.pos);
- mode = MODE_IN_HEAD_NOSCRIPT;
- return;
- }
- genericRawtext(t);
- return;
- }
- if (t.name === "script") {
- genericRawtext(t);
- return;
- }
- if (t.name === "template") {
- insertHtmlElement("template", t.attrs, t.pos);
- insertMarker();
- framesetOk = false;
- mode = MODE_IN_TEMPLATE;
- templateModes.push(MODE_IN_TEMPLATE);
- const el = cur();
- const fragment = _allocNode(NodeType.DocumentFragment, 0, 0);
- _nodeStrings[fragment] = "";
- // Parent link so the iterative walk can ascend out of the content.
- _nodeParents[fragment] = el;
- _nodeContentEnds[el] = fragment;
- _nodeFlags[el] |= FLAG_HAS_TEMPLATE;
- // The engine attaches this content to the parent as a shadow root and
- // drops the template from the tree; the one thing that leaves behind is
- // that the tree construction can no longer move it away from its host.
- const host = _nodeParents[el];
- const shadowRootMode = _findAttr(
- t.attrs.start,
- t.attrs.count,
- "shadowrootmode"
- );
- if (
- shadowRootMode !== 0 &&
- host !== 0 &&
- host !== open[0] &&
- (_nodeFlags[host] & FLAG_SHADOW_HOST) === 0 &&
- _nodeTypes[host] === NodeType.Element &&
- _isShadowHost(host)
- ) {
- const mode = _attrValueOf(shadowRootMode).toLowerCase();
- if (mode === "open" || mode === "closed") {
- _nodeFlags[el] |= FLAG_SHADOW_ROOT;
- _nodeFlags[host] |= FLAG_SHADOW_HOST;
- }
- }
- return;
- }
- if (t.name === "head") return;
- }
- if (t.type === TOKEN_END_TAG) {
- if (t.name === "head") {
- popOpen();
- mode = MODE_AFTER_HEAD;
- return;
- }
- if (BODY_HTML_BR.has(t.name)) {
- /* fallthrough */
- } else if (t.name === "template") {
- if (!open.some(isHtmlTemplateEl)) {
- return;
- }
- generateImpliedEndTagsThorough();
- popUntil("template");
- clearAfeToMarker();
- templateModes.pop();
- resetInsertionMode();
- return;
- } else {
- return;
- }
- }
- // anything else
- popOpen();
- mode = MODE_AFTER_HEAD;
- process(t);
- };
- modes.inHeadNoscript = (t) => {
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_END_TAG && t.name === "noscript") {
- popOpen();
- mode = MODE_IN_HEAD;
- return;
- }
- if (t.type === TOKEN_CHAR && isAllWs(t.data)) return modes.inHead(t);
- if (isCommentOrProcessingInstruction(t)) return modes.inHead(t);
- if (t.type === TOKEN_START_TAG && IN_HEAD_NOSCRIPT_PASSTHROUGH.has(t.name)) {
- return modes.inHead(t);
- }
- // A stray end tag other than </br>/</noscript> is ignored (the comment
- // or content stays inside <noscript>); only </br> and other content fall
- // back to popping <noscript>.
- if (t.type === TOKEN_END_TAG && t.name !== "br") return;
- if (
- t.type === TOKEN_START_TAG &&
- (t.name === "head" || t.name === "noscript")
- ) {
- return;
- }
- popOpen();
- mode = MODE_IN_HEAD;
- process(t);
- };
- modes.afterHead = (t) => {
- if (t.type === TOKEN_CHAR) {
- const r = leadingWs(t, true);
- if (!r) return;
- insertHtmlElement("body", EMPTY_ATTRS, null);
- mode = MODE_IN_BODY;
- process(r);
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG) {
- if (t.name === "html") return modes.inBody(t);
- if (t.name === "body") {
- insertHtmlElement("body", t.attrs, t.pos);
- framesetOk = false;
- mode = MODE_IN_BODY;
- return;
- }
- if (t.name === "frameset") {
- insertHtmlElement("frameset", t.attrs, t.pos);
- mode = MODE_IN_FRAMESET;
- return;
- }
- if (HEAD_ELEMENTS.has(t.name)) {
- const headEl = /** @type {HtmlElement} */ (head);
- open.push(headEl);
- modes.inHead(t);
- const idx = open.indexOf(headEl);
- if (idx !== -1) removeOpenAt(idx);
- return;
- }
- if (t.name === "head") return;
- }
- if (t.type === TOKEN_END_TAG) {
- if (t.name === "template") return modes.inHead(t);
- if (!BODY_HTML_BR.has(t.name)) return;
- }
- insertHtmlElement("body", EMPTY_ATTRS, null);
- mode = MODE_IN_BODY;
- process(t);
- };
- modes.inBody = (t) => {
- if (t.type === TOKEN_CHAR) {
- // Deliberately not guarded by `inputHasNul`: this `includes` flattens the
- // decoded rope, and skipping it makes the inserts below allocate more.
- if (t.data.includes("\0")) t = { ...t, data: t.data.replace(/\0/g, "") };
- if (t.data === "") return;
- reconstructAfe();
- insertCharacters(t.data, t.start, t.end);
- // `framesetOk` only ever goes true→false, so once it's false skip the
- // per-text-token whitespace scan entirely (it flips false very early in
- // real documents).
- if (framesetOk && !isAllWs(t.data)) framesetOk = false;
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG) return startTagInBody(t);
- if (t.type === TOKEN_END_TAG) return endTagInBody(t);
- if (t.type === TOKEN_EOF && templateModes.length) {
- return modes.inTemplate(t);
- }
- };
- const closeIfPInButtonScope = () => {
- if (inButtonScope("p")) closePElement();
- };
- // "any other end tag" in body: pop to the matching open element, stopping at
- // the first special element; also the adoption agency's no-element fallback.
- const anyOtherEndTag = (/** @type {string} */ name) => {
- for (let i = open.length - 1; i >= 0; i--) {
- const node = open[i];
- if (_namespaceOf(node) === NS_HTML && _tagNameOf(node) === name) {
- generateImpliedEndTags(name);
- while (open.length > i) {
- _nodeEnds[open[open.length - 1]] = tokenEnd;
- popOpen();
- }
- return;
- }
- if (isSpecial(node)) return;
- }
- };
- const startTagInBody = (/** @type {StartTagToken} */ t) => {
- const name = t.name;
- if (name === "html") {
- if (open.some(isHtmlTemplateEl)) {
- return;
- }
- mergeAttrs(open[0], t.attrs);
- return;
- }
- if (HEAD_ELEMENTS.has(name)) {
- return modes.inHead(t);
- }
- if (name === "body") {
- const second = open[1];
- if (
- !second ||
- _tagNameOf(second) !== "body" ||
- open.some(isHtmlTemplateEl)
- ) {
- return;
- }
- framesetOk = false;
- mergeAttrs(second, t.attrs);
- return;
- }
- if (name === "frameset") {
- const second = open[1];
- if (!second || _tagNameOf(second) !== "body") return;
- if (!framesetOk) return;
- detach(second);
- while (open.length > 1) popOpen();
- insertHtmlElement("frameset", t.attrs, t.pos);
- mode = MODE_IN_FRAMESET;
- return;
- }
- if (BLOCK_START.has(name)) {
- closeIfPInButtonScope();
- insertHtmlElement(name, t.attrs, t.pos);
- return;
- }
- if (HEADING.has(name)) {
- closeIfPInButtonScope();
- if (_namespaceOf(cur()) === NS_HTML && HEADING.has(_tagNameOf(cur()))) {
- popOpen();
- }
- insertHtmlElement(name, t.attrs, t.pos);
- return;
- }
- if (name === "pre" || name === "listing") {
- closeIfPInButtonScope();
- insertHtmlElement(name, t.attrs, t.pos);
- t.swallowNewline = true;
- framesetOk = false;
- return;
- }
- if (name === "form") {
- if (form && !open.some(isHtmlTemplateEl)) {
- return;
- }
- closeIfPInButtonScope();
- // The end tag this needs generates implied end tags where the source's did
- // not, so anything open between the two forms moves.
- if (_formResetWouldPop()) _printFromSource = true;
- const el = insertHtmlElement("form", t.attrs, t.pos);
- if (!open.some(isHtmlTemplateEl)) {
- form = el;
- }
- return;
- }
- if (name === "li") {
- framesetOk = false;
- for (let i = open.length - 1; i >= 0; i--) {
- const node = open[i];
- if (_namespaceOf(node) === NS_HTML && _tagNameOf(node) === "li") {
- generateImpliedEndTags("li");
- popUntil("li");
- break;
- }
- if (
- isSpecial(node) &&
- !(_namespaceOf(node) === NS_HTML && ADDRESS_DIV_P.has(_tagNameOf(node)))
- ) {
- // Inlined rather than a call: this runs for every `<li>` in a list.
- for (let j = i - 1; j >= 0 && !_printFromSource; j--) {
- const below = open[j];
- if (_namespaceOf(below) === NS_HTML && _tagNameOf(below) === "li") {
- _printFromSource = true;
- }
- }
- break;
- }
- }
- closeIfPInButtonScope();
- insertHtmlElement("li", t.attrs, t.pos);
- return;
- }
- if (name === "dd" || name === "dt") {
- framesetOk = false;
- for (let i = open.length - 1; i >= 0; i--) {
- const node = open[i];
- if (
- _namespaceOf(node) === NS_HTML &&
- (_tagNameOf(node) === "dd" || _tagNameOf(node) === "dt")
- ) {
- generateImpliedEndTags(_tagNameOf(node));
- popUntil(_tagNameOf(node));
- break;
- }
- if (
- isSpecial(node) &&
- !(_namespaceOf(node) === NS_HTML && ADDRESS_DIV_P.has(_tagNameOf(node)))
- ) {
- for (let j = i - 1; j >= 0 && !_printFromSource; j--) {
- const below = open[j];
- if (_namespaceOf(below) !== NS_HTML) continue;
- const bn = _tagNameOf(below);
- if (bn === "dd" || bn === "dt") _printFromSource = true;
- }
- break;
- }
- }
- closeIfPInButtonScope();
- insertHtmlElement(name, t.attrs, t.pos);
- return;
- }
- if (name === "plaintext") {
- closeIfPInButtonScope();
- insertHtmlElement("plaintext", t.attrs, t.pos);
- return;
- }
- if (name === "button") {
- if (inScope("button")) {
- generateImpliedEndTags();
- popUntil("button");
- }
- reconstructAfe();
- insertHtmlElement("button", t.attrs, t.pos);
- framesetOk = false;
- return;
- }
- if (name === "a") {
- // if there's an <a> in afe after last marker
- for (let i = afe.length - 1; i >= 0; i--) {
- if (afe[i] === AFE_MARKER) {
- if (_afeNameBelowMarker(i, "a")) _printFromSource = true;
- break;
- }
- if (_tagNameOf(afe[i]) === "a") {
- adoptionAgency("a", t.pos);
- const idx = afe.findIndex(
- (e) => e !== AFE_MARKER && _tagNameOf(e) === "a"
- );
- if (idx !== -1) {
- const el = afe[idx];
- afeRemoveAt(idx);
- const oi = open.indexOf(el);
- if (oi !== -1) removeOpenAt(oi);
- }
- break;
- }
- }
- reconstructAfe();
- const el = insertHtmlElement("a", t.attrs, t.pos);
- pushAfe(el);
- return;
- }
- if (FORMATTING.has(name) && name !== "a" && name !== "nobr") {
- reconstructAfe();
- const el = insertHtmlElement(name, t.attrs, t.pos);
- pushAfe(el);
- return;
- }
- if (name === "nobr") {
- reconstructAfe();
- if (inScope("nobr")) {
- // The adoption agency returns false when a marker shields the nobr
- // from the active formatting list; then act as "any other end tag".
- if (!adoptionAgency("nobr", t.pos)) anyOtherEndTag("nobr");
- reconstructAfe();
- }
- const el = insertHtmlElement("nobr", t.attrs, t.pos);
- pushAfe(el);
- return;
- }
- if (APPLET_MARQUEE_OBJECT.has(name)) {
- reconstructAfe();
- insertHtmlElement(name, t.attrs, t.pos);
- insertMarker();
- framesetOk = false;
- return;
- }
- if (name === "table") {
- if (!quirks) closeIfPInButtonScope();
- insertHtmlElement("table", t.attrs, t.pos);
- framesetOk = false;
- mode = MODE_IN_TABLE;
- return;
- }
- if (VOID_FORMATTING.has(name)) {
- reconstructAfe();
- insertHtmlElement(name, t.attrs, t.pos);
- popOpen();
- framesetOk = false;
- return;
- }
- if (name === "input") {
- // `<input>` inside a select is dropped; if a select is open it is
- // closed first (keygen/textarea no longer behave this way).
- if (inScope("select")) {
- popUntil("select");
- resetInsertionMode();
- } else if (
- fragment &&
- _namespaceOf(fragment) === NS_HTML &&
- _tagNameOf(fragment) === "select"
- ) {
- return;
- }
- reconstructAfe();
- insertHtmlElement("input", t.attrs, t.pos);
- popOpen();
- const ty = _findAttr(t.attrs.start, t.attrs.count, "type");
- if (ty === 0 || _attrValueOf(ty).toLowerCase() !== "hidden") {
- framesetOk = false;
- }
- return;
- }
- if (PARAM_SOURCE_TRACK.has(name)) {
- insertHtmlElement(name, t.attrs, t.pos);
- popOpen();
- return;
- }
- if (name === "hr") {
- closeIfPInButtonScope();
- // An `<hr>` is a `<select>`'s own separator, so inside one it leaves what
- // it was written in; outside one it is ordinary content and nests.
- if (inScope("select")) generateImpliedEndTags();
- insertHtmlElement("hr", t.attrs, t.pos);
- popOpen();
- framesetOk = false;
- return;
- }
- if (name === "image") {
- return startTagInBody({ ...t, name: "img" });
- }
- if (name === "textarea") {
- genericRcdata(t, true);
- framesetOk = false;
- return;
- }
- if (name === "xmp") {
- closeIfPInButtonScope();
- reconstructAfe();
- framesetOk = false;
- genericRawtext(t);
- return;
- }
- if (name === "iframe") {
- framesetOk = false;
- genericRawtext(t);
- return;
- }
- if (name === "noembed") {
- genericRawtext(t);
- return;
- }
- if (name === "select") {
- reconstructAfe();
- if (inScope("select")) {
- generateImpliedEndTags();
- popUntil("select");
- resetInsertionMode();
- return;
- }
- insertHtmlElement("select", t.attrs, t.pos);
- framesetOk = false;
- return;
- }
- if (name === "optgroup" || name === "option") {
- // Inside a select these end what is open like a list item does, an
- // `<option>` sparing its `<optgroup>`; outside one only an option gives way.
- if (inScope("select")) {
- generateImpliedEndTags(name === "option" ? "optgroup" : "");
- } else if (
- _namespaceOf(cur()) === NS_HTML &&
- _tagNameOf(cur()) === "option"
- ) {
- popOpen();
- }
- reconstructAfe();
- insertHtmlElement(name, t.attrs, t.pos);
- return;
- }
- if (name === "rb" || name === "rtc") {
- if (inScope("ruby")) generateImpliedEndTags();
- insertHtmlElement(name, t.attrs, t.pos);
- return;
- }
- if (name === "rp" || name === "rt") {
- if (inScope("ruby")) generateImpliedEndTags("rtc");
- insertHtmlElement(name, t.attrs, t.pos);
- return;
- }
- if (name === "math") {
- reconstructAfe();
- const attrs = adjustForeignAttrs(adjustMathmlAttrs(t.attrs), NS_MATHML);
- insertForeignElement("math", NS_MATHML, attrs, t.pos);
- if (t.selfClosing) popOpen();
- return;
- }
- if (name === "svg") {
- reconstructAfe();
- const attrs = adjustForeignAttrs(t.attrs, NS_SVG);
- insertForeignElement("svg", NS_SVG, attrs, t.pos);
- if (t.selfClosing) popOpen();
- return;
- }
- if (IGNORED_BODY_TABLE_STARTS.has(name)) {
- return;
- }
- // any other start tag
- reconstructAfe();
- insertHtmlElement(name, t.attrs, t.pos);
- };
- const endTagInBody = (/** @type {EndTagToken} */ t) => {
- const name = t.name;
- if (name === "template") return modes.inHead(t);
- if (name === "select") {
- if (!inScope("select")) return;
- generateImpliedEndTags();
- popUntil("select");
- return;
- }
- if (name === "body" || name === "html") {
- if (!inScope("body")) return;
- mode = MODE_AFTER_BODY;
- if (name === "html") process(t);
- return;
- }
- if (BLOCK_END.has(name)) {
- if (!inScope(name)) return;
- generateImpliedEndTags();
- popUntil(name);
- return;
- }
- if (name === "form") {
- if (!open.some(isHtmlTemplateEl)) {
- const node = form;
- form = 0;
- if (!node || !inScopeEl(node)) return;
- generateImpliedEndTags();
- const idx = open.indexOf(node);
- if (idx !== -1) removeOpenAt(idx);
- } else {
- if (!inScope("form")) return;
- generateImpliedEndTags();
- popUntil("form");
- }
- return;
- }
- if (name === "p") {
- if (!inButtonScope("p")) insertHtmlElement("p", EMPTY_ATTRS, t.pos);
- closePElement();
- return;
- }
- if (name === "li") {
- if (!inListItemScope("li")) return;
- generateImpliedEndTags("li");
- popUntil("li");
- return;
- }
- if (name === "dd" || name === "dt") {
- if (!inScope(name)) return;
- generateImpliedEndTags(name);
- popUntil(name);
- return;
- }
- if (HEADING.has(name)) {
- let anyHeadingInScope = false;
- for (const h of HEADING) {
- if (inScope(h)) {
- anyHeadingInScope = true;
- break;
- }
- }
- if (!anyHeadingInScope) return;
- generateImpliedEndTags();
- popUntilOneOf(HEADING);
- return;
- }
- if (name === "sarcasm") {
- /* take a deep breath */
- }
- if (FORMATTING.has(name)) {
- // With no such element in the list, the algorithm hands the token back
- // (§13.2.6.4.7, "any other end tag") rather than swallowing it — which is
- // what Noah's Ark leaves behind: an element still open that the list no
- // longer names.
- if (!adoptionAgency(name, t.pos)) anyOtherEndTag(name);
- return;
- }
- if (APPLET_MARQUEE_OBJECT.has(name)) {
- if (!inScope(name)) return;
- generateImpliedEndTags();
- popUntil(name);
- clearAfeToMarker();
- return;
- }
- if (name === "br") {
- reconstructAfe();
- insertHtmlElement("br", EMPTY_ATTRS, t.pos);
- popOpen();
- framesetOk = false;
- return;
- }
- anyOtherEndTag(name);
- };
- // generic RCDATA/RAWTEXT: tokenizer already emits the text + end tag, so we
- // just insert the element and switch to "text" mode; text mode appends chars
- // and the matching end tag pops.
- const genericRawtext = (/** @type {StartTagToken} */ t) => {
- insertHtmlElement(t.name, t.attrs, t.pos);
- originalMode = mode;
- mode = MODE_TEXT;
- };
- const genericRcdata = (/** @type {StartTagToken} */ t, swallow = false) => {
- insertHtmlElement(t.name, t.attrs, t.pos);
- if (swallow) t.swallowNewline = true;
- originalMode = mode;
- mode = MODE_TEXT;
- };
- modes.text = (t) => {
- if (t.type === TOKEN_CHAR) {
- insertCharacters(t.data, t.start, t.end);
- return;
- }
- if (t.type === TOKEN_EOF) {
- if (open.length) popOpen();
- mode = originalMode;
- process(t);
- return;
- }
- if (t.type === TOKEN_END_TAG) {
- // Treat as rawtext rather than an end tag when it can't close the
- // current element: a non-matching name (e.g. a fragment context), or
- // a name that ran straight to EOF with no delimiter (`</script` at
- // EOF — the tokenizer still emits a partial tag there).
- if (
- cur() &&
- (t.name !== _tagNameOf(cur()) || t.pos.end === t.pos.nameEnd)
- ) {
- insertCharacters(
- source.slice(t.pos.start, t.pos.end),
- t.pos.start,
- t.pos.end
- );
- return;
- }
- // span the close tag, like every other explicit-close pop
- _nodeEnds[/** @type {HtmlElement} */ (popOpen())] = tokenEnd;
- mode = originalMode;
- }
- };
- // ---------- table modes ----------
- /** @type {{ list: CharToken[], hasNonWs: boolean } | null} */
- let pendingTableChars = null;
- modes.inTable = (t) => {
- if (t.type === TOKEN_CHAR) {
- const c = cur();
- if (TABLE_CONTEXT.has(_tagNameOf(c)) && _namespaceOf(c) === NS_HTML) {
- pendingTableChars = { list: [], hasNonWs: false };
- originalMode = mode;
- mode = MODE_IN_TABLE_TEXT;
- return process(t);
- }
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG) {
- const name = t.name;
- if (name === "caption") {
- clearStackToTableContext();
- insertMarker();
- insertHtmlElement("caption", t.attrs, t.pos);
- mode = MODE_IN_CAPTION;
- return;
- }
- if (name === "colgroup") {
- clearStackToTableContext();
- insertHtmlElement("colgroup", t.attrs, t.pos);
- mode = MODE_IN_COLUMN_GROUP;
- return;
- }
- if (name === "col") {
- clearStackToTableContext();
- insertHtmlElement("colgroup", EMPTY_ATTRS, t.pos);
- mode = MODE_IN_COLUMN_GROUP;
- return process(t);
- }
- if (TBODY_GROUP.has(name)) {
- clearStackToTableContext();
- insertHtmlElement(name, t.attrs, t.pos);
- mode = MODE_IN_TABLE_BODY;
- return;
- }
- if (TD_TH_TR.has(name)) {
- clearStackToTableContext();
- insertHtmlElement("tbody", EMPTY_ATTRS, t.pos);
- mode = MODE_IN_TABLE_BODY;
- return process(t);
- }
- if (name === "table") {
- if (!inTableScope("table")) return;
- popUntil("table");
- resetInsertionMode();
- return process(t);
- }
- if (STYLE_SCRIPT_TEMPLATE.has(name)) {
- return modes.inHead(t);
- }
- if (name === "input") {
- const ty = _findAttr(t.attrs.start, t.attrs.count, "type");
- if (ty !== 0 && _attrValueOf(ty).toLowerCase() === "hidden") {
- insertHtmlElement("input", t.attrs, t.pos);
- popOpen();
- return;
- }
- }
- if (name === "form") {
- if (form || open.some(isHtmlTemplateEl)) {
- return;
- }
- form = insertHtmlElement("form", t.attrs, t.pos);
- popOpen();
- return;
- }
- }
- if (t.type === TOKEN_END_TAG) {
- if (t.name === "table") {
- if (!inTableScope("table")) return;
- popUntil("table");
- resetInsertionMode();
- return;
- }
- if (IN_TABLE_IGNORED_ENDS.has(t.name)) {
- return;
- }
- if (t.name === "template") return modes.inHead(t);
- }
- if (t.type === TOKEN_EOF) return modes.inBody(t);
- // anything else: foster parenting
- fosterParenting = true;
- modes.inBody(t);
- fosterParenting = false;
- };
- modes.inTableText = (t) => {
- if (t.type === TOKEN_CHAR) {
- const data =
- inputHasNul && t.data.includes("\0") ? t.data.replace(/\0/g, "") : t.data;
- if (data === "") return;
- // Snapshot into a fresh token: these are buffered and replayed after
- // later tokens arrive, so they must not alias the reused token.
- /** @type {CharToken} */
- const tc = { type: TOKEN_CHAR, data, start: t.start, end: t.end };
- const pending = /** @type {{ list: CharToken[], hasNonWs: boolean }} */ (
- pendingTableChars
- );
- pending.list.push(tc);
- if (!isAllWs(tc.data)) pending.hasNonWs = true;
- return;
- }
- // flush
- const chars = /** @type {{ list: CharToken[], hasNonWs: boolean }} */ (
- pendingTableChars
- );
- pendingTableChars = null;
- mode = originalMode;
- for (const ct of chars.list) {
- if (chars.hasNonWs) {
- fosterParenting = true;
- modes.inBody(ct);
- fosterParenting = false;
- } else {
- insertCharacters(ct.data, ct.start, ct.end);
- }
- }
- process(t);
- };
- const clearStackToTableContext = () => {
- while (open.length) {
- const c = cur();
- if (_namespaceOf(c) === NS_HTML && CLEAR_TABLE.has(_tagNameOf(c))) {
- break;
- }
- popOpen();
- }
- };
- const clearStackToTableBodyContext = () => {
- while (open.length) {
- const c = cur();
- if (_namespaceOf(c) === NS_HTML && CLEAR_TABLE_BODY.has(_tagNameOf(c))) {
- break;
- }
- popOpen();
- }
- };
- const clearStackToTableRowContext = () => {
- while (open.length) {
- const c = cur();
- if (_namespaceOf(c) === NS_HTML && CLEAR_TABLE_ROW.has(_tagNameOf(c))) {
- break;
- }
- popOpen();
- }
- };
- modes.inCaption = (t) => {
- if (
- (t.type === TOKEN_END_TAG && t.name === "caption") ||
- (t.type === TOKEN_START_TAG && CAPTION_TABLE_STARTS.has(t.name)) ||
- (t.type === TOKEN_END_TAG && t.name === "table")
- ) {
- if (!inTableScope("caption")) return;
- generateImpliedEndTags();
- popUntil("caption");
- clearAfeToMarker();
- mode = MODE_IN_TABLE;
- if (!(t.type === TOKEN_END_TAG && t.name === "caption")) {
- return process(t);
- }
- return;
- }
- if (t.type === TOKEN_END_TAG && CAPTION_IGNORED_ENDS.has(t.name)) {
- return;
- }
- return modes.inBody(t);
- };
- modes.inColumnGroup = (t) => {
- if (t.type === TOKEN_CHAR) {
- const r = leadingWs(t, true);
- if (!r) return;
- if (_tagNameOf(cur()) !== "colgroup") return;
- popOpen();
- mode = MODE_IN_TABLE;
- process(r);
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_START_TAG && t.name === "col") {
- insertHtmlElement("col", t.attrs, t.pos);
- popOpen();
- return;
- }
- if (t.type === TOKEN_END_TAG && t.name === "colgroup") {
- if (_tagNameOf(cur()) !== "colgroup") return;
- popOpen();
- mode = MODE_IN_TABLE;
- return;
- }
- if (t.type === TOKEN_END_TAG && t.name === "col") return;
- if (
- (t.type === TOKEN_START_TAG || t.type === TOKEN_END_TAG) &&
- t.name === "template"
- ) {
- return modes.inHead(t);
- }
- if (t.type === TOKEN_EOF) return modes.inBody(t);
- if (_tagNameOf(cur()) !== "colgroup") return;
- popOpen();
- mode = MODE_IN_TABLE;
- process(t);
- };
- modes.inTableBody = (t) => {
- if (t.type === TOKEN_START_TAG && t.name === "tr") {
- clearStackToTableBodyContext();
- insertHtmlElement("tr", t.attrs, t.pos);
- mode = MODE_IN_ROW;
- return;
- }
- if (t.type === TOKEN_START_TAG && (t.name === "th" || t.name === "td")) {
- clearStackToTableBodyContext();
- insertHtmlElement("tr", EMPTY_ATTRS, t.pos);
- mode = MODE_IN_ROW;
- return process(t);
- }
- if (t.type === TOKEN_END_TAG && TBODY_GROUP.has(t.name)) {
- if (!inTableScope(t.name)) return;
- clearStackToTableBodyContext();
- popOpen();
- mode = MODE_IN_TABLE;
- return;
- }
- if (
- (t.type === TOKEN_START_TAG && TBODY_TRIGGER_STARTS.has(t.name)) ||
- (t.type === TOKEN_END_TAG && t.name === "table")
- ) {
- if (!inTableScope(TBODY_GROUP)) return;
- clearStackToTableBodyContext();
- popOpen();
- mode = MODE_IN_TABLE;
- return process(t);
- }
- if (t.type === TOKEN_END_TAG && TBODY_IGNORED_ENDS.has(t.name)) {
- return;
- }
- return modes.inTable(t);
- };
- modes.inRow = (t) => {
- if (t.type === TOKEN_START_TAG && (t.name === "th" || t.name === "td")) {
- clearStackToTableRowContext();
- insertHtmlElement(t.name, t.attrs, t.pos);
- mode = MODE_IN_CELL;
- insertMarker();
- return;
- }
- if (t.type === TOKEN_END_TAG && t.name === "tr") {
- if (!inTableScope("tr")) return;
- clearStackToTableRowContext();
- popOpen();
- mode = MODE_IN_TABLE_BODY;
- return;
- }
- if (
- (t.type === TOKEN_START_TAG && ROW_TRIGGER_STARTS.has(t.name)) ||
- (t.type === TOKEN_END_TAG && t.name === "table")
- ) {
- if (!inTableScope("tr")) return;
- clearStackToTableRowContext();
- popOpen();
- mode = MODE_IN_TABLE_BODY;
- return process(t);
- }
- if (t.type === TOKEN_END_TAG && TBODY_GROUP.has(t.name)) {
- if (!inTableScope(t.name)) return;
- if (!inTableScope("tr")) return;
- clearStackToTableRowContext();
- popOpen();
- mode = MODE_IN_TABLE_BODY;
- return process(t);
- }
- if (t.type === TOKEN_END_TAG && ROW_IGNORED_ENDS.has(t.name)) {
- return;
- }
- return modes.inTable(t);
- };
- modes.inCell = (t) => {
- if (t.type === TOKEN_END_TAG && (t.name === "td" || t.name === "th")) {
- if (!inTableScope(t.name)) return;
- generateImpliedEndTags();
- popUntil(t.name);
- clearAfeToMarker();
- mode = MODE_IN_ROW;
- return;
- }
- if (t.type === TOKEN_START_TAG && CAPTION_TABLE_STARTS.has(t.name)) {
- if (!inTableScope("td") && !inTableScope("th")) return;
- closeCell();
- return process(t);
- }
- if (t.type === TOKEN_END_TAG && TABLE_CONTEXT.has(t.name)) {
- if (!inTableScope(t.name)) return;
- closeCell();
- return process(t);
- }
- if (t.type === TOKEN_END_TAG && CELL_IGNORED_ENDS.has(t.name)) {
- return;
- }
- return modes.inBody(t);
- };
- const closeCell = () => {
- generateImpliedEndTags();
- popUntilOneOf(TD_TH);
- clearAfeToMarker();
- mode = MODE_IN_ROW;
- };
- modes.inTemplate = (t) => {
- if (
- t.type === TOKEN_CHAR ||
- isCommentOrProcessingInstruction(t) ||
- t.type === TOKEN_DOCTYPE
- ) {
- return modes.inBody(t);
- }
- if (t.type === TOKEN_START_TAG) {
- if (HEAD_ELEMENTS.has(t.name)) {
- return modes.inHead(t);
- }
- const target = TEMPLATE_START_TAG_MODES.get(t.name) || MODE_IN_BODY;
- templateModes[templateModes.length - 1] = target;
- mode = target;
- return process(t);
- }
- if (t.type === TOKEN_END_TAG) {
- if (t.name === "template") return modes.inHead(t);
- return;
- }
- if (t.type === TOKEN_EOF) {
- if (!open.some(isHtmlTemplateEl)) {
- return;
- }
- popUntil("template");
- clearAfeToMarker();
- templateModes.pop();
- resetInsertionMode();
- return process(t);
- }
- };
- modes.afterBody = (t) => {
- if (t.type === TOKEN_CHAR && isAllWs(t.data)) return modes.inBody(t);
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t, {
- parent: open[0],
- beforeNode: 0
- });
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_END_TAG && t.name === "html") {
- if (fragment) return;
- mode = MODE_AFTER_AFTER_BODY;
- return;
- }
- if (t.type === TOKEN_EOF) return;
- mode = MODE_IN_BODY;
- process(t);
- };
- modes.inFrameset = (t) => {
- if (t.type === TOKEN_CHAR) {
- const ws = t.data.replace(/[^\t\n\f\r ]/g, "");
- if (ws) insertCharacters(ws, t.start, t.end);
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_START_TAG && t.name === "frameset") {
- insertHtmlElement("frameset", t.attrs, t.pos);
- return;
- }
- if (t.type === TOKEN_END_TAG && t.name === "frameset") {
- if (_tagNameOf(cur()) === "html") return;
- popOpen();
- if (!fragment && _tagNameOf(cur()) !== "frameset") {
- mode = MODE_AFTER_FRAMESET;
- }
- return;
- }
- if (t.type === TOKEN_START_TAG && t.name === "frame") {
- insertHtmlElement("frame", t.attrs, t.pos);
- popOpen();
- return;
- }
- if (t.type === TOKEN_START_TAG && t.name === "noframes") {
- return modes.inHead(t);
- }
- };
- modes.afterFrameset = (t) => {
- if (t.type === TOKEN_CHAR) {
- const ws = t.data.replace(/[^\t\n\f\r ]/g, "");
- if (ws) insertCharacters(ws, t.start, t.end);
- return;
- }
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t);
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return;
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_END_TAG && t.name === "html") {
- mode = MODE_AFTER_AFTER_FRAMESET;
- return;
- }
- if (t.type === TOKEN_START_TAG && t.name === "noframes") {
- return modes.inHead(t);
- }
- };
- modes.afterAfterBody = (t) => {
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t, { parent: doc, beforeNode: 0 });
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return modes.inBody(t);
- if (t.type === TOKEN_CHAR && isAllWs(t.data)) return modes.inBody(t);
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_EOF) return;
- mode = MODE_IN_BODY;
- process(t);
- };
- modes.afterAfterFrameset = (t) => {
- if (isCommentOrProcessingInstruction(t)) {
- insertCommentOrProcessingInstruction(t, { parent: doc, beforeNode: 0 });
- return;
- }
- if (t.type === TOKEN_DOCTYPE) return modes.inBody(t);
- if (t.type === TOKEN_CHAR && isAllWs(t.data)) return modes.inBody(t);
- if (t.type === TOKEN_START_TAG && t.name === "html") return modes.inBody(t);
- if (t.type === TOKEN_START_TAG && t.name === "noframes") {
- return modes.inHead(t);
- }
- };
- const dispatch = () => {
- process(/** @type {Token} */ (/** @type {unknown} */ (tok)));
- // One boolean per token: only a stack change can complete a subtree.
- if (_streamEligible && (_openStackChanged || _openSpliced)) {
- _openStackChanged = false;
- if (_streaming) {
- _streamSync();
- } else if (_nodeCount >= _STREAM_MIN_NODES) {
- // Big enough to be worth recycling: start streaming here. The tree so
- // far is unvisited, so the first sync enters the open stack and flushes
- // everything already finished under it.
- _streaming = true;
- _streamSync();
- }
- }
- };
- // Tokenizer callbacks, hoisted to module scope and closing over the module-
- // scope parser state. `tokenize`'s indirect `callbacks.text(...)` / `openTag`
- // / `attribute(...)` sites then see one stable function identity across every
- // parse, staying monomorphic so V8 can inline them — a fresh per-parse closure
- // made them megamorphic and blocked inlining. `parseHtml` only refreshes
- // `fragmentContext` on the shared object before each tokenize.
- /** @type {() => boolean} */
- const currentNodeIsForeign = () => {
- const adjustedCurrentNode = adjustedCurrent();
- return (
- open.length > 0 &&
- adjustedCurrentNode !== 0 &&
- _namespaceOf(adjustedCurrentNode) !== NS_HTML
- );
- };
- /**
- * Whether the tree builder skips the tokenizer switch for this start tag, so
- * its content stays markup: foreign content, and every frameset-mode tag but
- * `noframes`, which those modes ignore rather than hand to the head rules.
- * @param {string} name the start tag's lowercased name
- * @returns {boolean} true when the tokenizer stays in the data state
- */
- const vetoesContentMode = (name) => {
- if (currentNodeIsForeign()) return true;
- return (
- name !== "noframes" &&
- (mode === MODE_IN_FRAMESET ||
- mode === MODE_AFTER_FRAMESET ||
- mode === MODE_AFTER_AFTER_FRAMESET)
- );
- };
- /** @type {(input: string, code: string) => void} */
- const handleParseError = (input, code) => {
- if (code === "eof-in-tag") eofInTag = true;
- };
- /**
- * @param {string} input input
- * @param {number} rangeStart start of the range
- * @param {number} rangeEnd end of the range
- * @returns {string} the range with NULLs replaced, per the DOCTYPE states
- */
- const doctypePart = (input, rangeStart, rangeEnd) => {
- const raw = input.slice(rangeStart, rangeEnd);
- return inputHasNul ? raw.replace(/\0/g, "\uFFFD") : raw;
- };
- /** @type {(input: string, start: number, end: number, nameStart: number, nameEnd: number, publicStart: number, publicEnd: number, systemStart: number, systemEnd: number, forceQuirks: boolean) => number} */
- const handleDoctypeToken = (
- input,
- start,
- end,
- nameStart,
- nameEnd,
- publicStart,
- publicEnd,
- systemStart,
- systemEnd,
- forceQuirks
- ) => {
- // Only the token immediately after the start tag may be the swallowed
- // newline; anything else clears the flag (§13.2.6.4.7 "the next token").
- swallowNextNewline = false;
- tok.type = TOKEN_DOCTYPE;
- // A missing name and an empty one are the same to every consumer here: the
- // spec's name states always append at least one character.
- // ASCII-only folding: the name states lowercase ASCII upper alpha and append
- // every other character unchanged, so a name of `Ð` stays `Ð`.
- tok.name =
- nameStart === -1
- ? ""
- : _asciiLowerCase(doctypePart(input, nameStart, nameEnd));
- tok.publicId =
- publicStart === -1 ? null : doctypePart(input, publicStart, publicEnd);
- tok.systemId =
- systemStart === -1 ? null : doctypePart(input, systemStart, systemEnd);
- tok.forceQuirks = forceQuirks;
- tok.start = start;
- tok.end = end;
- dispatch();
- return end;
- };
- // A `<?…>` bogus comment is a processing instruction when its target is
- // well-formed (§13.2.5). The tokenizer reports the source range; the DOM-level
- // distinction is drawn here, so `tokenize` keeps reporting one comment token.
- const PI_SCAN_COMMENT = 0;
- const PI_SCAN_MATCH = 1;
- // Cut short by EOF — the spec emits only an end-of-file token, so no node.
- const PI_SCAN_DROP = 2;
- // Sub-ranges of the instruction last scanned, read only right after a match.
- let _piScanTargetEnd = 0;
- let _piScanDataStart = 0;
- let _piScanDataEnd = 0;
- /**
- * @param {string} input source
- * @param {number} start offset of the `<`
- * @param {number} end offset past the token
- * @returns {number} one of the `PI_SCAN_*` outcomes
- */
- const _scanProcessingInstruction = (input, start, end) => {
- const targetStart = start + 2;
- // Nothing after `<?`, so EOF landed in the open state and drops the token.
- if (targetStart >= end) return PI_SCAN_DROP;
- const first = input.charCodeAt(targetStart);
- if (!isAsciiAlpha(first) && first !== CC_LOW_LINE) return PI_SCAN_COMMENT;
- let i = targetStart + 1;
- while (i < end) {
- const cc = input.charCodeAt(i);
- if (isSpace(cc) || cc === CC_QUESTION_MARK || cc === CC_GREATER_THAN) break;
- if (
- !isAsciiAlphanumeric(cc) &&
- cc !== CC_HYPHEN_MINUS &&
- cc !== CC_LOW_LINE
- ) {
- return PI_SCAN_COMMENT;
- }
- i++;
- }
- // EOF while still reading the target drops the token, whatever it spells; a
- // target that ends at a delimiter is judged below first.
- if (i === end) return PI_SCAN_DROP;
- // Reserved for XML, and common enough in legacy pages to keep as comments.
- if (
- rangeEqualsLowerCase(input, targetStart, i, "xml") ||
- rangeEqualsLowerCase(input, targetStart, i, "xml-stylesheet")
- ) {
- return PI_SCAN_COMMENT;
- }
- _piScanTargetEnd = i;
- // The rest ran into EOF instead of the closing `>`, so the token is dropped.
- if (input.charCodeAt(end - 1) !== CC_GREATER_THAN) return PI_SCAN_DROP;
- let contentEnd = end - 1;
- while (i < contentEnd && isSpace(input.charCodeAt(i))) i++;
- // A `?` right before the `>` closes the instruction instead of being data.
- if (contentEnd > i && input.charCodeAt(contentEnd - 1) === CC_QUESTION_MARK) {
- contentEnd--;
- }
- _piScanDataStart = i;
- _piScanDataEnd = contentEnd;
- return PI_SCAN_MATCH;
- };
- /** @type {(input: string, start: number, end: number, dataStart: number, dataEnd: number) => number} */
- const handleCommentToken = (input, start, end, dataStart, dataEnd) => {
- // Only the token immediately after the start tag may be the swallowed
- // newline; anything else clears the flag (§13.2.6.4.7 "the next token").
- swallowNextNewline = false;
- // A CDATA section in foreign content arrives through this callback but is
- // character data, not a comment.
- if (
- input.startsWith("<![CDATA[", start) &&
- open.length > 0 &&
- adjustedCurrent() &&
- _namespaceOf(adjustedCurrent()) !== NS_HTML
- ) {
- const data = input.slice(dataStart, dataEnd).replace(/\r\n?/g, "\n");
- if (data !== "") {
- tok.type = TOKEN_CHAR;
- tok.data = data;
- tok.start = start;
- tok.end = end;
- dispatch();
- }
- return end;
- }
- if (input.charCodeAt(start + 1) === CC_QUESTION_MARK) {
- const scan = _scanProcessingInstruction(input, start, end);
- if (scan === PI_SCAN_DROP) return end;
- if (scan === PI_SCAN_MATCH) {
- tok.type = TOKEN_PROCESSING_INSTRUCTION;
- tok.name = input.slice(start + 2, _piScanTargetEnd);
- const piRaw = input.slice(_piScanDataStart, _piScanDataEnd);
- tok.data = inputHasNul ? piRaw.replace(/\0/g, "\uFFFD") : piRaw;
- tok.start = start;
- tok.end = end;
- dispatch();
- return end;
- }
- }
- tok.type = TOKEN_COMMENT;
- const commentRaw = input.slice(dataStart, dataEnd);
- tok.data = inputHasNul ? commentRaw.replace(/\0/g, "�") : commentRaw;
- tok.start = start;
- tok.end = end;
- dispatch();
- return end;
- };
- /** @type {(input: string, start: number, end: number) => number} */
- const handleTextToken = (input, start, end) => {
- // `skip.text` fast path: node dropped, so only whitespace-ness matters.
- // With no `&`/`\0`/`\r` (and no pending newline-swallow) the decoded value
- // equals the raw range — dispatch a canonical marker (`" "`/`"x"`) without
- // the slice + entity decode; anything trickier falls through.
- if (skipText && !swallowNextNewline) {
- let hasNonWhitespace = false;
- let i = start;
- for (; i < end; i++) {
- const charCode = input.charCodeAt(i);
- // 2 = break (& / NUL / CR), 1 = whitespace, 0 = other.
- const scanClass = charCode < 128 ? _TEXT_SCAN_CLASS[charCode] : 0;
- if (scanClass === 2) break;
- if (scanClass === 0) hasNonWhitespace = true;
- }
- if (i === end) {
- tok.type = TOKEN_CHAR;
- tok.data = hasNonWhitespace ? "x" : " ";
- tok.start = start;
- tok.end = end;
- dispatch();
- return end;
- }
- }
- const raw = input.slice(start, end);
- // CR normalization only when a CR is actually present (common case has none).
- const normalized =
- inputHasCr && raw.includes("\r") ? raw.replace(/\r\n?/g, "\n") : raw;
- const top = adjustedCurrent();
- const rawMode =
- mode === MODE_TEXT ||
- (top && _namespaceOf(top) === NS_HTML && _tagNameOf(top) === "plaintext");
- const noDecode =
- top &&
- _namespaceOf(top) === NS_HTML &&
- NO_DECODE_TEXT.has(_tagNameOf(top)) &&
- (mode === MODE_TEXT || mode === MODE_IN_BODY);
- let data = noDecode ? normalized : decode(normalized, false);
- // In RAWTEXT/RCDATA/script/PLAINTEXT, NULL becomes U+FFFD (tokenizer
- // rule); in data state NULLs pass through to be dropped in "in body".
- if (rawMode && inputHasNul && data.includes("\0")) {
- data = data.replace(/\0/g, "�");
- }
- // pre/listing/textarea swallow a leading newline (post entity decode).
- if (swallowNextNewline) {
- swallowNextNewline = false;
- if (data[0] === "\n") data = data.slice(1);
- }
- if (data === "") return end;
- tok.type = TOKEN_CHAR;
- tok.data = data;
- tok.start = start;
- tok.end = end;
- dispatch();
- return end;
- };
- /** @type {(input: string, nameStart: number, nameEnd: number, valueStart: number, valueEnd: number, quoteType: number) => number} */
- const handleAttributeToken = (
- input,
- nameStart,
- nameEnd,
- valueStart,
- valueEnd,
- quoteType
- ) => {
- const name = internLowerName(ATTR_NAME_INTERN, input, nameStart, nameEnd);
- if (pendAttrStart === 0) pendAttrStart = _attrCount + 1;
- // Drop duplicate attribute names (per spec). Plain loop avoids a
- // per-attribute closure allocation on this hot path.
- let dup = false;
- for (let i = pendAttrStart; i <= _attrCount; i++) {
- if (_attrNames[i] === name) {
- dup = true;
- break;
- }
- }
- if (!dup) {
- // The raw (undecoded) value is read from the source by offset on
- // demand: consumers re-resolve requests from it and the offsets must
- // stay aligned with the source. Only a valueless attribute stores an
- // override ("").
- _allocAttr(
- name,
- valueStart !== -1 ? null : "",
- nameStart,
- nameEnd,
- valueStart,
- valueEnd
- );
- }
- if (valueStart === -1) return nameEnd;
- return quoteType !== QUOTE_NONE ? valueEnd + 1 : valueEnd;
- };
- /** @type {(input: string, start: number, end: number, nameStart: number, nameEnd: number, selfClosing: boolean) => number} */
- const handleStartTagToken = (
- input,
- start,
- end,
- nameStart,
- nameEnd,
- selfClosing
- ) => {
- // Only the token immediately after the start tag may be the swallowed
- // newline; anything else clears the flag (§13.2.6.4.7 "the next token").
- swallowNextNewline = false;
- // A start tag the tokenizer only emitted because it hit EOF mid-tag
- // is dropped, matching the spec's eof-in-tag handling.
- if (eofInTag) {
- // Any attribute slots already allocated for it are orphaned.
- pendAttrStart = 0;
- return end;
- }
- const name = internLowerName(TAG_NAME_INTERN, input, nameStart, nameEnd);
- if (name === "selectedcontent") sawSelectedContent = true;
- tok.type = TOKEN_START_TAG;
- tok.name = name;
- // The reused token carries the tag's attribute run; every consumer of
- // `t.attrs` runs synchronously within this dispatch.
- tok.attrs.start = pendAttrStart;
- tok.attrs.count = pendAttrStart === 0 ? 0 : _attrCount + 1 - pendAttrStart;
- pendAttrStart = 0;
- tok.selfClosing = selfClosing;
- tok.swallowNewline = false;
- tok.pos.start = start;
- tok.pos.end = end;
- tok.pos.tagEnd = end;
- tok.pos.nameEnd = nameEnd;
- dispatch();
- if (tok.swallowNewline) swallowNextNewline = true;
- return end;
- };
- /** @type {(input: string, start: number, end: number, nameStart: number, nameEnd: number) => number} */
- const handleEndTagToken = (input, start, end, nameStart, nameEnd) => {
- // Only the token immediately after the start tag may be the swallowed
- // newline; anything else clears the flag (§13.2.6.4.7 "the next token").
- swallowNextNewline = false;
- // Most end tags close the current element: compare the raw range against
- // its already-interned tag name — one case-folding walk, no hash/probe.
- let name;
- if (open.length > 0) {
- const topName = _nodeStrings[open[open.length - 1]];
- if (rangeEqualsLowerCase(input, nameStart, nameEnd, topName)) {
- name = topName;
- }
- }
- if (name === undefined) {
- name = internLowerName(TAG_NAME_INTERN, input, nameStart, nameEnd);
- }
- // End tags drop any parsed attributes (the slots are orphaned).
- pendAttrStart = 0;
- tok.type = TOKEN_END_TAG;
- tok.name = name;
- tok.pos.start = start;
- tok.pos.end = end;
- tok.pos.tagEnd = end;
- tok.pos.nameEnd = nameEnd;
- dispatch();
- return end;
- };
- /** @type {HtmlTokenCallbacks} */
- const PARSE_CALLBACKS = {
- isForeign: currentNodeIsForeign,
- vetoesContentMode,
- fragmentContext: undefined,
- parseError: handleParseError,
- doctype: handleDoctypeToken,
- comment: handleCommentToken,
- text: handleTextToken,
- attribute: handleAttributeToken,
- openTag: handleStartTagToken,
- closeTag: handleEndTagToken
- };
- /**
- * @param {string} input HTML source
- * @param {number=} pos start byte offset (string input only; default `0`)
- * @param {HtmlParseOptions=} options parse options (fragment context, AST skips)
- * @returns {HtmlDocument} ref to the document node — read through `A`; valid until the next parse
- */
- const parseHtml = (input, pos = 0, options = {}) => {
- source = input;
- inputHasCr = input.includes("\r");
- inputHasNul = input.includes("\0");
- // Only an unknown name's entry is a slice of the last document; one naming a
- // known element is a shared constant, and stays warm across parses.
- _dropSourceBackedNames(TAG_NAME_INTERN);
- _dropSourceBackedNames(ATTR_NAME_INTERN);
- const { fragmentContext, skip = EMPTY_SKIP } = options;
- skipText = skip.text === true;
- skipComments = skip.comments === true;
- skipDoctype = skip.doctype === true;
- _resetAstColumns();
- _entered.length = 0;
- _skipFrom = -1;
- _streaming = false;
- _docSkipped = false;
- _openStackChanged = false;
- _openSpliced = false;
- _streamHalted = false;
- // Pre-size the columns from the input length: a document big enough to
- // exceed `_COLUMN_SHRINK_CAPACITY` re-enters with released (empty) columns,
- // and growing through the doubling cascade re-allocates and copies every
- // column several times per parse. ~12 bytes/node and ~50 bytes/attribute
- // are the densest realistic HTML, so `len/12` (`len/48`) reaches the final
- // capacity in one allocation; a sparser document over-allocates bounded
- // scratch that the post-walk release frees, and the doubling growth remains
- // as the safety net when the estimate is short. `skip.text` (what
- // `HtmlParser` uses) drops every text node — roughly half of all nodes — so
- // its node estimate uses `len/24`; attributes are unaffected by the skip and
- // keep `len/48`.
- let nodeEstimate = (input.length / (skipText ? 24 : 12)) | 0;
- const attrEstimate = (input.length / 48) | 0;
- if (_streamEligible) {
- // Node ids are recycled once streaming starts, so a document past the
- // threshold never needs the whole estimate — one allocation covering the
- // activation point plus a batch of headroom is its high-water mark. A
- // document below it is unaffected and still pre-sizes exactly. Attributes
- // are not recycled (`_attrCount` only resets between parses), so their
- // estimate stands either way.
- const cap = _STREAM_MIN_NODES + _FLUSH_BATCH * 2;
- if (nodeEstimate > cap) nodeEstimate = cap;
- }
- if (nodeEstimate > _nodeCapacity) _growNodeColumns(nodeEstimate, true);
- if (attrEstimate > _attrCapacity) _growAttrColumns(attrEstimate, true);
- _htmlSource = source;
- doc = _allocNode(NodeType.Document, 0, 0);
- _nodeStrings[doc] = "";
- mode = MODE_INITIAL;
- originalMode = 0;
- open.length = 0;
- afe.length = 0;
- head = 0;
- form = 0;
- framesetOk = true;
- pendAttrStart = 0;
- fosterParenting = false;
- _fosteredLog = null;
- _printFromSource = false;
- _fosteredRuns = null;
- _fosteredNodes = null;
- _fosterHolders = null;
- // Node slots are reused between parses, so the marks come off with the sets.
- if (_fosterVerbatim !== null) {
- for (const n of _fosterVerbatim) _nodeFlags[n] &= ~FLAG_FOSTER_REGION;
- }
- _fosterVerbatim = null;
- _fosterOpenTail = null;
- _fosterSpanEnd = 0;
- _fosterMoves = null;
- templateModes.length = 0;
- quirks = false;
- fragment = 0;
- tokenEnd = 0;
- sharedPlace.parent = doc;
- sharedPlace.beforeNode = 0;
- pendingTableChars = null;
- swallowNextNewline = false;
- eofInTag = false;
- sawSelectedContent = false;
- // ---------- fragment setup ----------
- if (fragmentContext) {
- let ctxName = fragmentContext.toLowerCase();
- let ctxNs = NS_HTML;
- if (ctxName.startsWith("svg ")) {
- ctxNs = NS_SVG;
- ctxName = ctxName.slice(4);
- } else if (ctxName.startsWith("math ")) {
- ctxNs = NS_MATHML;
- ctxName = ctxName.slice(5);
- }
- fragment = mkEl(ctxName, ctxNs, EMPTY_ATTRS, null);
- const htmlEl = mkEl("html", NS_HTML, EMPTY_ATTRS, null);
- _appendChild(doc, htmlEl);
- open.push(htmlEl);
- if (ctxNs !== NS_HTML) {
- mode = MODE_IN_BODY;
- } else if (["title", "textarea"].includes(ctxName)) {
- originalMode = MODE_IN_BODY;
- mode = MODE_TEXT;
- } else if (
- ["style", "xmp", "iframe", "noembed", "noframes", "script"].includes(
- ctxName
- )
- ) {
- originalMode = MODE_IN_BODY;
- mode = MODE_TEXT;
- } else if (ctxName === "noscript" || ctxName === "plaintext") {
- mode = MODE_IN_BODY;
- } else {
- resetInsertionMode();
- }
- if (ctxName === "template") {
- templateModes.push(MODE_IN_TEMPLATE);
- mode = MODE_IN_TEMPLATE;
- }
- }
- // Refresh the per-parse fragment context on the shared module-scope
- // callbacks object; its function properties stay identity-stable so the
- // tokenizer's callback call sites remain monomorphic.
- PARSE_CALLBACKS.fragmentContext = fragment ? _tagNameOf(fragment) : undefined;
- // A `skipChildren()` on the root means nothing under it is visited, so the
- // walk never starts and only the document's own `exit` is left to fire.
- if (_streamEligible && !_enterNode(doc)) {
- _streamEligible = false;
- _docSkipped = true;
- }
- tokenize(source, pos, PARSE_CALLBACKS);
- tok.type = TOKEN_EOF;
- dispatch();
- // Not while printing: the mirror is content the engine makes, with no source
- // text behind it, and the printer emits what it walks — so serializing it
- // would write the selected option's subtree into the document, where the
- // engine that reads the output mirrors it a second time.
- if (sawSelectedContent && _writer === undefined) {
- mirrorSelectedContent(doc, 0);
- }
- if (_streamEligible) _streamFinish(doc);
- else if (_docSkipped) _exitNode(doc);
- return doc;
- };
- /**
- * Deep-clone a node into fresh ids, dropping attribute source offsets and the
- * raw-text body span so cloned content does not re-emit dependencies.
- * @param {HtmlNodeRef} node node
- * @returns {HtmlNodeRef} clone
- */
- const cloneSubtree = (node) => {
- const ty = _nodeTypes[node];
- const clone = _allocNode(ty, _nodeStarts[node], _nodeEnds[node]);
- _nodeStrings[clone] = _nodeStrings[node];
- if (ty !== NodeType.Element) return clone;
- _nodeFlags[clone] = _nodeFlags[node];
- const attrs = cloneAttrs(node);
- _nodeAttrStarts[clone] = attrs.start;
- _nodeAttrCounts[clone] = attrs.count;
- _nodeTagEnds[clone] = _nodeTagEnds[node];
- _nodeNameEnds[clone] = _nodeNameEnds[node];
- // Empty body span: the clone re-emits no raw-text dependency.
- _nodeContentEnds[clone] = _nodeTagEnds[node];
- for (let k = _nodeFirstChildren[node]; k !== 0; k = _nodeNextSiblings[k]) {
- _appendChild(clone, cloneSubtree(k));
- }
- // Clone `<template>` content into the clone's own fragment.
- const tc = _templateContentOf(node);
- if (tc !== 0) {
- const fragment = _allocNode(NodeType.DocumentFragment, 0, 0);
- _nodeStrings[fragment] = "";
- // Parent link so the iterative walk can ascend out of the content.
- _nodeParents[fragment] = clone;
- _nodeContentEnds[clone] = fragment;
- _nodeFlags[clone] |= FLAG_HAS_TEMPLATE;
- for (let k = _nodeFirstChildren[tc]; k !== 0; k = _nodeNextSiblings[k]) {
- _appendChild(fragment, cloneSubtree(k));
- }
- }
- return clone;
- };
- /**
- * The selected option of a select: the last `<option selected>`, else the
- * first option (scanning direct children and `<optgroup>` children).
- * @param {HtmlElement} select select element
- * @returns {HtmlElement} selected option (0 = none)
- */
- const selectedOption = (select) => {
- /** @type {HtmlElement[]} */
- const options = [];
- /** @param {HtmlElement} el element */
- const collect = (el) => {
- for (let c = _nodeFirstChildren[el]; c !== 0; c = _nodeNextSiblings[c]) {
- if (_nodeTypes[c] !== NodeType.Element || _namespaceOf(c) !== NS_HTML) {
- continue;
- }
- if (_nodeStrings[c] === "option") options.push(c);
- else if (_nodeStrings[c] === "optgroup") collect(c);
- }
- };
- collect(select);
- if (options.length === 0) return 0;
- for (let i = options.length - 1; i >= 0; i--) {
- const el = options[i];
- if (_findAttr(_nodeAttrStarts[el], _nodeAttrCounts[el], "selected") !== 0) {
- return el;
- }
- }
- return options[0];
- };
- /**
- * Fill each `<selectedcontent>` with a clone of its `<select>`'s selected
- * option subtree (the customizable-select mirroring behavior). Not while
- * printing: the clone has no source behind it, and the printer writes what it
- * walks (see the call site).
- * @param {HtmlNodeRef} node node
- * @param {HtmlElement} select nearest ancestor select (0 = none)
- */
- const mirrorSelectedContent = (node, select) => {
- // A `<template>`'s children live in its content fragment.
- const tc = _templateContentOf(node);
- const container = tc !== 0 ? tc : node;
- for (
- let child = _nodeFirstChildren[container];
- child !== 0;
- child = _nodeNextSiblings[child]
- ) {
- if (_nodeTypes[child] !== NodeType.Element) continue;
- if (_namespaceOf(child) === NS_HTML && _nodeStrings[child] === "select") {
- mirrorSelectedContent(child, child);
- } else if (
- select !== 0 &&
- _namespaceOf(child) === NS_HTML &&
- _nodeStrings[child] === "selectedcontent"
- ) {
- const option = selectedOption(select);
- if (option !== 0) {
- // Replace the children with clones of the option's subtree.
- _nodeFirstChildren[child] = 0;
- _nodeLastChildren[child] = 0;
- for (
- let k = _nodeFirstChildren[option];
- k !== 0;
- k = _nodeNextSiblings[k]
- ) {
- _appendChild(child, cloneSubtree(k));
- }
- // A `<select>` carried into the clone is a select of its own there, so
- // its `<selectedcontent>` mirrors inside the copy as it would anywhere.
- mirrorSelectedContent(child, 0);
- }
- } else if (
- select !== 0 &&
- _namespaceOf(child) === NS_HTML &&
- _nodeStrings[child] === "option"
- ) {
- // An option is what a `<selectedcontent>` mirrors, so nothing inside one
- // mirrors: not a `<selectedcontent>` written there, and not a `<select>`
- // nested under it either.
- continue;
- } else {
- mirrorSelectedContent(child, select);
- }
- }
- };
- /** @typedef {HtmlNode | HtmlDocument | HtmlDocumentFragment} HtmlVisitableNode */
- // HTML-typed views over the generic visitor machinery (`util/SourceProcessor`).
- /**
- * @typedef {import("../util/SourceProcessor").VisitorFn<HtmlPath>} VisitorFn
- * @typedef {import("../util/SourceProcessor").VisitorBucket<HtmlPath>} VisitorBucket
- * @typedef {import("../util/SourceProcessor").VisitorMap<HtmlPath>} VisitorMap
- * @typedef {import("../util/SourceProcessor").CompiledVisitorMap<HtmlPath>} CompiledVisitorMap
- */
- /**
- * What the minifying printer may rewrite. Every entry is on unless it is
- * `false`, so a document one transform breaks can still be minified by the rest.
- * @typedef {object} HtmlTransformOptions
- * @property {boolean=} collapseBooleanAttributes write a boolean attribute spelled with its own name (`disabled="disabled"`) as the bare name
- * @property {(boolean | "all" | "some" | string | RegExp | ((comment: string) => boolean))=} comments which comments survive: `"some"` (the default) the ones that carry something, `true` / `"all"` every one, `false` none, or the ones a pattern matches / a predicate accepts, over the comment's own text. A comment a parser or a server reads is kept whatever this says
- * @property {boolean=} minifyJson strip the whitespace between a JSON `<script>`'s tokens
- * @property {boolean=} minifyStyles run the CSS minifier over an inline `<style>` and every `style=""`
- * @property {boolean=} normalizeAttributeQuotes drop or re-pick an attribute value's quotes
- * @property {boolean=} normalizeEnumeratedAttributes fold an enumerated value to the keyword it names
- * @property {boolean=} normalizeListAttributes normalize a space-, comma- or descriptor-separated list value (`class`, `rel`, `srcset`, `sizes`, the viewport `content`)
- * @property {boolean=} normalizeNumericAttributes write an integer attribute the one way its rules read it
- * @property {boolean=} removeOptionalTags leave out an optional tag other than the `<html>` / `<head>` / `<body>` shell, which is `removeImpliedTags`
- */
- /**
- * @typedef {object} HtmlProcessOptions
- * @property {string=} fragmentContext context element tag name for fragment parsing (see `parseHtml`); the HTML analog of the CSS parser's `as` parse-mode option
- * @property {HtmlAstSkip=} skip node kinds to omit from the AST for speed/memory (see `HtmlAstSkip`)
- * @property {boolean=} minimize print the safely-minified serialization (nodes rebuilt from source, inert comments dropped, opening-tag whitespace collapsed) as `process` walks, and return it (default false = walk only, return `""`)
- * @property {CssEnvironment=} environment CSS's, not HTML's: handed to the CSS minifier that runs over an inline `<style>` and every `style=""`, so the inline copy of a declaration agrees with the `.css` asset
- * @property {boolean=} convertLengthUnits CSS's too, handed over with `environment` (see `HtmlPrintOptions`)
- * @property {boolean=} rewriteCustomProperties CSS's too, handed over with `environment` (see `HtmlPrintOptions`)
- * @property {CssTransformOptions=} cssTransforms CSS's per-transform switches too, handed over with `environment` (see `HtmlPrintOptions`)
- * @property {HtmlTransformOptions=} transforms which of the meaning-preserving rewrites the minifying print makes; each is on unless it is `false`
- * @property {(boolean | "conservative" | "smart" | "all")=} collapseWhitespace collapse each run of whitespace in text to a single space, except where an ancestor renders it verbatim; `"smart"` also drops what sits against a block edge and `"all"` drops every edge (default false)
- * @property {boolean=} removeEmptyAttributes drop an attribute whose empty value leaves it in the state its absence gives (default false)
- * @property {boolean=} removeEmptyElements drop an element with no children and no attributes, unless its bare form is meaningful (default false)
- * @property {boolean=} mergeStyles print a run of adjacent `<style>` elements as one sheet, which removes elements (default false)
- * @property {boolean=} sortAttributes print an element's attributes commonest name first, ties by name, which nothing in HTML reads (default false)
- * @property {boolean=} sortTokenLists print every token list the DOM reads as a set (`class`, `rel`, `part`, …) in token order (default false)
- * @property {(boolean | "smart" | "all")=} removeRedundantAttributes drop an attribute whose value is the element's own default; `true` is `"smart"`, and `"all"` also drops the spec defaults a selector can match (default false)
- * @property {(boolean | "smart" | "all")=} removeImpliedTags how much of the `<html>` / `<head>` / `<body>` shell §13.1.2.4 lets the parser imply may be left out: `"smart"` leaves out only the `<html>` start tag, `true` (or `"all"`) all six, `false` none (default `"smart"`)
- * @property {boolean=} deferSrcdoc whether an `<iframe srcdoc>` is among what `deferEmbeddedSource` collects (default true); false for a caller that minifies them itself, which keeps the attribute on the normal path and its shorter delimiter
- * @property {DeferredEmbeddedSource[]=} deferEmbeddedSource collects what `renderEmbeddedSource` would be offered instead of offering it, for a caller whose renderer is asynchronous: the print leaves a marker for each and `finish` puts the answers in their place, so one parse serves both. A `style=""` stays with the built-in CSS minifier, whose text this print reads back to decide how the attribute is written. Takes precedence over `renderEmbeddedSource`
- * @property {EmbeddedSourceRenderer=} renderEmbeddedSource renders each nested body this document embeds — an inline `<style>`, every `style=""` (handed over as a whole stylesheet, SVG's and MathML's included), a `<script>` holding JSON or JavaScript, an `<svg>` subtree, and the document an `<iframe srcdoc>` holds (decoded, and written back escaped). Replaces the built-in CSS and JSON minifiers wherever it answers, and returning anything but text falls back to them; it is the only way inline JavaScript, SVG and a nested document are reached at all
- */
- /**
- * What the HTML printer may be told, on top of the `mode` every language has.
- * The first two are not HTML's own: an inline `<style>` and every `style=""` run
- * through the CSS minifier, so `optimization.minimize.css` rides along to reach
- * it (see `lib/config/defaults.js`) and the two must not disagree about what the
- * target can read. The rest is `optimization.minimize.html`.
- * @typedef {Pick<CssProcessOptions, "environment" | "convertLengthUnits" | "rewriteCustomProperties"> & { cssTransforms?: CssTransformOptions, transforms?: HtmlTransformOptions, collapseWhitespace?: boolean | "conservative" | "smart" | "all", mergeStyles?: boolean, removeEmptyAttributes?: boolean, removeEmptyElements?: boolean, removeRedundantAttributes?: boolean | "smart" | "all", sortAttributes?: boolean, sortTokenLists?: boolean, removeImpliedTags?: boolean | "smart" | "all", renderEmbeddedSource?: EmbeddedSourceRenderer, deferEmbeddedSource?: DeferredEmbeddedSource[], deferSrcdoc?: boolean }} HtmlPrintOptions
- */
- /** @typedef {import("../util/SourceProcessor").PrintContext<HtmlPath, HtmlNodeRef, HtmlPrintOptions>} PrintContext */
- // Whether the text ends inside an unterminated character reference, so the next
- // sibling's first characters could complete it (`a &am` + `p;`).
- /**
- * @param {string} s source text
- * @returns {boolean} true when a character reference is still open at the end
- */
- const _hasOpenReference = (s) => {
- const last = s.lastIndexOf("&");
- if (last === -1) return false;
- // Read in place: slicing the tail off to match it against a pattern allocates
- // a string per text node, and every one of them is thrown away here.
- for (let i = last + 1; i < s.length; i++) {
- const c = s.charCodeAt(i);
- if (!isAsciiAlphanumeric(c) && c !== CC_NUMBER_SIGN) return false;
- }
- return true;
- };
- /**
- * Whether the `&` at `i` would be consumed as a character reference when the
- * text is read back (§13.2.5.72): only before `#` or a name the table holds.
- * Every other `&` the tokenizer flushes as literal text, so `R&D` needs no
- * escape — and neither does the `&` in a query string.
- * @param {string} s decoded text
- * @param {number} i the `&`'s index
- * @returns {boolean} true when it has to be escaped
- */
- const _startsCharacterReference = (s, i) => {
- // Nothing terminates the run before the node ends, so the next sibling's
- // first characters could complete it once a comment between them is dropped
- // (`a &am` + `p;`) — the same reason the source fast path keeps out.
- let end = i + 1;
- while (end < s.length) {
- const c = s.charCodeAt(end);
- if (!isAsciiAlphanumeric(c) && c !== 0x23) break;
- end++;
- }
- if (end === s.length) return true;
- const next = s.charCodeAt(i + 1);
- // `#` opens a numeric reference.
- if (next === 0x23) return true;
- if (!isAsciiAlphanumeric(next)) return false;
- // The named-reference table holds the names with and without their `;`, so
- // any prefix hit is a reference the tokenizer would take.
- const longest = Math.min(s.length, i + 1 + MAX_ENTITY_NAME_LEN);
- for (let j = i + 2; j <= longest; j++) {
- if (HTML_ENTITIES[s.slice(i + 1, j)] !== undefined) return true;
- }
- return false;
- };
- /**
- * Escape a text node's decoded data for re-serialization: `&` and `<`, which
- * start a character reference and a tag, plus `>` outside minification, where
- * the §13.3 serialization is followed to the letter. A CR is written as a
- * reference either way — §13.2.3.5 rewrites a literal one to LF before the
- * tokenizer reads it. Unlike the exported `escapeText` (which also numerically
- * encodes newlines and U+00A0 for single-line `data:` URIs), this keeps
- * newlines and U+00A0 literal — byte-lean and the same DOM.
- * @param {string} s decoded text
- * @param {boolean} minify whether the shorter (still equivalent) form is wanted
- * @returns {string} text-content-safe string
- */
- const _escapeTextContent = (s, minify) => {
- // A bare `>` is only ever a character in text, so §13.3's escape of it buys
- // nothing but bytes. `<` always starts a tag, and an `&` only sometimes
- // starts a character reference — §13.2.5.72 flushes the rest as literal text.
- if (minify) {
- if (!s.includes("&") && !s.includes("<") && !s.includes("\r")) return s;
- let out = "";
- let last = 0;
- for (let i = 0; i < s.length; i++) {
- const c = s.charCodeAt(i);
- if (c === 0x3c) {
- out += `${s.slice(last, i)}<`;
- last = i + 1;
- } else if (c === 0x26 && _startsCharacterReference(s, i)) {
- out += `${s.slice(last, i)}&`;
- last = i + 1;
- } else if (c === 0x0d) {
- // §13.2.3.5 rewrites a literal CR to LF before the tokenizer sees it,
- // so one a character reference put in the text has to stay one.
- out += `${s.slice(last, i)} `;
- last = i + 1;
- }
- }
- return last === 0 ? s : out + s.slice(last);
- }
- if (
- !s.includes("&") &&
- !s.includes("<") &&
- !s.includes(">") &&
- !s.includes("\r")
- ) {
- return s;
- }
- return s
- .replace(/&/g, "&")
- .replace(/</g, "<")
- .replace(/>/g, ">")
- .replace(/\r/g, " ");
- };
- /**
- * Rebuild an opening tag from the node's name and attributes, for elements with
- * no matching source tag (adoption-agency clones, reconstructed formatting
- * elements, renamed tokens like `<image>`). Attribute values are raw
- * (undecoded), so they re-parse to the original decoded values.
- * @param {HtmlPath} path the accessor positioned on the element
- * @returns {string} serialized opening tag
- */
- const _synthesizeOpenTag = (path) => {
- let out = `<${path.tagName()}`;
- for (const attribute of path.attributes()) {
- const name =
- attribute.serializedName !== undefined
- ? attribute.serializedName
- : attribute.name;
- out +=
- attribute.value === ""
- ? ` ${name}`
- : ` ${name}="${attribute.value.replace(/"/g, """)}"`;
- }
- return `${out}>`;
- };
- /**
- * Whether an element's default display puts whitespace at its edge outside every
- * line box, so none of it renders.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node node (0 = none)
- * @returns {boolean} true when it is a block-level HTML element
- */
- const _isBlockLevel = (path, node) =>
- node !== 0 &&
- path.type(node) === NodeType.Element &&
- path.namespace(node) === NS_HTML &&
- BLOCK_LEVEL_ELEMENTS.has(path.tagName(node));
- /**
- * Whether a text node's leading whitespace sits outside every line box, which is
- * so when it opens a block container. A block *sibling* in front of it would do
- * the same, but reading backwards costs a scan of the whole child list — so this
- * only keeps a space that renders, never drops one that does not.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the text node
- * @param {HtmlNodeRef} parent its parent (0 = none)
- * @returns {boolean} true when the leading whitespace renders as nothing
- */
- const _startsOutsideLineBox = (path, node, parent) =>
- parent !== 0 &&
- _nodeFirstChildren[parent] === node &&
- _isBlockLevel(path, parent);
- /**
- * Whether a text node's trailing whitespace sits outside every line box: it
- * closes a block container, or a block element follows it.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the text node
- * @param {HtmlNodeRef} parent its parent (0 = none)
- * @returns {boolean} true when the trailing whitespace renders as nothing
- */
- const _endsOutsideLineBox = (path, node, parent) => {
- const next = path.nextSibling(node);
- if (next === 0) return parent !== 0 && _isBlockLevel(path, parent);
- return _isBlockLevel(path, next);
- };
- /**
- * Whether a node prints nothing at all, so what follows it in the output is its
- * next sibling. The two that can: a comment the inert-comment rule drops, and a
- * whitespace-only text node the active `collapseWhitespace` tier deletes — the
- * same two questions the printer asks when it reaches them, asked here so the
- * §13.1.2.4 tests below read the output rather than the tree.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the node
- * @returns {boolean} true when it contributes no output
- */
- const _printsNothing = (path, node) => {
- const type = path.type(node);
- if (type === NodeType.Comment) {
- return _minifying && !_keepComment(path.data(node), path.source(node));
- }
- if (type !== NodeType.Text) return false;
- const data = path.data(node);
- if (!isAllWs(data)) return false;
- const parent = path.parentOf(node);
- if (parent === 0 || path.namespace(parent) !== NS_HTML) return false;
- const parentName = path.tagName(parent);
- // A literal-text body is data — its whitespace is printed as written.
- if (LITERAL_TEXT_PARENTS.has(parentName)) return false;
- // Outside any block formatting context, so nothing ever renders it — which is
- // a reason to drop it, not a claim that a re-serializing pass already did.
- if (parentName === "head" || parentName === "html") return _minifying;
- if (
- _collapseWhitespace === "none" ||
- _collapseWhitespace === "conservative"
- ) {
- return false;
- }
- for (
- let ancestor = parent;
- ancestor !== 0;
- ancestor = path.parentOf(ancestor)
- ) {
- if (
- path.namespace(ancestor) === NS_HTML &&
- LEADING_NEWLINE_ELEMENTS.has(path.tagName(ancestor))
- ) {
- return false;
- }
- }
- // Whitespace-only, so it collapses to one space; either edge going empties it.
- return (
- _collapseWhitespace === "all" ||
- _startsOutsideLineBox(path, node, parent) ||
- _endsOutsideLineBox(path, node, parent)
- );
- };
- /**
- * The next sibling that prints something, skipping the ones that print nothing.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the node to start after
- * @returns {HtmlNodeRef} that sibling, or 0
- */
- const _nextPrintedSibling = (path, node) => {
- let next = path.nextSibling(node);
- while (next !== 0 && _printsNothing(path, next)) {
- next = path.nextSibling(next);
- }
- return next;
- };
- // The elements whose start tag closes the one before it only inside a `ruby`.
- const RUBY_SEGMENTS = new Set(["rb", "rp", "rt", "rtc"]);
- /**
- * Whether a `ruby` is in scope above `node`. A `<rb>` / `<rt>` / `<rp>` start tag
- * only closes the one before it when there is one (§13.2.6.4.7), so without it
- * an omitted end tag nests the next element inside this one instead.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when a `ruby` encloses it
- */
- const _hasRubyInScope = (path, node) => {
- for (
- let parent = path.parentOf(node);
- parent !== 0 && path.type(parent) === NodeType.Element;
- parent = path.parentOf(parent)
- ) {
- if (path.namespace(parent) !== NS_HTML) return false;
- const name = path.tagName(parent);
- if (name === "ruby") return true;
- if (HTML_SCOPE.has(name)) return false;
- }
- return false;
- };
- /**
- * Whether an element's end tag may be omitted (§13.1.2.4). Reads what will be
- * printed rather than the tree: a comment minifying drops, and whitespace the
- * collapse tier deletes, are not content the tag has to stay for.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {string} name the element's lowercased tag name
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when the end tag can be left out
- */
- const _canOmitEndTag = (path, name, node) => {
- if (path.namespace(node) !== NS_HTML) return false;
- if (name === "html" || name === "head" || name === "body") {
- return _removeImpliedTags !== "none" && _canOmitShellEnd(path, name, node);
- }
- if (!_transforms.removeOptionalTags) return false;
- if (OPTIONAL_END_TAG_UNLESS_TRAILING_NODE.has(name)) {
- // Whitespace or a comment behind it would move inside once the tag goes;
- // an element closes it through the insertion mode instead.
- const trailing = _nextPrintedSibling(path, node);
- if (trailing === 0) return true;
- if (path.type(trailing) !== NodeType.Element) return false;
- // A `<template>` does not close what it stands behind: §13.2.6.4.12 hands it
- // to the head rules, which insert it where it is.
- return !(
- path.namespace(trailing) === NS_HTML &&
- path.tagName(trailing) === "template"
- );
- }
- const followers = OPTIONAL_END_TAG_FOLLOWERS.get(name);
- if (followers === undefined) return false;
- const next = _nextPrintedSibling(path, node);
- if (next === 0) {
- if (!OPTIONAL_END_TAG_AT_END.has(name)) return false;
- const parent = path.parentOf(node);
- const parentIsHtml = parent !== 0 && path.namespace(parent) === NS_HTML;
- // Left open inside a formatting element, this is the furthest block the
- // adoption agency needs — the parent's end tag would then restructure both.
- if (
- parentIsHtml &&
- SPECIAL.has(name) &&
- FORMATTING.has(path.tagName(parent))
- ) {
- return false;
- }
- if (!parentIsHtml) return false;
- // A `<template>`'s content is a fragment, not an element, and `</template>`
- // closes what is open inside it.
- if (path.type(parent) !== NodeType.Element) return true;
- // "Any other end tag" walks past a node that is not special, so a parent's
- // end tag closes one of those whatever the parent is.
- if (!SPECIAL.has(name)) return true;
- // A special one blocks that walk instead, so only a parent whose end tag
- // generates implied end tags closes it. Under any other — `<canvas>`, a
- // custom element — it stays open and swallows what follows the parent.
- return P_ENDS_ON_PARENT_END_TAG.has(path.tagName(parent));
- }
- if (
- path.type(next) !== NodeType.Element ||
- path.namespace(next) !== NS_HTML ||
- !followers.has(path.tagName(next))
- ) {
- return false;
- }
- // A `<table>` start tag closes an open `p` only outside quirks mode — the
- // one follower whose reading the document mode decides.
- if (quirks && name === "p" && path.tagName(next) === "table") return false;
- // Outside a `ruby` the follower's start tag closes nothing, so this one would
- // go on to hold it.
- if (RUBY_SEGMENTS.has(name) && !_hasRubyInScope(path, node)) return false;
- // A parser-implied `<tbody>` / `<tr>` is printed transparently, so there is
- // no start tag to close this element — its rows would continue this one.
- return (
- path.openTag(next) !== "" ||
- path.attributeCount(next) !== 0 ||
- !TRANSPARENT_IMPLIED_ELEMENTS.has(path.tagName(next))
- );
- };
- /**
- * The last sibling before `node` that prints something, or 0 — the accessor only
- * walks forward, so this walks the chain. Skipping what prints nothing is what
- * keeps it the mirror of `_nextPrintedSibling`: the two decide whether one of
- * `</thead>` and a `<tbody>` start tag may go, and reading a dropped whitespace
- * node from one side only would let both go and merge the rows into the head.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the element
- * @returns {HtmlNodeRef} the preceding printed sibling, or 0
- */
- const _previousPrintedSibling = (path, node) => {
- let previous = 0;
- for (
- let c = path.firstChild(path.parentOf(node));
- c !== 0 && c !== node;
- c = path.nextSibling(c)
- ) {
- if (!_printsNothing(path, c)) previous = c;
- }
- return previous;
- };
- /**
- * Whether an element is shaped like one the parser would re-imply: no attributes
- * to carry, and `child` opening it. Anything before that child (whitespace, a
- * comment) would land outside the implied element, so it blocks the omission.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the element
- * @param {string} child the tag name the implied element must start with
- * @returns {boolean} true when the shape allows dropping the start tag
- */
- const _impliedStartTagShape = (path, node, child) => {
- if (path.attributeCount(node) !== 0 || path.namespace(node) !== NS_HTML) {
- return false;
- }
- const first = path.firstChild(node);
- return (
- first !== 0 &&
- path.type(first) === NodeType.Element &&
- path.namespace(first) === NS_HTML &&
- path.tagName(first) === child
- );
- };
- /**
- * Whether a `<tbody>` start tag may be omitted (§13.1.2.4): the parser re-implies
- * it when a `<tr>` opens a table section, but only if no earlier section ran into
- * it — a preceding `tbody` / `thead` / `tfoot` that also dropped its end tag would
- * swallow these rows instead.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when the start tag can be left out
- */
- const _canOmitTbodyStart = (path, node) => {
- if (!_impliedStartTagShape(path, node, "tr")) return false;
- const previous = _previousPrintedSibling(path, node);
- if (previous === 0 || path.type(previous) !== NodeType.Element) return true;
- const previousName = path.tagName(previous);
- if (
- previousName !== "tbody" &&
- previousName !== "thead" &&
- previousName !== "tfoot"
- ) {
- return true;
- }
- return !_canOmitEndTag(path, previousName, previous);
- };
- /**
- * Whether a `<colgroup>` start tag may be omitted (§13.1.2.4): a `<col>` in a
- * table implies one, but only when the `<col>` is the first thing inside and no
- * preceding `<colgroup>` dropped its end tag — that one would take these columns.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when the start tag can be left out
- */
- const _canOmitColgroupStart = (path, node) => {
- if (!_impliedStartTagShape(path, node, "col")) return false;
- const previous = _previousPrintedSibling(path, node);
- if (
- previous === 0 ||
- path.type(previous) !== NodeType.Element ||
- path.tagName(previous) !== "colgroup"
- ) {
- return true;
- }
- return !_canOmitEndTag(path, "colgroup", previous);
- };
- // The CSS abilities of the target, forwarded into the nested CSS minifier so an
- // inline `<style>` or `style=""` is gated exactly like a `.css` asset. Set per
- // print, since the printer is the only place the options are in scope.
- /** @type {CssEnvironment | undefined} */
- let _cssEnvironment;
- // Whether the nested CSS minifier may rewrite a length's unit — forwarded like
- // `_cssEnvironment` so inline CSS agrees with the `.css` assets.
- let _cssConvertLengthUnits = false;
- // And whether it may shorten a custom property's value, forwarded the same way.
- let _cssRewriteCustomProperties = false;
- // And which of its rewrites it makes at all, forwarded the same way — inline CSS
- // is minified by the same switches a `.css` asset is.
- /** @type {CssTransformOptions | undefined} */
- let _cssTransforms;
- // Every rewrite on, which is what a print with no `transforms` option makes and
- // what the walk-only passes hold. Frozen, since it is shared rather than copied.
- /** @type {Required<HtmlTransformOptions>} */
- const _ALL_TRANSFORMS = Object.freeze({
- collapseBooleanAttributes: true,
- comments: "some",
- minifyJson: true,
- minifyStyles: true,
- normalizeAttributeQuotes: true,
- normalizeEnumeratedAttributes: true,
- normalizeListAttributes: true,
- normalizeNumericAttributes: true,
- removeOptionalTags: true
- });
- /**
- * The rewrites this print makes, resolved once: each name is what the options
- * set it to, and what `_ALL_TRANSFORMS` gives it where they set nothing — so no
- * read of the result asks whether the option was given. `comments` keeps
- * whatever it was given, which is still falsy only when it is `false`.
- *
- * Always a copy, and always made the one way: spreading the table and then
- * writing over a name it already has leaves the object's shape alone, so every
- * read of the result — and there is one per token — sees a single hidden class.
- * Handing the frozen table itself back where nothing is set would save this
- * object and cost each of those reads a second shape, `Object.freeze` giving a
- * frozen object a map of its own.
- * @param {HtmlTransformOptions | undefined} options the `transforms` option
- * @returns {Required<HtmlTransformOptions>} every rewrite resolved
- */
- const _transformsFrom = (options) => {
- /** @type {EXPECTED_ANY} */
- const out = { ..._ALL_TRANSFORMS };
- if (options !== undefined) {
- for (const name of Object.keys(_ALL_TRANSFORMS)) {
- const value = /** @type {EXPECTED_ANY} */ (options)[name];
- if (value !== undefined) out[name] = value;
- }
- }
- return out;
- };
- // The same table, as a copy, for the reason `_transformsFrom` always makes one.
- /** @type {Required<HtmlTransformOptions>} */
- const _DEFAULT_TRANSFORMS = _transformsFrom(undefined);
- /**
- * The `comments` option resolved to what it means per comment: `true` every
- * one, `false` none, `"some"` the ones that carry something, or a predicate
- * over the comment's own text — which is what a pattern compiles to here, so no
- * print re-reads one per comment. A pattern is matched from the start each
- * time, since a `g` flag would otherwise carry an index between comments.
- * @param {Required<HtmlTransformOptions>["comments"]} comments the option
- * @returns {boolean | "some" | ((comment: string) => boolean)} what it keeps
- */
- const _keptComments = (comments) => {
- if (comments === "all") return true;
- if (typeof comments === "boolean" || comments === "some") return comments;
- if (typeof comments === "function") return comments;
- const pattern =
- typeof comments === "string" ? new RegExp(comments) : comments;
- return (comment) => {
- pattern.lastIndex = 0;
- return pattern.test(comment);
- };
- };
- // Which rewrites this print makes. One object rather than a flag apiece: it is
- // read straight off the options, and a fixed shape keeps each read monomorphic.
- /** @type {Required<HtmlTransformOptions>} */
- let _transforms = _DEFAULT_TRANSFORMS;
- /**
- * The per-transform switches out of a wider options object — what
- * `optimization.minimize.html` names beside the transforms that are off until
- * asked for. Exported so the minify function the minimizer plugin ships to its
- * workers can pick them without repeating the names.
- * @param {EXPECTED_OBJECT} options an options object that may carry them
- * @returns {HtmlTransformOptions | undefined} the switches, or undefined when none is set
- */
- const pickTransforms = (options) => {
- /** @type {EXPECTED_ANY} */
- let out;
- for (const name of Object.keys(_ALL_TRANSFORMS)) {
- const value = /** @type {EXPECTED_ANY} */ (options)[name];
- if (value === undefined) continue;
- if (out === undefined) out = {};
- out[name] = value;
- }
- return out;
- };
- // Minified `style` attributes by raw text — a pure function of the value and
- // the options above, which a document repeats far more often than it adds one.
- /** @type {Map<string, string>} */
- const _styleAttributeCache = new Map();
- // How far text whitespace may be collapsed (see `HtmlProcessOptions`) — set per
- // print, like `_cssEnvironment`. "none" is off; the rest name how much of the
- // whitespace against a boundary no line box reaches goes with it.
- /** @type {"none" | "conservative" | "smart" | "all"} */
- let _collapseWhitespace = "none";
- // Whether this run is minifying, for the tests that run before `printer` does and
- // have to predict what it will drop.
- let _minifying = false;
- // The caller's renderer for source nested in this document, set per print. It
- // replaces the built-in CSS/JSON minifiers and is the only way inline
- // JavaScript and SVG are reached — webpack ships no minifier for either.
- /** @type {EmbeddedSourceRenderer | undefined} */
- let _renderEmbeddedSource;
- // Where an asynchronous caller's embedded sources are recorded instead of
- // rendered, set per print. A body whose text this print reads back to decide
- // what to emit is still minified by the built-in for that decision — only the
- // text that reaches the output is deferred.
- /** @type {DeferredEmbeddedSource[] | undefined} */
- let _deferEmbeddedSource;
- // Whether an `<iframe srcdoc>` is among what is deferred. A caller minifying
- // them itself keeps the attribute on the normal path, which re-quotes it to
- // whichever delimiter its content makes shorter.
- let _deferSrcdoc = true;
- /**
- * Hand one nested body to the caller's renderer, which must be present.
- * Anything it throws is a renderer that did not answer: a document must not
- * fail to serialize because something inside it did not parse.
- * @param {string} source the nested body
- * @param {string} type its language, e.g. `"css"` / `"javascript"` / `"svg"`
- * @param {string=} as which production of that language it is, for one with more than one
- * @returns {string | null} the rendered body, or null when it declined
- */
- const _renderEmbeddedOrNull = (source, type, as) => {
- try {
- const rendered = /** @type {EmbeddedSourceRenderer} */ (
- _renderEmbeddedSource
- )(source, {
- type,
- hostType: HTML_TYPE,
- ...(as === undefined ? undefined : { as })
- });
- // Anything but text is a renderer that did not answer, not a body to write.
- return typeof rendered === "string" ? rendered : null;
- } catch (_err) {
- return null;
- }
- };
- /**
- * Record one nested body for a caller that answers asynchronously and print the
- * marker standing in for it. `onDeclined` spells what an untapped run spells,
- * so declining costs nothing the built-ins would have done.
- * @param {string} source the nested body
- * @param {string} type its language, e.g. `"css"` / `"javascript"` / `"svg"`
- * @param {(rendered: string | undefined) => string} build the text to print once the answer is in
- * @param {string=} as which production of that language it is, for one with more than one
- * @returns {string} the marker to print
- */
- const _deferEmbedded = (source, type, build, as) => {
- const holes = /** @type {DeferredEmbeddedSource[]} */ (_deferEmbeddedSource);
- const id = holes.length;
- holes.push({
- type,
- hostType: HTML_TYPE,
- source,
- build,
- ...(as === undefined ? undefined : { as })
- });
- return deferredWrite(id);
- };
- /**
- * `_renderEmbeddedOrNull` for a body webpack has no minifier of its own for, so
- * a renderer that declines leaves it exactly as written.
- * @param {string} source the nested body
- * @param {string} type its language, e.g. `"css"` / `"javascript"` / `"svg"`
- * @returns {string} the rendered body, or the original
- */
- const _renderEmbedded = (source, type) => {
- if (_deferEmbeddedSource !== undefined) {
- return _deferEmbedded(source, type, (rendered) =>
- typeof rendered === "string" ? rendered : source
- );
- }
- const rendered = _renderEmbeddedOrNull(source, type);
- return rendered === null ? source : rendered;
- };
- /**
- * Whether this element is the outermost `<svg>` of an SVG subtree — the whole
- * of which is offered to the renderer, so a nested one is part of its parent's.
- * @param {HtmlPath} path the accessor positioned on the element
- * @returns {boolean} true for an SVG root
- */
- const _isSvgRoot = (path) => {
- if (path.tagName() !== "svg" || path.namespace() !== NS_SVG) return false;
- const parent = path.parentOf();
- return parent === 0 || path.namespace(parent) !== NS_SVG;
- };
- // Which attribute defaults may be dropped (see `HtmlProcessOptions`), off
- // unless asked for: whichever tier drops one, the page stops carrying an
- // attribute that `getAttribute` and every attribute selector read.
- /** @type {"none" | "smart" | "all"} */
- let _removeRedundantAttributes = "none";
- // Whether an empty `class` / `id` / `style` / `dir` may go (see
- // `HtmlProcessOptions`), off unless asked for: a presence selector reads it.
- let _removeEmptyAttributes = false;
- // Whether a childless, attribute-less element may go (see
- // `HtmlProcessOptions`), off unless asked for: CSS can give one a size or a
- // `::before`, and this cannot see the stylesheet.
- let _removeEmptyElements = false;
- /**
- * Whether each element ends up empty in the output, memoized: the answer is
- * fixed for a parse, and the walk that finds it is the whole subtree.
- * @type {Map<HtmlNodeRef, boolean>}
- */
- const _emptyElements = new Map();
- // Which comments this print keeps, with a pattern already compiled to the
- // predicate it stands for — `true` every one, `false` none, `"some"` the ones
- // that carry something (see `_keepComment`).
- /** @type {boolean | "some" | ((comment: string) => boolean)} */
- let _commentsKept = "some";
- // The two reorderings (see `HtmlProcessOptions`), off unless asked for: neither
- // changes what anything matches, but both change what a script reading
- // `element.attributes` or `className` back sees.
- let _sortAttributes = false;
- let _sortTokenLists = false;
- // Whether a run of adjacent `<style>` elements prints as one (see
- // `HtmlProcessOptions`), off unless asked for: it removes elements, which
- // `document.styleSheets` and a `:nth-child` selector read.
- let _mergeStyles = false;
- // A run of adjacent `<style>` elements prints as one, so the elements after the
- // first print nothing at all. Filled by the first one's text as it prints, which
- // is before any of theirs does; null until a document actually has a run.
- /** @type {Set<HtmlElement> | null} */
- let _absorbedStyles = null;
- // Foster parenting puts a node before the table it was written inside, so tree
- // order and source order disagree there. Recorded where it happens.
- // The tree holds something no arrangement of tags reaches: an element no source
- // tag wrote, or a list item a special element stopped the loop from closing.
- let _printFromSource = false;
- /** @type {HtmlNodeRef[] | null} this parse's fosters, flat: table, then node */
- let _fosteredLog = null;
- /** @type {Map<HtmlNodeRef, HtmlNodeRef[]> | null} */
- let _fosteredRuns = null;
- /** @type {Map<HtmlNodeRef, HtmlNodeRef> | null} */
- let _fosteredNodes = null;
- /** @type {Set<HtmlNodeRef> | null} */
- let _fosterHolders = null;
- /** @type {Map<HtmlNodeRef, boolean> | null} */
- let _fosterMoves = null;
- // Tables whose run has to move but cannot be replayed by re-nesting it. Their own
- // source is what produced the tree, so it is what prints.
- /** @type {Set<HtmlNodeRef> | null} */
- let _fosterVerbatim = null;
- // That holder and everything still open around it: the slice runs to where the
- // source stopped, so an end tag none of them was given cannot be invented here.
- /** @type {Set<HtmlNodeRef> | null} */
- let _fosterOpenTail = null;
- /** @type {number} how far the furthest verbatim span reached */
- let _fosterSpanEnd = 0;
- // How much of the `<html>` / `<head>` / `<body>` shell §13.1.2.4 lets the
- // parser imply may be left out (see `HtmlProcessOptions`). Every *other*
- // optional tag goes unconditionally — nothing can observe that — but these six
- // are what a consumer reading the page with a regexp rather than a parser looks
- // for. `"smart"` leaves out the one such a reader never matches: the `<html>`
- // start tag, omittable only when it carries no attribute at all, so the
- // `<html lang=en>` anyone greps for keeps its tag anyway. `</html>` stays with
- // it — a truncation check reads a page as complete by finding one.
- /** @type {"none" | "smart" | "all"} */
- let _removeImpliedTags = "smart";
- /**
- * Whether the `<html>` / `<head>` / `<body>` start tag §13.1.2.4 lets the parser
- * imply may be left out. A comment minification keeps, whitespace, or an element
- * the parser still reads as head content would move above the tag once it goes;
- * a comment it drops is gone from the output, so it is skipped over.
- * @param {HtmlPath} path the accessor
- * @param {string} name the element's lowercased tag name
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when the start tag can be dropped
- */
- const _canOmitShellStart = (path, name, node) => {
- if (name !== "html" && name !== "head" && name !== "body") return false;
- if (path.attributeCount(node) !== 0) return false;
- let first = path.firstChild(node);
- while (first !== 0 && _printsNothing(path, first)) {
- first = path.nextSibling(first);
- }
- if (first === 0) return name === "html" || _removeImpliedTags === "all";
- const type = path.type(first);
- if (type === NodeType.Comment) return false;
- if (name === "html") return true;
- if (_removeImpliedTags !== "all") return false;
- if (name === "head") return type === NodeType.Element;
- if (type === NodeType.Text) return !startsWithWs(path.data(first));
- return !BODY_START_KEPT_BEFORE.has(path.tagName(first));
- };
- /**
- * Whether `</html>` / `</head>` / `</body>` can be left out without the parser
- * handing what follows to the element it closed: §4.13 lets an `html` or `body`
- * end tag go when a comment does not follow it, and a `head` one when neither a
- * comment nor whitespace does.
- * @param {HtmlPath} path the accessor
- * @param {string} name the element's lowercased tag name
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when nothing behind the tag would move inside
- */
- const _shellEndTagOmittable = (path, name, node) => {
- let next = path.nextSibling(node);
- while (next !== 0 && _printsNothing(path, next)) {
- next = path.nextSibling(next);
- }
- if (next === 0) return true;
- const type = path.type(next);
- if (type === NodeType.Comment) return false;
- return (
- name !== "head" || type !== NodeType.Text || !startsWithWs(path.data(next))
- );
- };
- /**
- * The same, for the end tag — `"all"` only: `"smart"` keeps every one of them,
- * `</html>` included, because a reader checking a page downloaded whole looks
- * for it. A comment behind the tag would move inside the element it closed,
- * and whitespace behind `</head>` into `<body>`.
- * @param {HtmlPath} path the accessor
- * @param {string} name the element's lowercased tag name
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when the end tag can be dropped
- */
- const _canOmitShellEnd = (path, name, node) =>
- _removeImpliedTags === "all" && _shellEndTagOmittable(path, name, node);
- /**
- * Whether a parser-inserted table group has to print a start tag the source
- * never spelled. With `removeOptionalTags` off its end tag prints, and a
- * `</tbody>` with no `<tbody>` is not the document that was parsed.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {string} name its lowercased tag name
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when the start tag has to be materialized
- */
- const _impliedTableGroupStartNeeded = (path, name, node) => {
- if (!_transforms.removeOptionalTags) {
- return name === "tbody" || name === "colgroup";
- }
- // §4.13 lets the start tag go only when the group before it kept its end
- // tag, or that one would take these columns over.
- return name === "colgroup" && !_canOmitColgroupStart(path, node);
- };
- /**
- * Whether an element's start tag may be left out: one the parser re-implies, or
- * a shell tag the option lets go. A shell element the source never spelled falls
- * through to the transparent-implied path, which already prints nothing for it.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {string} name its lowercased tag name
- * @param {HtmlNodeRef} node the element
- * @param {string} open what its start tag printed as, `""` when the source
- * spelled none
- * @returns {boolean} true when the start tag can be dropped
- */
- const _canOmitStartTag = (path, name, node, open) => {
- if (name === "tbody") {
- return _transforms.removeOptionalTags && _canOmitTbodyStart(path, node);
- }
- if (name === "colgroup") {
- return _transforms.removeOptionalTags && _canOmitColgroupStart(path, node);
- }
- return (
- open !== "" &&
- _removeImpliedTags !== "none" &&
- _canOmitShellStart(path, name, node)
- );
- };
- /**
- * Minify a `<style>` body with webpack's own CSS minifier — the one that runs on
- * CSS assets, so an inline sheet is held to the same rules. `null` when it
- * declines the text, which a caller joining two sheets has to tell apart from a
- * sheet that was already minimal: text the minifier cannot parse may be
- * unterminated, and then it swallows whatever is concatenated after it.
- * @param {string} css the element's text
- * @param {{ text: string }=} decided filled with the text this print may read back, which is the built-in's rather than the marker when the caller's renderer is deferred
- * @returns {string | null} the minified text, or null
- */
- const _minifyStyleBody = (css, decided) => {
- if (!_transforms.minifyStyles) return null;
- if (css.trim() === "") return css;
- try {
- if (_deferEmbeddedSource !== undefined) {
- // The built-in still runs: what this print does next is decided by
- // reading the minified text back (whether a run merges), and a marker
- // cannot be read. Only the text that reaches the output is the caller's,
- // and `decided` carries the readable text to whoever needs it.
- const builtin = _builtinStyleBody(css);
- if (builtin === null) return null;
- if (decided !== undefined) decided.text = builtin;
- return _deferEmbedded(css, CSS_TYPE, (rendered) =>
- typeof rendered === "string" ? rendered : builtin
- );
- }
- if (_renderEmbeddedSource !== undefined) {
- // Declining falls back to the built-in, so a renderer that covers some
- // languages does not turn the ones it does not cover off.
- const rendered = _renderEmbeddedOrNull(css, CSS_TYPE);
- if (rendered !== null) return rendered;
- }
- return _builtinStyleBody(css);
- } catch (_err) {
- return null;
- }
- };
- /**
- * The CSS minify webpack ships, for a body no caller's renderer answered for.
- * @param {string} css the element's text
- * @param {("stylesheet" | "block-contents")=} as which production it is — a `style=""` holds a block's contents, an element a stylesheet
- * @returns {string | null} the minified text, or null
- */
- const _builtinStyleBody = (css, as) => {
- try {
- const cssSyntax = _cssSyntax || (_cssSyntax = require("../css/syntax"));
- const options = {
- environment: _cssEnvironment,
- convertLengthUnits: _cssConvertLengthUnits,
- rewriteCustomProperties: _cssRewriteCustomProperties,
- transforms: _cssTransforms
- };
- return new cssSyntax.SourceProcessor().process(css, {
- mode: "minify",
- ...options,
- as
- }).code;
- } catch (_err) {
- return null;
- }
- };
- /**
- * `_minifyStyleBody`, with text it rejects handed back untouched rather than
- * failing the build.
- * @param {string} css the element's text
- * @returns {string} the minified text
- */
- const _minifyInlineCss = (css) => {
- const minified = _minifyStyleBody(css);
- return minified === null ? css : minified;
- };
- // The throwaway rule a `style` attribute's declaration list is minified inside.
- // The selector is only ever written and read back here, so it is the shortest
- // one the printer cannot rewrite — and the two halves are named so the read back
- // out cannot drift from what was written in.
- /**
- * How one attribute is written out, and whether it leaves the tag ending in an
- * unquoted value — the shortest of the minimal reference spelling, the value
- * bare, and the value quoted. `null` when none of them beats the source.
- * @param {string} attributeName the attribute's name
- * @param {string} value its rewritten value
- * @param {string} rawValue its value as the source spelled it
- * @param {string | null} decoded the value decoded, for one carrying character references, or null
- * @returns {{ text: string, unquotedTail: boolean } | null} what to emit, or null to leave the source alone
- */
- const _writeAttribute = (attributeName, value, rawValue, decoded) => {
- const minimal =
- _transforms.normalizeAttributeQuotes && decoded !== null
- ? _minimalAttributeValue(decoded)
- : null;
- const unquotable =
- _transforms.normalizeAttributeQuotes && _canUnquoteAttributeValue(value);
- if (
- minimal !== null &&
- minimal.length < (unquotable ? value.length : rawValue.length + 2)
- ) {
- const opened = minimal.charCodeAt(0);
- return {
- text: ` ${attributeName}=${minimal}`,
- unquotedTail: opened !== CC_QUOTATION_MARK && opened !== CC_APOSTROPHE
- };
- }
- if (unquotable) {
- return { text: ` ${attributeName}=${value}`, unquotedTail: true };
- }
- if (value !== rawValue) {
- const quotedValue = _quoteAttributeValue(value);
- if (quotedValue !== null) {
- return { text: ` ${attributeName}=${quotedValue}`, unquotedTail: false };
- }
- }
- return null;
- };
- /**
- * Minify a `style` attribute's declaration list with webpack's CSS minifier, as
- * the block-contents production it is — which is what gives the list the
- * duplicate-drop and shorthand merges a rule's block gets.
- * @param {string} raw raw (undecoded) attribute value
- * @returns {string} the minified declarations
- */
- const _minifyStyleAttribute = (raw) => {
- // Ahead of the memo: with the switch off nothing is minified, so there is
- // nothing to keep and nothing to read back.
- if (!_transforms.minifyStyles) return raw;
- // A caller's renderer is handed every `style=""` the document holds, so it is
- // not a function of the value alone and nothing about it may be reused.
- if (_renderEmbeddedSource !== undefined) {
- return _minifyStyleAttributeUncached(raw);
- }
- const memoized = _styleAttributeCache.get(raw);
- if (memoized !== undefined) return memoized;
- const minified = _minifyStyleAttributeUncached(raw);
- _styleAttributeCache.set(raw, minified);
- return minified;
- };
- /**
- * `_minifyStyleAttribute` without the memo — every path here runs the CSS
- * minifier, and every return is safe to keep, none of them reading state the
- * print goes on to change.
- * @param {string} raw raw (undecoded) attribute value
- * @returns {string} the minified declarations
- */
- const _minifyStyleAttributeUncached = (raw) => {
- if (raw.trim() === "") return raw;
- // Offered as the CSS it is, with `as` saying which production — the wrapper a
- // declaration list needs for the transforms a block's contents get is the CSS
- // layer's own business. The deferring path offers it at the attribute
- // instead, where the shape an answer has to land in is known.
- if (
- _renderEmbeddedSource !== undefined &&
- _deferEmbeddedSource === undefined
- ) {
- const rendered = _renderEmbeddedOrNull(raw, CSS_TYPE, BLOCK_CONTENTS);
- if (rendered !== null) return rendered;
- }
- const out = _builtinStyleBody(raw, BLOCK_CONTENTS);
- return out === null ? raw : out;
- };
- /**
- * Trim ASCII whitespace, which is all these grammars skip — `String#trim` also
- * eats NBSP and the other Unicode spaces, which they read as ordinary
- * characters (a leading NBSP makes an integer unparsable, and is a class token
- * of its own).
- * @param {string} value the value
- * @returns {string} the trimmed value
- */
- const _asciiTrim = (value) => {
- let start = 0;
- let end = value.length;
- while (start < end && isSpace(value.charCodeAt(start))) start++;
- while (end > start && isSpace(value.charCodeAt(end - 1))) end--;
- return start === 0 && end === value.length ? value : value.slice(start, end);
- };
- /**
- * Normalize a `<meta name=viewport>` content list: its grammar ignores the
- * whitespace around the `,` / `;` / `=` separators, so dropping that keeps every
- * pair intact.
- * @param {string} raw raw (undecoded) attribute value
- * @returns {string} the normalized list
- */
- const _normalizeViewport = (raw) =>
- _asciiTrim(raw)
- .replace(/[\t\n\f\r ]*([,;=])[\t\n\f\r ]*/g, "$1")
- .replace(/[\t\n\f\r ]+/g, " ");
- /**
- * Wrap a rewritten attribute value in the shortest quote it admits, or null when
- * it carries both kinds and the source spelling has to stand.
- * @param {string} value the value
- * @returns {string | null} the quoted value, or null
- */
- const _quoteAttributeValue = (value) => {
- if (!value.includes('"')) return `"${value}"`;
- if (!value.includes("'")) return `'${value}'`;
- return null;
- };
- /**
- * The shortest way to write a decoded value: bare where the grammar allows it,
- * else under whichever quote costs less. `<`, `>` and the quote that is not the
- * delimiter carry no meaning inside a quoted value, so `alt="say "hi""`
- * is `alt='say "hi"'` and `title="<b>"` is `title="<b>"`.
- * @param {string} value the decoded attribute value
- * @returns {string} the value as written, delimiters included
- */
- const _minimalAttributeValue = (value) => {
- const underDouble = escapeAttribute(value, CC_QUOTATION_MARK, true);
- const underSingle = escapeAttribute(value, CC_APOSTROPHE, true);
- let best = `"${underDouble}"`;
- if (underSingle.length + 2 < best.length) best = `'${underSingle}'`;
- // A bare value costs no delimiters, so it wins whenever it is legal at all.
- for (const bare of [underDouble, underSingle]) {
- if (bare.length < best.length && _canUnquoteAttributeValue(bare)) {
- best = bare;
- }
- }
- return best;
- };
- /**
- * Whether an attribute only restates the element's own default, so dropping it
- * leaves the DOM reading the same value — but leaves an attribute selector,
- * which matches the content attribute, with nothing to match. Both tables are
- * looked up per element by the caller, not per attribute here: the answer is the
- * same for every attribute of one tag.
- * @param {Record<string, string> | undefined} markers the element's marker table
- * @param {Record<string, string> | undefined} defaults the element's spec-default table
- * @param {string} tagName lowercased element name
- * @param {string} name lowercased attribute name
- * @param {string} value the attribute's raw value
- * @returns {boolean} true when the attribute may be dropped
- */
- const _isRedundantAttribute = (markers, defaults, tagName, name, value) => {
- const lowered = _asciiLowerCase(_asciiTrim(value));
- // `<script type>`: only a spelling that names JavaScript, which is what the
- // element already runs. `module` and the empty spelling say something.
- if (tagName === "script" && name === "type") {
- return (
- lowered !== "" &&
- lowered !== "module" &&
- JAVASCRIPT_SCRIPT_TYPES.has(lowered)
- );
- }
- if (markers !== undefined && markers[name] === lowered) return true;
- return defaults !== undefined && defaults[name] === lowered;
- };
- /**
- * Whether an empty attribute reads as its own absence, so `removeEmptyAttributes`
- * may drop it — only on the elements the spec defines it on, never `<x-foo rel="">`.
- * @param {string} elementName lowercased element name
- * @param {string} name lowercased attribute name
- * @param {string} value the attribute's raw value
- * @returns {boolean} true when the attribute may be dropped
- */
- const _isRemovableEmptyAttribute = (elementName, name, value) => {
- if (!_removeEmptyAttributes) return false;
- const on = EMPTY_REMOVABLE_ATTRIBUTES.get(name);
- if (on === undefined || (on !== null && !on.has(elementName))) return false;
- // Read raw, so a value spelled with a character reference is kept whatever it
- // decodes to. `_asciiTrim` returns the input when there is nothing to trim.
- return _asciiTrim(value) === "";
- };
- /**
- * Whether an element carries nothing at all, so `removeEmptyElements` may drop
- * it. Every guard but the name table is a rule: a void element is childless by
- * definition, a `<template>`'s children hang off a content fragment, foreign
- * content is not ours to judge, and an element spelling any attribute was
- * written for a reason — which is what keeps `<script src>`, `<iframe src>` and
- * `<div id=mount>`, each of which html-minifier-terser drops.
- *
- * What no guard can cover is CSS: `.spacer { height: 4px }` or a `::before`
- * makes an empty element visible, and this cannot see the stylesheet. That is
- * the risk the option is off by default for.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {HtmlElement} element the element
- * @returns {boolean} true when the element may be dropped
- */
- // Every test is pure, so the O(1) ones come first: `_childrenPrintNothing`
- // walks the subtree, and the printer asks this of every element. Memoized for
- // the same reason — without it a nested run costs one walk per ancestor.
- const _isRemovableEmptyElement = (path, element) => {
- if (!_removeEmptyElements) return false;
- // Asked before the memo: on component markup nearly every element answers
- // here, and the count costs less than the memo round trip.
- if (path.attributeCount(element) !== 0) return false;
- const memo = _emptyElements.get(element);
- if (memo !== undefined) return memo;
- const answer =
- path.templateContent(element) === 0 &&
- path.namespace(element) === NS_HTML &&
- // Void is otherwise disqualifying — a `<br>` is childless and still
- // renders. A void element that belongs in the head is the exception: with
- // no attributes it states nothing.
- (!VOID.has(path.tagName(element)) ||
- EMPTY_METADATA_ELEMENTS.has(path.tagName(element))) &&
- !EMPTY_ELEMENT_KEPT.has(path.tagName(element)) &&
- _childrenPrintNothing(path, element);
- _emptyElements.set(element, answer);
- return answer;
- };
- /**
- * Whether every child of an element prints nothing, so the element ends up
- * empty in the output whatever the source tree holds. Recursive with
- * `_isRemovableEmptyElement`, which is what lets a run of nested empties go
- * together rather than one layer per build.
- * @param {HtmlPath} path the accessor
- * @param {HtmlElement} element the element
- * @returns {boolean} true when nothing inside it reaches the output
- */
- const _childrenPrintNothing = (path, element) => {
- for (
- let child = _nodeFirstChildren[element];
- child !== 0;
- child = path.nextSibling(child)
- ) {
- if (
- path.type(child) === NodeType.Element
- ? !_isRemovableEmptyElement(path, /** @type {HtmlElement} */ (child))
- : !_printsNothing(path, child)
- ) {
- return false;
- }
- }
- return true;
- };
- /**
- * Whether an attribute is boolean on this element, so its presence alone is the
- * value. `null` in the table means the attribute is global.
- * @param {string} tagName lowercased element name
- * @param {string} name lowercased attribute name
- * @returns {boolean} true when the value carries nothing
- */
- const _isBooleanAttribute = (tagName, name) => {
- const elements = BOOLEAN_ATTRIBUTES.get(name);
- if (elements === undefined) return BOOLEAN_ATTRIBUTES.has(name);
- return elements === null || elements.has(tagName);
- };
- /**
- * Whether a space-separated list names the same token twice, read span by span
- * so nothing is allocated. The lists are short — `class` is a handful of tokens
- * — so comparing each pair beats building a set of them.
- * @param {string} list a collapsed, trimmed token list
- * @returns {boolean} true when some token appears more than once
- */
- const _repeatsAToken = (list) => {
- const length = list.length;
- for (let start = 0; start < length;) {
- let end = list.indexOf(" ", start);
- if (end === -1) end = length;
- const width = end - start;
- for (let other = end + 1; other < length;) {
- let otherEnd = list.indexOf(" ", other);
- if (otherEnd === -1) otherEnd = length;
- if (otherEnd - other === width) {
- let i = 0;
- while (
- i < width &&
- list.charCodeAt(start + i) === list.charCodeAt(other + i)
- ) {
- i++;
- }
- if (i === width) return true;
- }
- other = otherEnd + 1;
- }
- start = end + 1;
- }
- return false;
- };
- /**
- * Collapse an ASCII-whitespace-separated token list (`class`) to single spaces,
- * and — where the DOM reflects the list as a `DOMTokenList`, an ordered *set*,
- * so a repeat was never a second token — drop a token already in it. Order is
- * kept either way: the first spelling of each token stays put.
- * @param {string} raw raw (undecoded) value
- * @param {boolean} unique whether a repeated token may go (`DOM_TOKEN_LIST_ATTRIBUTES`)
- * @param {boolean} sort whether to print the tokens in order (`sortTokenLists`)
- * @returns {string} the collapsed list
- */
- const _normalizeTokenList = (raw, unique, sort) => {
- // One walk answers "already collapsed?" — the common already-minified list
- // rewrites nothing, so it must not pay the trim + regex replace either.
- let dirty = false;
- let hasSpace = false;
- let previousSpace = true;
- for (let i = 0; i < raw.length; i++) {
- const c = raw.charCodeAt(i);
- if (
- (c >= 0x09 && c <= 0x0d && c !== 0x0b) ||
- (c === 0x20 && previousSpace)
- ) {
- dirty = true;
- break;
- }
- if (c === 0x20) {
- hasSpace = true;
- previousSpace = true;
- } else {
- previousSpace = false;
- }
- }
- if (!dirty && previousSpace && raw.length !== 0) dirty = true;
- const collapsed = dirty ? _asciiTrim(raw).replace(/[\t\n\f\r ]+/g, " ") : raw;
- if (dirty ? !collapsed.includes(" ") : !hasSpace) return collapsed;
- // Cutting the list into tokens is the allocation this runs often enough to
- // care about, and almost no list repeats one — so the check reads the spans
- // in place and only a list that does pay for the pieces. Sorting needs them
- // either way.
- if (!sort && !(unique && _repeatsAToken(collapsed))) return collapsed;
- if (sort) return _sortedTokenList(collapsed, unique);
- return [...new Set(collapsed.split(" "))].join(" ");
- };
- // One list's token spans, reused: ordering a list allocates only its result.
- /** @type {number[]} */
- const _tokenSpanStarts = [];
- /** @type {number[]} */
- const _tokenSpanEnds = [];
- /**
- * Compare two of `list`'s tokens where they sit, in the code unit order
- * `Array#sort` puts strings in — a prefix sorts before what extends it.
- * @param {string} list the collapsed list
- * @param {number} aStart first token's start
- * @param {number} aEnd first token's end
- * @param {number} bStart second token's start
- * @param {number} bEnd second token's end
- * @returns {number} negative, zero or positive
- */
- const _compareTokens = (list, aStart, aEnd, bStart, bEnd) => {
- const shared = Math.min(aEnd - aStart, bEnd - bStart);
- for (let i = 0; i < shared; i++) {
- const difference =
- list.charCodeAt(aStart + i) - list.charCodeAt(bStart + i);
- if (difference !== 0) return difference;
- }
- return aEnd - aStart - (bEnd - bStart);
- };
- /**
- * `collapsed`'s tokens in sorted order, deduplicated when it is a set. They are
- * ordered where they sit, so the pieces `split` would build are never built.
- * @param {string} collapsed a single-space-separated list
- * @param {boolean} unique whether repeats collapse
- * @returns {string} the ordered list
- */
- const _sortedTokenList = (collapsed, unique) => {
- const starts = _tokenSpanStarts;
- const ends = _tokenSpanEnds;
- let count = 0;
- for (let at = 0; ;) {
- const space = collapsed.indexOf(" ", at);
- starts[count] = at;
- ends[count] = space === -1 ? collapsed.length : space;
- count++;
- if (space === -1) break;
- at = space + 1;
- }
- // A list carries a handful of tokens, so insertion sort beats `Array#sort`'s
- // setup and leaves equal tokens adjacent for the pass below to drop.
- for (let i = 1; i < count; i++) {
- const start = starts[i];
- const end = ends[i];
- let j = i - 1;
- while (
- j >= 0 &&
- _compareTokens(collapsed, starts[j], ends[j], start, end) > 0
- ) {
- starts[j + 1] = starts[j];
- ends[j + 1] = ends[j];
- j--;
- }
- starts[j + 1] = start;
- ends[j + 1] = end;
- }
- let out = "";
- for (let i = 0; i < count; i++) {
- if (
- unique &&
- i !== 0 &&
- _compareTokens(
- collapsed,
- starts[i - 1],
- ends[i - 1],
- starts[i],
- ends[i]
- ) === 0
- ) {
- continue;
- }
- if (out !== "") out += " ";
- out += collapsed.slice(starts[i], ends[i]);
- }
- return out;
- };
- /**
- * Drop the whitespace a comma-separated list's grammar ignores: "split a string
- * on commas" strips each token's edges and nothing else. Whitespace inside a
- * token stays — it is part of the token `accept` keeps, and a separator of its
- * own in the number list `coords` parses.
- * @param {string} raw raw (undecoded) value
- * @returns {string} the normalized list
- */
- const _normalizeCommaList = (raw) =>
- raw
- .split(",")
- .map((item) => _asciiTrim(item))
- .join(",");
- /**
- * Normalize an integer attribute: the parse rules skip leading ASCII whitespace
- * and stop at the first non-digit, so the ends and any leading zeros carry
- * nothing. A value those rules would not read as one integer is left alone —
- * only the signed rules accept a `+` or `-`, so a signed value on a
- * non-negative attribute stays as it is (it does not parse at all).
- * @param {string} raw raw (undecoded) value
- * @param {boolean} signed whether the attribute parses with the signed rules
- * @returns {string} the normalized integer
- */
- const _normalizeInteger = (raw, signed) => {
- const value = _asciiTrim(raw);
- if (!(signed ? /^[+-]?\d+$/ : /^\d+$/).test(value)) return raw;
- const negative = value.charCodeAt(0) === 0x2d;
- const digits = value.replace(/^[+-]/, "").replace(/^0+(?=\d)/, "");
- return negative && digits !== "0" ? `-${digits}` : digits;
- };
- // The `*` bucket, read on every attribute the element does not scope itself.
- const ENUMERATED_GLOBAL_KEYWORDS = ENUMERATED_KEYWORDS["*"];
- /**
- * The value rewrites that keep an attribute's parsed meaning: a srcset's
- * whitespace, a `style` declaration list, and the viewport `content` list.
- * @param {string} element lowercased element name
- * @param {string} name lowercased attribute name
- * @param {string} raw raw (undecoded) value
- * @param {boolean} viewport whether the element is `<meta name=viewport>`
- * @param {Record<string, Set<string>> | undefined} enumeratedOn what the element scopes itself, looked up once per element by the caller
- * @returns {string} the rewritten value, or `raw` when nothing applies
- */
- const _rewriteAttributeValue = (element, name, raw, viewport, enumeratedOn) => {
- // One question for the whole chain below, and most attributes on a page — a
- // `data-*`, an `id`, an `aria-*` — are done here rather than after a miss in
- // each table in turn.
- if (!REWRITABLE_ATTRIBUTES.has(name)) return raw;
- // An enumerated value matches ASCII-case-insensitively, so folding one the
- // spec names changes nothing the DOM reads.
- if (
- _transforms.normalizeEnumeratedAttributes &&
- ENUMERATED_ATTRIBUTE_NAMES.has(name)
- ) {
- const own = enumeratedOn === undefined ? undefined : enumeratedOn[name];
- const enumerated =
- own === undefined ? ENUMERATED_GLOBAL_KEYWORDS[name] : own;
- if (enumerated !== undefined) {
- // Case-insensitively, and nothing else: an enumerated attribute names a
- // keyword, so `type="time "` is not `time` — it is the value no keyword
- // matches, which puts the input in its default state.
- const lowered = _asciiLowerCase(raw);
- // A value the spec does not enumerate is left exactly as written:
- // `target` names a browsing context, and `type` on a custom element
- // means what it says.
- if (enumerated.has(lowered)) return lowered;
- }
- }
- if (SRCSET_ATTRIBUTES.has(name)) {
- return _transforms.normalizeListAttributes ? _normalizeSrcset(raw) : raw;
- }
- if (name === "style") return _minifyStyleAttribute(raw);
- const tokenListOn = _transforms.normalizeListAttributes
- ? TOKEN_LIST_ATTRIBUTES.get(name)
- : undefined;
- if (
- tokenListOn !== undefined &&
- (tokenListOn === null || tokenListOn.has(element))
- ) {
- // Only the lists the DOM reflects as a `DOMTokenList` are sets, so only
- // they may be reordered — `ping` is the order its requests go out in and
- // `accesskey` the order its keys are tried.
- const isSet = DOM_TOKEN_LIST_ATTRIBUTES.has(name);
- return _normalizeTokenList(raw, isSet, _sortTokenLists && isSet);
- }
- if (_transforms.normalizeListAttributes) {
- if (viewport && name === "content") return _normalizeViewport(raw);
- if (COMMA_LIST_ATTRIBUTES.has(name)) return _normalizeCommaList(raw);
- }
- const urlOn = URL_ATTRIBUTES.get(name);
- if (urlOn !== undefined && (urlOn === null || urlOn.has(element))) {
- return _asciiTrim(raw);
- }
- const integerOn = _transforms.normalizeNumericAttributes
- ? INTEGER_ATTRIBUTES.get(name)
- : undefined;
- if (
- integerOn !== undefined &&
- (integerOn === null || integerOn.has(element))
- ) {
- return _normalizeInteger(raw, SIGNED_INTEGER_ATTRIBUTES.has(name));
- }
- return raw;
- };
- /**
- * A `<script>`'s lowercased `type`, or `""` when it has none.
- * @param {HtmlPath} path the accessor
- * @param {HtmlElement} element the `<script>` element
- * @returns {string} the type
- */
- const _scriptType = (path, element) => {
- for (const attribute of path.attributes(element)) {
- if (attribute.name === "type") {
- return _asciiLowerCase(_asciiTrim(attribute.value));
- }
- }
- return "";
- };
- /**
- * Whether an element is the viewport `<meta>`, whose content list is the only
- * `content` this rewrites.
- * @param {HtmlPath} path the accessor positioned on the `<meta>`
- * @returns {boolean} true for `<meta name=viewport>`
- */
- const _isViewportMeta = (path) => {
- for (const attribute of path.attributes()) {
- if (attribute.name === "name") {
- return _asciiLowerCase(_asciiTrim(attribute.value)) === "viewport";
- }
- }
- return false;
- };
- const _JSON_SUBTYPE_REGEXP =
- /^[!#$%&'*+.^_`|~\w-]+\/[!#$%&'*+.^_`|~\w-]*\+json$/;
- /**
- * Strip the whitespace between a JSON body's tokens. Every literal is copied
- * byte for byte — re-serializing would reorder nothing but would round numbers
- * through a double, drop a duplicate key and rewrite escapes, none of which is
- * this transform's to do. Escapes are also the security case: unescaping a
- * `<` lets a string close the `<script>` early.
- * @param {string} json a `<script>` body
- * @returns {string} the stripped body
- */
- const _minifyInlineJson = (json) => {
- if (!_transforms.minifyJson) return json;
- if (json.trim() === "") return json;
- if (_deferEmbeddedSource !== undefined) {
- return _deferEmbedded(json, JSON_TYPE, (rendered) =>
- typeof rendered === "string" ? rendered : _builtinInlineJson(json)
- );
- }
- if (_renderEmbeddedSource !== undefined) {
- const rendered = _renderEmbeddedOrNull(json, JSON_TYPE);
- if (rendered !== null) return rendered;
- }
- return _builtinInlineJson(json);
- };
- /**
- * The JSON strip webpack ships, for a body no caller's renderer answered for.
- * @param {string} json a `<script>` body
- * @returns {string} the stripped body
- */
- const _builtinInlineJson = (json) => {
- try {
- JSON.parse(json);
- } catch (_err) {
- // Not JSON after all (a template, a placeholder) — not ours to touch.
- return json;
- }
- let out = "";
- let inString = false;
- let escaped = false;
- for (let i = 0; i < json.length; i++) {
- const c = json.charCodeAt(i);
- if (inString) {
- out += json[i];
- if (escaped) escaped = false;
- else if (c === 0x5c) escaped = true;
- else if (c === 0x22) inString = false;
- continue;
- }
- if (c === 0x22) {
- inString = true;
- out += json[i];
- continue;
- }
- // JSON whitespace (RFC 8259): tab, LF, CR, space.
- if (c === 0x09 || c === 0x0a || c === 0x0d || c === 0x20) continue;
- out += json[i];
- }
- return out;
- };
- /**
- * Whether a `<style>` element's body is CSS: the attribute is optional, and the
- * only value that keeps it CSS is `text/css`.
- * @param {HtmlPath} path the accessor
- * @param {HtmlElement} element the `<style>` element
- * @returns {boolean} true when the body may be minified as CSS
- */
- const _isCssStyleElement = (path, element) => {
- for (const attribute of path.attributes(element)) {
- if (attribute.name !== "type") continue;
- const value = _asciiLowerCase(_asciiTrim(attribute.value));
- return value === "" || value === "text/css";
- }
- return true;
- };
- // An at-rule that only applies at the top of a sheet, so a second sheet carrying
- // one cannot be appended to a first.
- const _SHEET_LEADING_AT_RULE_REGEXP = /@(?:charset|import|namespace)\b/i;
- /**
- * A `<style>`'s only child when that is one text node — the shape a merge needs,
- * since `<style>` is parsed as raw text and anything else means it is empty.
- * @param {HtmlElement} element the `<style>` element
- * @returns {HtmlNodeRef} the text node, or 0
- */
- const _styleTextChild = (element) => {
- const first = _nodeFirstChildren[element];
- return first !== 0 &&
- _nodeTypes[first] === NodeType.Text &&
- _nodeNextSiblings[first] === 0
- ? first
- : 0;
- };
- /**
- * Whether two elements carry exactly the same attributes. Values compare raw:
- * two spellings of one value are rare enough not to be worth decoding for, and
- * reading them as different only declines a merge.
- * @param {HtmlPath} path the accessor
- * @param {HtmlElement} one an element
- * @param {HtmlElement} other another
- * @returns {boolean} true when the two agree on every attribute
- */
- const _sameAttributes = (path, one, other) => {
- const count = path.attributeCount(one);
- if (count !== path.attributeCount(other)) return false;
- for (let i = 0; i < count; i++) {
- const attribute = path.attributeAt(i, one);
- const match = path.findAttribute(path.attributeName(attribute), other);
- if (
- match === 0 ||
- path.attributeValue(match) !== path.attributeValue(attribute)
- ) {
- return false;
- }
- }
- return true;
- };
- /**
- * The `<style>` that may be folded into `element`: the next sibling, when only
- * whitespace lies between them and the two agree on every attribute. Adjacency
- * is what makes the fold safe — nothing moves past anything, so the cascade is
- * the one the source wrote. Media, `title` and `blocking` all ride on the
- * attribute test, and `type` on both being CSS in the first place.
- * @param {HtmlPath} path the accessor
- * @param {HtmlElement} element a `<style>` element
- * @returns {HtmlElement} the next `<style>` of the run, or 0
- */
- const _mergeableStyleAfter = (path, element) => {
- for (
- let next = _nodeNextSiblings[element];
- next !== 0;
- next = _nodeNextSiblings[next]
- ) {
- const type = _nodeTypes[next];
- if (type === NodeType.Text) {
- if (!isAllWs(_nodeStrings[next])) return 0;
- continue;
- }
- return type === NodeType.Element &&
- path.namespace(next) === NS_HTML &&
- path.tagName(next) === "style" &&
- _styleTextChild(next) !== 0 &&
- _isCssStyleElement(path, next) &&
- _sameAttributes(path, element, next)
- ? next
- : 0;
- }
- return 0;
- };
- /**
- * The text the first `<style>` of a run prints: every sheet in it, each minified
- * on its own and then joined. Marks the rest of the run absorbed, which is what
- * makes their tags and text print as nothing — it runs first because a run's
- * first element prints before any of the others in both print paths.
- *
- * Each sheet is minified before it is joined, never after: only text the
- * minifier accepted is known to be terminated, and appending to a sheet that is
- * not makes the next one part of the last rule of this one. A sheet after the
- * first carrying `@import` (or `@charset` / `@namespace`) is not folded either —
- * those only apply at the top of a sheet, and would silently stop applying.
- * @param {HtmlPath} path the accessor
- * @param {HtmlElement} element the run's first `<style>`
- * @param {string} css its own text
- * @returns {string} the run's text
- */
- const _mergedStyleRun = (path, element, css) => {
- const head = _minifyStyleBody(css);
- if (head === null) return css;
- let out = head;
- for (
- let next = _mergeableStyleAfter(path, element);
- next !== 0;
- next = _mergeableStyleAfter(path, next)
- ) {
- const decided = { text: "" };
- const body = _minifyStyleBody(
- _nodeStrings[/** @type {HtmlNodeRef} */ (_styleTextChild(next))],
- decided
- );
- if (body === null) break;
- // Read the built-in's text, not the marker: whether this sheet may be
- // absorbed decides what its own element prints, so it cannot wait for an
- // answer.
- if (_SHEET_LEADING_AT_RULE_REGEXP.test(decided.text || body)) break;
- if (_absorbedStyles === null) _absorbedStyles = new Set();
- _absorbedStyles.add(next);
- out += body;
- }
- return out;
- };
- /**
- * Re-serialize a srcset with its whitespace reduced to what the grammar needs.
- * The candidate list is unchanged — only the bytes between candidates go — and
- * anything the parser rejects is handed back untouched.
- * @param {string} raw raw (undecoded) attribute value
- * @returns {string} the normalized value
- */
- const _normalizeSrcset = (raw) => {
- let candidates;
- try {
- candidates = parseSrcset(raw);
- } catch (_err) {
- return raw;
- }
- if (candidates.length === 0) return raw;
- let out = "";
- let previousBare = false;
- for (let i = 0; i < candidates.length; i++) {
- const url = candidates[i][0];
- const end = candidates[i][2];
- const nextStart =
- i + 1 < candidates.length ? candidates[i + 1][1] : raw.length;
- let descriptor = raw.slice(end, nextStart);
- const comma = descriptor.indexOf(",");
- if (comma !== -1) descriptor = descriptor.slice(0, comma);
- descriptor = _asciiTrim(descriptor).replace(/[\t\n\f\r ]+/g, " ");
- // A descriptor-less URL runs to the next whitespace, so without one it
- // would swallow the separating comma and the URL behind it.
- if (i !== 0) out += previousBare ? ", " : ",";
- out += descriptor === "" ? url : `${url} ${descriptor}`;
- previousBare = descriptor === "";
- }
- return out;
- };
- /**
- * ASCII-lowercase a name or a keyword value, which is all HTML folds — a Unicode
- * `toLowerCase` maps characters the parser leaves alone (U+212A onto `k`).
- * Returns the input untouched when there is nothing to fold (the common case).
- * @param {string} value source name or value
- * @returns {string} the value as the parser would match it
- */
- const _asciiLowerCase = (value) => {
- for (let i = 0; i < value.length; i++) {
- const c = value.charCodeAt(i);
- if (c >= 0x41 && c <= 0x5a) {
- return (
- value.slice(0, i) +
- value
- .slice(i)
- .replace(/[A-Z]/g, (u) => String.fromCharCode(u.charCodeAt(0) + 0x20))
- );
- }
- }
- return value;
- };
- /**
- * Whether a raw attribute value keeps its meaning unquoted: §13.1.2.3 admits a
- * non-empty run with no whitespace, quote, backtick, `=`, `<` or `>`. Character
- * references decode identically either way, so `&` needs no guard.
- * @param {string} raw raw (undecoded) value, without its quotes
- * @returns {boolean} true when the unquoted form re-parses to the same value
- */
- const _canUnquoteAttributeValue = (raw) => {
- for (let i = 0; i < raw.length; i++) {
- const c = raw.charCodeAt(i);
- if (
- c === 0x20 ||
- c === 0x22 ||
- c === 0x27 ||
- c === 0x3c ||
- c === 0x3d ||
- c === 0x3e ||
- c === 0x60 ||
- (c >= 0x09 && c <= 0x0d)
- ) {
- return false;
- }
- }
- return raw.length !== 0;
- };
- // How often each attribute name occurs in the document being printed, or null
- // until `sortAttributes` asks for it. Built from the parse's own attribute
- // column rather than by walking the tree: every attribute of every element is
- // already interned there, in one flat run.
- /** @type {Map<string, number> | null} */
- let _attributeFrequencies = null;
- // Reused by `_sortedAttributes`: one element's attribute indices and their
- // ranks, so ordering a tag allocates nothing.
- /** @type {number[]} */
- const _sortOrderScratch = [];
- /** @type {number[]} */
- const _sortKeyScratch = [];
- // The print order is a property of the document, so resolving a name to its
- // rank once leaves the per-element sort comparing two integers.
- /** @type {Map<string, number> | null} */
- let _attributeRanks = null;
- /**
- * @returns {Map<string, number>} each attribute name's place in the print order
- */
- const _attributeRank = () => {
- if (_attributeRanks !== null) return _attributeRanks;
- const frequency = _attributeFrequency();
- const names = [...frequency.keys()].sort((a, b) => {
- // Commonest names first, so the run two elements share is one run of bytes
- // a compressor pays for once. Name order breaks ties, keeping this total.
- const byFrequency =
- /** @type {number} */ (frequency.get(b)) -
- /** @type {number} */ (frequency.get(a));
- if (byFrequency !== 0) return byFrequency;
- return a < b ? -1 : 1;
- });
- const ranks = new Map();
- for (let i = 0; i < names.length; i++) ranks.set(names[i], i);
- _attributeRanks = ranks;
- return ranks;
- };
- /**
- * @returns {Map<string, number>} how often each attribute name occurs
- */
- const _attributeFrequency = () => {
- if (_attributeFrequencies !== null) return _attributeFrequencies;
- const frequency = new Map();
- for (let i = 1; i <= _attrCount; i++) {
- const name = _attrNames[i];
- frequency.set(name, (frequency.get(name) || 0) + 1);
- }
- _attributeFrequencies = frequency;
- return frequency;
- };
- /**
- * The element's attribute indices in the order `sortAttributes` prints them:
- * the document's commonest names first, ties by name. Nothing in HTML reads
- * attribute order, so a stable spelling across pages is worth the repeats it
- * hands a compressor.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {number} count its attribute count
- * @returns {number[]} the indices, in print order
- */
- const _sortedAttributes = (path, count) => {
- const ranks = _attributeRank();
- const order = _sortOrderScratch;
- const keys = _sortKeyScratch;
- for (let i = 0; i < count; i++) {
- order[i] = i;
- // The ranks were counted off the same column this name is read from, so
- // every name printed here has one.
- keys[i] = /** @type {number} */ (
- ranks.get(path.attributeName(path.attributeAt(i)))
- );
- }
- // An element carries a handful of attributes, so a stable insertion sort over
- // their ranks beats `Array#sort`'s setup and the comparator it would call.
- for (let i = 1; i < count; i++) {
- const key = keys[i];
- const index = order[i];
- let j = i - 1;
- while (j >= 0 && keys[j] > key) {
- keys[j + 1] = keys[j];
- order[j + 1] = order[j];
- j--;
- }
- keys[j + 1] = key;
- order[j + 1] = index;
- }
- return order;
- };
- /**
- * Rewrite an opening tag from its own source spans, collapsing the whitespace
- * between its attributes to one space and dropping the quotes around any value
- * that is legal unquoted (§13.1.2.3). webpack's own late asset passes match the
- * sentinels they left behind with the quote optional, so they still see them
- * (see `INTEGRITY_SENTINEL_REGEXP`); a third-party post-processor that regexes
- * emitted HTML assuming quotes would not, which is why `removeAttributeQuotes`
- * is off by default in html-minifier.
- *
- * Falls back to the source tag when an attribute lies outside it — a repeated
- * `<html>` / `<body>` tag merges its attributes onto the element already open,
- * and those cannot be sliced from here.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {string} open the opening tag's source text
- * @returns {string} the minified opening tag
- */
- const _minifyOpenTag = (path, open) => {
- const tagStart = path.start();
- const tagEnd = path.tagEnd();
- const count = path.attributeCount();
- // HTML parsing ASCII-lowercases names, so folding one changes nothing. Foreign
- // content keeps its source bytes: SVG admits element names no adjustment table
- // would case-correct on the way back in.
- const html = path.namespace() === NS_HTML;
- const nameEnd = path.nameEnd() - tagStart;
- // A bare `<name>` — half of all tags — is what the rest would rebuild byte
- // for byte; decided before the name is sliced, so it allocates nothing.
- if (count === 0 && open.length === nameEnd + 1) {
- let folds = false;
- if (html) {
- for (let i = 1; i < nameEnd; i++) {
- const c = open.charCodeAt(i);
- if (c >= 0x41 && c <= 0x5a) {
- folds = true;
- break;
- }
- }
- }
- if (!folds) return open;
- }
- const sourceName = open.slice(1, nameEnd);
- const elementName = html ? _asciiLowerCase(sourceName) : sourceName;
- let out = `<${elementName}`;
- const isViewportMeta =
- html && path.tagName() === "meta" && _isViewportMeta(path);
- let unquotedTail = false;
- let consumedEnd = path.nameEnd();
- // HTML only, which is also what keeps `consumedEnd` meaningful: it is read
- // once more below, and only for a foreign element's `/>`.
- // Per element, not per attribute: every attribute of one tag gets the same
- // answer out of these, and they are the hottest lookups in the printer.
- const markers =
- html && _removeRedundantAttributes !== "none"
- ? REDUNDANT_TYPE_ATTRIBUTES[elementName]
- : undefined;
- const defaults =
- html && _removeRedundantAttributes === "all"
- ? REDUNDANT_DEFAULT_ATTRIBUTES[elementName]
- : undefined;
- const mayDrop = markers !== undefined || defaults !== undefined;
- const enumeratedOn = html ? ENUMERATED_KEYWORDS[elementName] : undefined;
- const order =
- _sortAttributes && html && count > 1
- ? _sortedAttributes(path, count)
- : null;
- for (let k = 0; k < count; k++) {
- const i = order === null ? k : order[k];
- const attribute = path.attributeAt(i);
- const nameStart = path.attributeNameStart(attribute);
- const nameEnd = path.attributeNameEnd(attribute);
- if (nameStart < tagStart || nameEnd > tagEnd) return open;
- // The parse interned exactly `_asciiLowerCase(<name slice>)` — read it back
- // rather than re-slicing and re-folding. Foreign content keeps source case.
- const attributeName = html
- ? path.attributeName(attribute)
- : open.slice(nameStart - tagStart, nameEnd - tagStart);
- const valueStart = path.attributeValueStart(attribute);
- if (valueStart === -1) {
- // Valueless is the empty value, so it goes with the empty ones.
- if (html && _isRemovableEmptyAttribute(elementName, attributeName, "")) {
- consumedEnd = nameEnd;
- continue;
- }
- out += ` ${attributeName}`;
- unquotedTail = false;
- consumedEnd = nameEnd;
- continue;
- }
- // `attributeValueStart` points past the opening quote, `attributeValueEnd`
- // at the closing one, so a quoted value's span is one wider on each side.
- const quote = open.charCodeAt(valueStart - 1 - tagStart);
- const quoted = quote === 34 || quote === 39;
- const rawEnd = path.attributeValueEnd(attribute);
- const valueEnd = rawEnd + (quoted ? 1 : 0);
- const rawValue = open.slice(valueStart - tagStart, rawEnd - tagStart);
- // Dropped before any rewrite, and whether or not the source quoted it:
- // the value only restates the default, or carries nothing at all.
- if (
- html &&
- ((mayDrop &&
- _isRedundantAttribute(
- markers,
- defaults,
- elementName,
- attributeName,
- rawValue
- )) ||
- _isRemovableEmptyAttribute(elementName, attributeName, rawValue))
- ) {
- consumedEnd = valueEnd;
- continue;
- }
- if (quoted) {
- const hasReference = rawValue.includes("&");
- // A whole document, so it is offered decoded and written back escaped —
- // the one attribute whose reference-carrying value is still rewritten,
- // since a document is mostly references. The renderer cannot parse it
- // here (the walk is mid-parse), so it looks up what a walk-only pass
- // prepared, exactly as an async minifier does.
- if (
- html &&
- (_renderEmbeddedSource !== undefined ||
- (_deferEmbeddedSource !== undefined && _deferSrcdoc)) &&
- attributeName === "srcdoc" &&
- elementName === "iframe" &&
- rawValue !== ""
- ) {
- const document = decodeEntities(rawValue, true);
- // The whole attribute is deferred, not the document inside it: what
- // escaping it needs is decided by what it holds, which is the answer.
- // The same test the gate above makes: with both renderers set, a
- // caller asking for `srcdoc` on the normal path gets it there.
- if (_deferEmbeddedSource !== undefined && _deferSrcdoc) {
- // Declining leaves the attribute exactly as it was written, down to
- // the delimiter: re-escaping it into double quotes would lengthen a
- // value the source spelled with single ones.
- const delimiter = open[valueStart - tagStart - 1];
- const asWritten = ` ${attributeName}=${delimiter}${rawValue}${delimiter}`;
- out += _deferEmbedded(document, HTML_TYPE, (rendered) =>
- typeof rendered !== "string" || rendered === document
- ? asWritten
- : ` ${attributeName}="${escapeAttribute(rendered, 34, true)}"`
- );
- unquotedTail = false;
- consumedEnd = valueEnd;
- continue;
- }
- // The synchronous renderer: `_renderEmbedded` defers where a collector
- // is set, and this attribute is the one kept off that path.
- const rendered = _renderEmbeddedOrNull(document, HTML_TYPE);
- if (rendered !== null && rendered !== document) {
- out += ` ${attributeName}="${escapeAttribute(rendered, 34, true)}"`;
- unquotedTail = false;
- consumedEnd = valueEnd;
- continue;
- }
- }
- // A character reference would decode to something these grammars read
- // differently, so only a reference-free value is rewritten.
- let value = rawValue;
- if (!hasReference) {
- if (html) {
- value = _rewriteAttributeValue(
- elementName,
- attributeName,
- rawValue,
- isViewportMeta,
- enumeratedOn
- );
- } else if (attributeName === "style") {
- // The one foreign-content attribute HTML's rules still fit: SVG and
- // MathML `style` is a CSS declaration list, same as HTML's. Every
- // other name is adjusted or means something else there.
- value = _minifyStyleAttribute(rawValue);
- }
- }
- // An empty value is the bare name: the tokenizer's after-attribute-name
- // state reads `x` in `<p x>` and `<p x="">` as the same attribute with
- // the same empty value, in foreign content too. A boolean attribute
- // spelled with its own name (`disabled="disabled"`) is the same empty
- // presence, and that one the spec does canonicalize.
- // Both spellings are the same empty presence, and writing either as the
- // bare name is the `=""` going — a quoting decision, so it answers to
- // the switch that names the rest of them.
- if (
- _transforms.normalizeAttributeQuotes &&
- (rawValue === "" ||
- (html &&
- _transforms.collapseBooleanAttributes &&
- rawValue.length === attributeName.length &&
- _isBooleanAttribute(elementName, attributeName) &&
- _asciiLowerCase(rawValue) === attributeName))
- ) {
- out += ` ${attributeName}`;
- unquotedTail = false;
- consumedEnd = valueEnd;
- continue;
- }
- // Guarded on the reference, so the decode stays off the hot path — and
- // tried before unquoting, which keeps every reference it writes bare.
- const decoded = hasReference ? decodeEntities(rawValue, true) : null;
- // A reference spelling whitespace is deliberate: writing it out would
- // let `removeEmptyAttributes` read the attribute as empty on a re-run.
- const spellable =
- decoded !== null &&
- (_asciiTrim(decoded) !== "" || _asciiTrim(rawValue) === "")
- ? decoded
- : null;
- const isStyle = attributeName === "style";
- // A `style=""` carrying a character reference is a declaration list like
- // any other once decoded — `"` is a quote to CSS, not an entity — so
- // it is minified from the decoded text and spelled back from there.
- const declarations =
- isStyle && spellable !== null ? _minifyStyleAttribute(spellable) : null;
- /**
- * @param {string} minified the declaration list to write
- * @returns {{ text: string, unquotedTail: boolean } | null} what to emit
- */
- const writeStyle = (minified) =>
- declarations !== null
- ? _writeAttribute(attributeName, rawValue, rawValue, minified)
- : _writeAttribute(attributeName, minified, rawValue, spellable);
- const written = isStyle
- ? writeStyle(declarations !== null ? declarations : value)
- : _writeAttribute(attributeName, value, rawValue, spellable);
- if (written !== null) {
- // The list is deferred, not the shape around it: whether this value
- // leaves the tag ending unquoted is read back when the tag closes and
- // cannot wait for an answer, so one that would land in the other shape
- // is declined and the built-in's text stands.
- out =
- isStyle && _deferEmbeddedSource !== undefined
- ? out +
- _deferEmbedded(
- // Decoded when the source spelled it with references, since
- // that is the list itself rather than one attribute's spelling.
- spellable !== null ? spellable : rawValue,
- CSS_TYPE,
- (rendered) => {
- if (typeof rendered !== "string") return written.text;
- const answered = writeStyle(rendered);
- return answered !== null &&
- answered.unquotedTail === written.unquotedTail
- ? answered.text
- : written.text;
- },
- BLOCK_CONTENTS
- )
- : out + written.text;
- unquotedTail = written.unquotedTail;
- consumedEnd = valueEnd;
- continue;
- }
- }
- // A quoted value carries its quotes over; an unquoted one starts at the
- // value itself.
- out += ` ${attributeName}=${open.slice(
- (quoted ? valueStart - 1 : valueStart) - tagStart,
- valueEnd - tagStart
- )}`;
- unquotedTail = !quoted;
- consumedEnd = valueEnd;
- }
- // A source `/>` self-closes a foreign element, so it survives there — but only
- // when the `/` really is the flag: in `<a href=x/>` it is the last character of
- // an unquoted value instead, and after one of those it needs a space or it
- // would fuse into the value. An HTML element's parser ignores it, so it goes.
- if (
- path.namespace() !== NS_HTML &&
- open.charCodeAt(open.length - 2) === 47 &&
- tagEnd - 2 >= consumedEnd
- ) {
- return `${out}${unquotedTail ? " /" : "/"}>`;
- }
- return `${out}>`;
- };
- /**
- * Whether a comment must survive minification: downlevel conditional comments
- * (`<!--[if …]>` / `<![endif]-->`) drive IE branching, server-side includes
- * (`<!--#…-->`) are directives, and a `<?…?>` bogus comment is a server-side
- * template tag — all behavior-bearing. Every other comment is inert, so
- * dropping it preserves meaning.
- * @param {string} data comment data (between `<!--` and `-->`)
- * @param {string} source the comment's source text, including its delimiters
- * @returns {boolean} true to keep the comment
- */
- const _keepComment = (data, source) => {
- // `<?php … ?>` / `<%… %>` reach the tree as bogus comments (§13.2.5.42) but
- // are server-side directives, and the forms below drive IE branching or a
- // server-side include — every one of them is code rather than a comment, so
- // no level drops it.
- if (source.charCodeAt(1) === 63) return true;
- const trimmed = data.trimStart();
- // `<![` covers the downlevel-revealed forms: `<!--<![endif]-->` closers and
- // standalone `<![if …]>` openers.
- if (
- trimmed.startsWith("[if") ||
- trimmed.startsWith("[endif") ||
- trimmed.startsWith("<![") ||
- trimmed.startsWith("#")
- ) {
- return true;
- }
- const kept = _commentsKept;
- // Every other comment is inert, so `"some"` is where the two languages part:
- // nothing left here carries anything a parser reads.
- if (typeof kept === "boolean") return kept;
- if (kept === "some") return false;
- // A pattern or a predicate of the author's, over the comment's own text.
- return kept(data) === true;
- };
- // The source open tag `_elementOpenTag` last read — its second result, passed
- // out of band so the per-element pair costs no object.
- let _elementOpenSource = "";
- /**
- * An element's open tag as it stands in the source, before any minifying; also
- * left in `_elementOpenSource`.
- * @param {HtmlPath} path positioned on the element
- * @returns {string} its source open tag
- */
- const _openTagSource = (path) => {
- // A repeated `<html>` / `<body>` start tag adds its unseen attributes to the
- // open element and is itself dropped, so that element's source tag no longer
- // spells its attributes — rebuild it rather than echo a stale one.
- const source =
- _mergedAttrNodes.size !== 0 && _mergedAttrNodes.has(path.node)
- ? _synthesizeOpenTag(path)
- : path.openTag();
- _elementOpenSource = source;
- return source;
- };
- /**
- * An element's open tag as the printer would emit it; the source tag it was
- * read from is left in `_elementOpenSource`.
- * @param {HtmlPath} path positioned on the element
- * @param {boolean} minify whether minifying
- * @returns {string} the printed open tag
- */
- const _elementOpenTag = (path, minify) => {
- const source = _openTagSource(path);
- return minify && source !== "" ? _minifyOpenTag(path, source) : source;
- };
- /**
- * An element's end tag as the printer would emit it.
- * @param {HtmlPath} path positioned on the element
- * @param {string} name its tag name
- * @param {string} open its source open tag
- * @param {boolean} innerEmpty whether it printed no children
- * @param {boolean} minify whether minifying
- * @returns {string} the end tag, `""` when it carries nothing
- */
- const _elementCloseTag = (path, name, open, innerEmpty, minify) => {
- // Everything after `<plaintext>` is text, so an emitted end tag would re-parse
- // as literal text; a source self-closing tag (`<rect/>` in foreign content)
- // with no children needs none either.
- if (
- name === "plaintext" ||
- (innerEmpty && open.endsWith("/>")) ||
- (minify && _canOmitEndTag(path, name, path.node))
- ) {
- return "";
- }
- // `</name>` spelled by the source, falling back to the element's own name when
- // there is no source end tag to echo. Read here rather than up front: an
- // omitted end tag throws it away, and that is what minifying mostly does.
- if (minify && path.namespace() === NS_HTML) {
- // The source name folds to the element's own — the common case, answered
- // off the raw range without building the source spelling to cut apart.
- if (
- rangeEqualsLowerCase(_htmlSource, path.start() + 1, path.nameEnd(), name)
- ) {
- return `</${name}>`;
- }
- return `</${_asciiLowerCase(path.closeTag().slice(2, -1)).trim() || name}>`;
- }
- const sourceClose = path.closeTag();
- return sourceClose === "" ? `</${name}>` : sourceClose;
- };
- /**
- * Whether re-parsing an implied `<html>` / `<body>`'s children without its start
- * tag would not hand them back: a start tag `inBody` ignores still closed
- * `<head>` on the way there, so a head element written after one belongs to the
- * body, and §13.2.6.4.6 gives it back to the head the modes had already left. A
- * comment goes further out still, to the element above.
- * @param {HtmlNodeRef} node the implied element
- * @returns {boolean} true when its start tag has to materialize
- */
- const _contentOpensOutside = (node) => {
- const first = _nodeFirstChildren[node];
- if (first === 0) return false;
- const type = _nodeTypes[first];
- if (type === NodeType.Comment) return true;
- return (
- type === NodeType.Element &&
- _namespaceOf(first) === NS_HTML &&
- HEAD_ELEMENTS.has(_tagNameOf(first))
- );
- };
- // What `_streamTags` decided for the element just entered — passed out of band
- // so each streamed element costs no result object.
- let _streamOpen = "";
- let _streamClose = "";
- let _streamImplied = "";
- /**
- * An element printed in pieces prints as `open + children + close`, and these are
- * those two — the printer's `Element` case, minus everything in it that can only
- * be decided once the children's text is in. They are left in `_streamOpen` /
- * `_streamClose` / `_streamImplied`; `false` for an element not of that
- * shape: a leading newline to round-trip, a self-closing tag, a `<template>`'s
- * content fragment, or a source `/>` whose end tag hangs on whether anything
- * printed inside it. `_streamImplied` names an omitted `<html>` / `<body>` whose
- * open tag still materializes if the first thing inside it turns out to be
- * whitespace.
- * @param {HtmlPath} path positioned on the element
- * @param {boolean} minify whether minifying
- * @returns {boolean} whether the element prints in pieces
- */
- const _streamTags = (path, minify) => {
- _streamImplied = "";
- // Folded into the `<style>` before it, which already printed its text, or
- // carrying nothing at all.
- if (
- minify &&
- ((_absorbedStyles !== null && _absorbedStyles.has(path.node)) ||
- _isRemovableEmptyElement(path, path.node))
- ) {
- _streamOpen = "";
- _streamClose = "";
- return true;
- }
- if (path.selfClosing() || path.templateContent() !== 0) return false;
- // A fostered run that moves prints inside the table it came out of, which needs
- // both of them held rather than emitted as the walk reaches them. One printing
- // from source holds nothing: its content is a slice, so it still streams.
- if (
- _fosterHolders !== null &&
- _fosterHolders.has(path.node) &&
- ((_nodeFlags[path.node] & FLAG_FOSTER_REGION) === 0 ||
- !(/** @type {Set<HtmlNodeRef>} */ (_fosterVerbatim).has(path.node)))
- ) {
- return false;
- }
- const name = path.tagName();
- if (LEADING_NEWLINE_ELEMENTS.has(name)) return false;
- if (name === "form" && _formNeedsPointerReset(path)) return false;
- // An SVG subtree is offered to the renderer whole, so it has to be composed
- // rather than streamed out in pieces that never form one string.
- if (
- (_renderEmbeddedSource !== undefined ||
- _deferEmbeddedSource !== undefined) &&
- name === "svg" &&
- _isSvgRoot(path)
- ) {
- return false;
- }
- // Read before minifying: bailing afterwards would throw the work away and
- // leave `printer` to redo it, minifying every `style=""` on a `/>`-spelled
- // tag twice — and running a renderer twice over it.
- const source = _openTagSource(path);
- if (source.endsWith("/>")) return false;
- const tag = minify && source !== "" ? _minifyOpenTag(path, source) : source;
- if (minify && _canOmitStartTag(path, name, path.node, source)) {
- _streamOpen = "";
- _streamClose = _canOmitEndTag(path, name, path.node) ? "" : `</${name}>`;
- return true;
- }
- if (source === "") {
- if (path.attributeCount() === 0 && TRANSPARENT_IMPLIED_ELEMENTS.has(name)) {
- // The parser re-implies the start tag, but not the end tag the source
- // carried: without it whatever follows moves inside this element.
- _streamClose = (
- SHELL_ELEMENTS.has(name)
- ? _shellEndTagOmittable(path, name, path.node)
- : _canOmitEndTag(path, name, path.node)
- )
- ? ""
- : `</${name}>`;
- // Leading whitespace is decided on the first piece emitted; what the modes
- // hand back out is known here, from the node itself.
- const shell = name === "body" || name === "html";
- const opensOutside = shell && _contentOpensOutside(path.node);
- _streamOpen =
- opensOutside || _impliedTableGroupStartNeeded(path, name, path.node)
- ? `<${name}>`
- : "";
- _streamImplied = shell && !opensOutside ? name : "";
- } else {
- _streamOpen = _synthesizeOpenTag(path);
- _streamClose = `</${name}>`;
- }
- return true;
- }
- _streamOpen = tag;
- _streamClose = _elementCloseTag(path, name, source, false, minify);
- return true;
- };
- /**
- * Whether a `<form>` needs a `</form>` in front of it: one written inside another
- * reaches the tree only because an end tag cleared the form element pointer on
- * the way — §13.2.6.4.7 ignores the start tag while the pointer is set, so
- * without it the element is dropped on re-parsing. A `<template>` between the two
- * suspends the pointer instead, and needs nothing.
- * @param {HtmlPath} path the accessor positioned on the `<form>`
- * @returns {boolean} true when the end tag has to be written first
- */
- const _formNeedsPointerReset = (path) => {
- // §13.2.6.5 inserts a foreign `<form>` without consulting the pointer at all.
- if (path.namespace() !== NS_HTML) return false;
- let found = false;
- for (
- let ancestor = path.parentOf();
- ancestor !== 0;
- ancestor = path.parentOf(ancestor)
- ) {
- // A template's content is a fragment, not an element: reaching it means the
- // walk left a `<template>`, which suspends the pointer.
- if (path.type(ancestor) === NodeType.DocumentFragment) return false;
- if (path.type(ancestor) !== NodeType.Element) return found;
- if (path.namespace(ancestor) !== NS_HTML) continue;
- const name = path.tagName(ancestor);
- // A template anywhere above suspends the pointer, whichever side of the
- // outer form it is on, so the start tag is never the one that is ignored.
- if (name === "template") return false;
- if (name === "form") found = true;
- }
- return found;
- };
- /**
- * Whether printing a fostered run back inside its table hands the parser the same
- * tree: only when "in table" fosters every node of it straight back out. What it
- * keeps instead — its own content, and the start tags §13.2.6.4.9 gives a rule of
- * their own — would stay in the table, and a formatting element §13.2.4.3 rebuilt
- * has no source tag to place it with.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the run's root
- * @param {0 | 1} needed 1 once a start tag in it is known to end something open
- * @returns {0 | 1 | 2 | 3} 0 = stays put, 1 = replays, 2 = replays and is needed,
- * 3 = has to move but cannot replay, so the table prints from its own source
- */
- const _fosterRunReplays = (path, node, needed) => {
- let blocked = false;
- for (let n = node; ;) {
- if (path.type(n) === NodeType.Element) {
- const html = path.namespace(n) === NS_HTML;
- const name = html ? path.tagName(n) : "";
- // Only the root has to be one "in table" fosters: everything under it was
- // inserted at the same current node either way.
- if (n === node && html && TABLE_CONTEXT.has(name)) blocked = true;
- if (
- html &&
- ((FORMATTING.has(name) && path.openTag(n) === "") ||
- (n === node &&
- (name === "input"
- ? _isHiddenInput(path, n)
- : name === "form"
- ? path.firstChild(n) !== 0
- : FOSTER_KEPT.has(name))))
- ) {
- blocked = true;
- }
- if (!html) {
- const child = path.firstChild(n);
- if (child !== 0) {
- n = child;
- continue;
- }
- }
- // A scope rule ends the nearest match on the stack, not the parent, so
- // every element still open above it is a candidate.
- let a = html ? path.parentOf(n) : 0;
- while (needed === 0 && a !== 0) {
- if (
- path.type(a) === NodeType.Element &&
- path.namespace(a) === NS_HTML &&
- _startTagEnds(path.tagName(a), name)
- ) {
- needed = 1;
- }
- a = path.parentOf(a);
- }
- }
- const child = path.firstChild(n);
- if (child !== 0) {
- n = child;
- continue;
- }
- for (;;) {
- if (n === node) {
- return needed === 1 ? (blocked ? 3 : 2) : blocked ? 0 : 1;
- }
- const next = path.nextSibling(n);
- if (next !== 0) {
- n = next;
- break;
- }
- n = path.parentOf(n);
- }
- }
- };
- /**
- * Whether an `<input>` is the one §13.2.6.4.9 keeps rather than fosters: every
- * other one takes the "anything else" arc out of the table.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the `<input>`
- * @returns {boolean} true when the table would keep it
- */
- /**
- * Whether a `</form>` written in front of the form about to be inserted would
- * take something with it: §13.2.6.4.7 generates implied end tags before removing
- * the form, and the source's own end tag ran with nothing of the sort open.
- * @returns {boolean} true when one is open between here and the enclosing form
- */
- const _formResetWouldPop = () => {
- // §13.2.6.4.7 removes the form from the stack rather than popping to it, so the
- // enclosing one is in the tree and not on the stack — the walk is up the tree.
- let popped = false;
- for (
- let n = open.length === 0 ? 0 : open[open.length - 1];
- n !== 0;
- n = _nodeParents[n]
- ) {
- if (_nodeTypes[n] !== NodeType.Element || _namespaceOf(n) !== NS_HTML) {
- continue;
- }
- const name = _tagNameOf(n);
- if (name === "form") return popped;
- if (IMPLIED.has(name)) popped = true;
- }
- return false;
- };
- /**
- * Whether an entry of this name sits below the marker a §13.2.4.3 scan stopped
- * at. The marker is what hides it, and the marker is there because an element
- * the source left open put it there — so the tree is a shape only that element
- * staying open reproduces.
- * @param {number} from index the scan stopped at
- * @param {string} name the tag name it was looking for
- * @returns {boolean} true when the marker is hiding one
- */
- const _afeNameBelowMarker = (from, name) => {
- for (let i = from - 1; i >= 0; i--) {
- const e = afe[i];
- if (e === AFE_MARKER) continue;
- if (_namespaceOf(e) === NS_HTML && _tagNameOf(e) === name) return true;
- }
- return false;
- };
- /**
- * Whether an `<input>` is the one §13.2.6.4.9 keeps rather than fosters: every
- * other one takes the "anything else" arc out of the table.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the `<input>`
- * @returns {boolean} true when the table would keep it
- */
- const _isHiddenInput = (path, node) => {
- for (let i = 0, n = path.attributeCount(node); i < n; i++) {
- const a = path.attributeAt(i, node);
- if (path.attributeName(a) === "type") {
- return _asciiLowerCase(path.attributeValue(a)) === "hidden";
- }
- }
- return false;
- };
- /**
- * Whether a start tag ends an open element of `openName` through a scope check —
- * §4.13's followers, a second tag of the same name, one heading behind another,
- * and the set whose members end each other by generating implied end tags. Inside
- * a table none of them reach it, every scope stopping at the table; printed back
- * with no table between them they do.
- * @param {string} openName the open element's lowercased tag name
- * @param {string} name the start tag's lowercased name
- * @returns {boolean} true when the start tag ends it
- */
- const _startTagEnds = (openName, name) => {
- if (name === openName) return true;
- const ended = SCOPE_ENDED_BY.get(openName);
- if (ended !== undefined && ended !== null && ended.has(name)) return true;
- const closers = OPTIONAL_END_TAG_FOLLOWERS.get(openName);
- if (closers !== undefined && closers.has(name)) return true;
- // §4.13's tag-omission list is not §13.2.6.4.7's close-a-p list — `center` is in
- // one and not the other, and the parser's is what decides the tree.
- if (openName === "p" && BLOCK_START.has(name)) return true;
- if (HEADING.has(openName)) return HEADING.has(name);
- return IMPLIED.has(openName) && IMPLIED.has(name);
- };
- /**
- * Whether `table`'s fostered run prints back inside it, memoized so the holder
- * that skips the run and the table that takes it in answer this the same way.
- * Moving it is a repair, not a preference: tree order is left alone unless a
- * start tag in the run would close the holder on the way back in, and the run
- * replays out of the table it is moved into.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} table the table the run came out of
- * @returns {boolean} true when the run moves back in
- */
- const _fosterRunMoves = (path, table) => {
- const moves = /** @type {Map<HtmlNodeRef, boolean>} */ (_fosterMoves);
- const cached = moves.get(table);
- if (cached !== undefined) return cached;
- moves.set(table, false);
- const run = /** @type {Map<HtmlNodeRef, HtmlNodeRef[]>} */ (
- _fosteredRuns
- ).get(table);
- if (run === undefined) return false;
- if (path.parentOf(table) === 0) return false;
- let closes = false;
- let text = false;
- let blocked = false;
- for (let i = 0; i < run.length; i++) {
- const verdict = _fosterRunReplays(path, run[i], 0);
- if (verdict === 0 || verdict === 3) {
- blocked = true;
- break;
- }
- if (verdict === 2) closes = true;
- if (path.type(run[i]) === NodeType.Text) text = true;
- }
- // Interleaving reads both in source order, which holds only while they do not
- // overlap: one written inside a sibling's span has no place among them.
- if (!blocked) {
- for (let i = 0; i < run.length && !blocked; i++) {
- const at = path.start(run[i]);
- for (let c = path.firstChild(table); c !== 0; c = path.nextSibling(c)) {
- if (at > path.start(c) && at < _fosterRegionEnd(path, c)) {
- blocked = true;
- break;
- }
- }
- }
- }
- // A table keeps whitespace-only text and fosters the rest: moved back in beside
- // each other the two fuse, and one non-whitespace character fosters all of it.
- // That blocks the re-nesting like any other node that cannot replay.
- if (text && !blocked) {
- for (let c = path.firstChild(table); c !== 0; c = path.nextSibling(c)) {
- if (path.type(c) === NodeType.Text) {
- blocked = true;
- break;
- }
- }
- }
- // The run moves or stays whole, so one node needing to move and another unable
- // to replay is the verbatim case — worth a second look only once blocked.
- if (blocked) {
- closes = false;
- for (let i = 0; i < run.length && !closes; i++) {
- const verdict = _fosterRunReplays(path, run[i], 0);
- if (verdict === 2 || verdict === 3) closes = true;
- }
- }
- if (!closes) return false;
- // Nothing re-nested reproduces this one, so it prints the source it was built
- // from and the run prints nothing: it is already inside that.
- if (blocked) {
- if (_fosterVerbatim === null) _fosterVerbatim = new Set();
- // The holder, not the table: an earlier run in it prints end tags that leave
- // the list of active formatting elements elsewhere before this one is read.
- const holder = path.parentOf(table);
- _fosterVerbatim.add(holder);
- // Anything still open around it: the slice runs to where the source stopped,
- // so an end tag none of them was given cannot be invented after it.
- if (_fosterOpenTail === null) _fosterOpenTail = new Set();
- for (let a = holder; a !== 0; a = path.parentOf(a)) {
- _fosterOpenTail.add(a);
- _nodeFlags[a] |= FLAG_FOSTER_REGION;
- }
- const reach = _fosterRegionEnd(path, table);
- if (reach > _fosterSpanEnd) _fosterSpanEnd = reach;
- moves.set(table, true);
- return true;
- }
- moves.set(table, closes);
- return closes;
- };
- /**
- * Decide every fostered run before the walk starts and forget the ones that stay
- * put: a holder whose run moves cannot print in pieces, and that costs its whole
- * subtree the streamed path — a price only the runs that actually move should
- * pay. With none left the walk runs exactly as it does for a document that never
- * fostered anything.
- * @param {HtmlPath} path the accessor
- * @returns {void}
- */
- const _settleFosteredRuns = (path) => {
- const log = /** @type {HtmlNodeRef[]} */ (_fosteredLog);
- // Nothing is built for a document whose runs all stay put, which is most of
- // them: one pass says whether any node would be moved at all.
- let moves = false;
- for (let i = 1; i < log.length; i += 2) {
- const verdict = _fosterRunReplays(path, log[i], 0);
- if (verdict === 2 || verdict === 3) {
- moves = true;
- break;
- }
- }
- if (!moves) return;
- /** @type {Map<HtmlNodeRef, HtmlNodeRef[]>} */
- const runs = new Map();
- for (let i = 0; i < log.length; i += 2) {
- const run = runs.get(log[i]);
- if (run === undefined) runs.set(log[i], [log[i + 1]]);
- else run.push(log[i + 1]);
- }
- _fosteredRuns = runs;
- _fosterMoves = new Map();
- /** @type {Map<HtmlNodeRef, HtmlNodeRef>} */
- const nodes = new Map();
- /** @type {Set<HtmlNodeRef>} */
- const holders = new Set();
- for (let i = 0; i < log.length; i += 2) {
- if (!_fosterRunMoves(path, log[i])) continue;
- // The adoption agency can move a node after it is fostered, so the holder
- // that skips the run is read from the tree, not from where it went in.
- holders.add(path.parentOf(log[i + 1]));
- nodes.set(log[i + 1], log[i]);
- }
- for (const table of runs.keys()) {
- if (!_fosterRunMoves(path, table)) runs.delete(table);
- }
- _fosteredNodes = nodes;
- _fosterHolders = holders;
- };
- /**
- * The tree as a string, for the one check the printer cannot make from the tree
- * alone: whether what it printed parses back to what it printed from. Values are
- * compared as the DOM holds them, so the same attribute written with a different
- * quote or escape is the same attribute.
- * @param {HtmlNodeRef} root the node whose children are read
- * @returns {string} a rendering equal for equal trees
- */
- const _digest = (root) => {
- /** @type {string[]} */
- const out = [];
- // A stack rather than recursion: a document nests as deeply as its source says
- // it does, and the tree this reads is the one the printer just walked.
- /** @type {number[]} */
- const stack = [];
- const first = A.firstChild(root);
- if (first !== 0) stack.push(first, 0);
- while (stack.length !== 0) {
- const depth = /** @type {number} */ (stack.pop());
- const node = /** @type {HtmlNodeRef} */ (stack.pop());
- // The sibling goes on first, so this node's own subtree drains before it.
- const next = A.nextSibling(node);
- if (next !== 0) stack.push(next, depth);
- const type = A.type(node);
- if (type !== NodeType.Element) {
- out.push(`${depth}\u0001${type}\u0001${A.data(node)}`);
- continue;
- }
- out.push(
- `${depth}\u0001e\u0001${A.namespace(node)}\u0001${A.tagName(node)}`
- );
- const attributes = [];
- for (const a of A.attributes(node)) {
- // The value as the DOM holds it: the same value can be written with a
- // different quote and a different escape and still be the same attribute.
- attributes.push(
- `${a.serializedName || a.name}=${decodeEntities(a.value, true)}`
- );
- }
- attributes.sort();
- out.push(attributes.join("\u0002"));
- const content = A.templateContent(node);
- const inner = content !== 0 ? A.firstChild(content) : A.firstChild(node);
- if (inner !== 0) stack.push(inner, depth + 1);
- }
- return out.join("\n");
- };
- /**
- * Whether this element is one the settled runs concern — the parent a run was
- * fostered into, or the table it goes back inside. Every other element composes
- * its children the way it always did.
- * @param {HtmlNodeRef} node the element
- * @returns {boolean} true when it has to read the run
- */
- const _fosterTouches = (node) =>
- (_fosterVerbatim !== null && _fosterVerbatim.has(node)) ||
- /** @type {Set<HtmlNodeRef>} */ (_fosterHolders).has(node) ||
- /** @type {Map<HtmlNodeRef, HtmlNodeRef[]>} */ (_fosteredRuns).has(node);
- /**
- * Whether this element still owes an end tag once its content printed from
- * source: one the source never wrote is not the printer's to invent, and one it
- * did write is already inside the slice when the span ran past it.
- * @param {HtmlPath} path the accessor positioned on the element
- * @returns {boolean} true when the end tag still has to be printed
- */
- const _endTagOutsideSpan = (path) =>
- path.sourceClosed() && path.end(path.node) > _fosterSpanEnd;
- /**
- * The source an element's content was built from. Nothing re-nested reproduces a
- * run that lands here, and the source did — anchored on the element itself, so no
- * child of it can start earlier and none is left out.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} node the element printing from source
- * @returns {string} its raw content
- */
- const _fosterSourceRegion = (path, node) => {
- let end = path.tagEnd(node);
- // A fragment, or an element the parser implied, has no opening tag to start
- // from — its content begins wherever the first thing in it was written.
- let from = -1;
- for (let c = path.firstChild(node); c !== 0; c = path.nextSibling(c)) {
- const reach = _fosterRegionEnd(path, c);
- if (reach > end) end = reach;
- // Offset 0 is a real position for the first node in the source; a node the
- // parser made has no range at all, which is both offsets at zero.
- const at = path.start(c);
- if ((at !== 0 || path.end(c) !== 0) && (from === -1 || at < from)) {
- from = at;
- }
- }
- if (from === -1 || from > end) from = path.tagEnd(node);
- return path.sourceSpanAt(from, end);
- };
- /**
- * An element's children's text when this document foster parented something: a
- * run that moves prints inside the table it was written in rather than where the
- * tree holds it, interleaved with the table's own children in source order —
- * which is the order they were written in, so the parser fosters them right back
- * out to where they are now.
- * @param {HtmlPath} path the accessor positioned on the element
- * @param {PrintContext} writer the print context
- * @returns {string} the children's text
- */
- const _composeAroundFoster = (path, writer) => {
- const node = path.node;
- const fostered = /** @type {Map<HtmlNodeRef, HtmlNodeRef>} */ (
- _fosteredNodes
- );
- const run = _fosterRunMoves(path, node)
- ? /** @type {Map<HtmlNodeRef, HtmlNodeRef[]>} */ (_fosteredRuns).get(node)
- : undefined;
- /** @type {HtmlNodeRef[]} */
- const order = run === undefined ? [] : [...run];
- const moved = order.length !== 0;
- for (let c = path.firstChild(); c !== 0; c = path.nextSibling(c)) {
- const from = fostered.get(c);
- if (from !== undefined && _fosterRunMoves(path, from)) continue;
- order.push(c);
- }
- if (moved) order.sort((a, b) => path.start(a) - path.start(b));
- if (
- (_nodeFlags[node] & FLAG_FOSTER_REGION) !== 0 &&
- /** @type {Set<HtmlNodeRef>} */ (_fosterVerbatim).has(node)
- ) {
- return _fosterSourceRegion(path, node);
- }
- let inner = "";
- for (let i = 0; i < order.length; i++) inner += writer.get(order[i]);
- return inner;
- };
- /**
- * The last source offset the table's own tokens reach. Its recorded end stops at
- * the start tag when the source never closed it, and the fostered nodes were
- * written inside it, so the whole subtree and the run answer together.
- * @param {HtmlPath} path the accessor
- * @param {HtmlNodeRef} table the table printing from source
- * @returns {number} end offset
- */
- const _fosterRegionEnd = (path, table) => {
- let end = path.end(table);
- const run = /** @type {Map<HtmlNodeRef, HtmlNodeRef[]>} */ (
- _fosteredRuns
- ).get(table);
- const roots = run === undefined ? [table] : [table, ...run];
- if (path.type(table) !== NodeType.Element) return end;
- for (const root of roots) {
- for (let n = root; ;) {
- if (path.end(n) > end) end = path.end(n);
- const child = path.firstChild(n);
- if (child !== 0) {
- n = child;
- continue;
- }
- for (;;) {
- if (n === root) break;
- const next = path.nextSibling(n);
- if (next !== 0) {
- n = next;
- break;
- }
- n = path.parentOf(n);
- }
- if (n === root) break;
- }
- }
- return end;
- };
- /**
- * The default HTML node printer — passed to the `SourceProcessor` and fired per
- * node once its children are printed (a developer could supply their own). It
- * takes the same `path` a visitor gets plus the print context as its `writer`,
- * and knows nothing of the walk: it switches on `path.type()`, serializes the node
- * — opening tags kept verbatim from source (attribute quoting / spacing
- * preserved), end tags generated, text re-escaped from its decoded value — pulling
- * its children's already-printed text from `writer.get`, and **returns** it. This
- * is the WHATWG serialization, so the output re-parses to the same DOM (text-node
- * offsets can overrun end tags, so source slices can't be used for text). The
- * minify transforms are dropping inert comments and collapsing the whitespace
- * between an opening tag's attributes; DOM-absent whitespace (between the
- * doctype and `<html>`, etc.) naturally falls away.
- * @param {HtmlPath} path the accessor positioned on the finished node
- * @param {PrintContext} writer the print context (children's printed text)
- * @returns {string} the node's serialized text
- * @experimental exposed as `webpack.html.syntax.printer`; unstable API
- */
- const printer = (path, writer) => {
- const minify = writer.options.mode === "minify";
- switch (path.type()) {
- case NodeType.Element: {
- // Folded into the `<style>` before it, which printed the whole run, or
- // carrying nothing at all.
- if (
- minify &&
- ((_absorbedStyles !== null && _absorbedStyles.has(path.node)) ||
- _isRemovableEmptyElement(path, path.node))
- ) {
- return "";
- }
- // Minifying rewrites the tag (attribute spacing / quoting) but never its
- // `/>`-ness, which the end-tag decision below reads off the source.
- const tag = _elementOpenTag(path, minify);
- const open = _elementOpenSource;
- // The whole subtree goes to the renderer, its nested bodies already
- // rendered — children print before their parent.
- const svgRoot =
- minify &&
- (_renderEmbeddedSource !== undefined ||
- _deferEmbeddedSource !== undefined) &&
- _isSvgRoot(path);
- // Void / self-closing elements have no children and no end tag; a
- // tag-less one (`<image>` → img, `</br>` → br) is rebuilt, not dropped.
- if (path.selfClosing()) {
- const selfClosed = open !== "" ? tag : _synthesizeOpenTag(path);
- return svgRoot ? _renderEmbedded(selfClosed, SVG_TYPE) : selfClosed;
- }
- // `<template>` holds its children in a content fragment, not the child
- // chain; every other element reconstructs from its children in order.
- const tc = path.templateContent();
- let inner = "";
- if (tc !== 0) {
- inner = writer.get(tc);
- } else {
- if (_fosteredRuns !== null && _fosterTouches(path.node)) {
- inner = _composeAroundFoster(path, writer);
- } else {
- for (let c = path.firstChild(); c !== 0; c = path.nextSibling(c)) {
- inner += writer.get(c);
- }
- }
- // Round-trip the parser's leading-newline strip (`<pre>` / `<textarea>`
- // / `<listing>`) so a value that starts with one survives re-parsing.
- if (
- inner.charCodeAt(0) === 10 &&
- path.namespace() === NS_HTML &&
- LEADING_NEWLINE_ELEMENTS.has(path.tagName())
- ) {
- inner = `\n${inner}`;
- }
- }
- const name = path.tagName();
- // The end tag is handled by the optional-end-tag rules like any other.
- if (minify && _canOmitStartTag(path, name, path.node, open)) {
- const implied = _canOmitEndTag(path, name, path.node)
- ? ""
- : `</${name}>`;
- return inner + implied;
- }
- if (open === "") {
- // A parser-implied structural element stays transparent — the parser
- // re-implies it — but only when attribute-less (repeated `<html>` /
- // `<body>` tags merge attributes onto the implied element).
- if (
- path.attributeCount() === 0 &&
- TRANSPARENT_IMPLIED_ELEMENTS.has(name)
- ) {
- // Leading whitespace only survives re-parsing once `<body>` has
- // started: before that the insertion modes drop it. Materializing
- // the tag is the cheapest way to keep the text node intact.
- if (
- (name === "body" || name === "html") &&
- (startsWithWs(inner) || _contentOpensOutside(path.node))
- ) {
- return `<${name}>${inner}`;
- }
- // §4.13 lets the start tag go only when the group before it kept
- // its end tag, or that one would take these columns over.
- if (_impliedTableGroupStartNeeded(path, name, path.node)) {
- return `<${name}>${inner}`;
- }
- // The parser re-implies the start tag, but not the end tag the
- // source carried: without it whatever follows moves inside.
- return (
- SHELL_ELEMENTS.has(name)
- ? _shellEndTagOmittable(path, name, path.node)
- : _canOmitEndTag(path, name, path.node)
- )
- ? inner
- : `${inner}</${name}>`;
- }
- // Anything else tag-less (adoption-agency clone, reconstructed
- // formatting element) must materialize or its formatting is lost.
- return `${_synthesizeOpenTag(path) + inner}</${name}>`;
- }
- const composed =
- (name === "form" && _formNeedsPointerReset(path) ? "</form>" : "") +
- tag +
- inner +
- // Its content came out of the source unchanged, so an end tag the
- // source never wrote would leave the parser somewhere else.
- ((_nodeFlags[path.node] & FLAG_FOSTER_REGION) !== 0 &&
- !_endTagOutsideSpan(path)
- ? ""
- : _elementCloseTag(path, name, open, inner === "", minify));
- return svgRoot ? _renderEmbedded(composed, SVG_TYPE) : composed;
- }
- case NodeType.Text: {
- const data = path.data();
- // Literal-text elements (`script` / `style` / …) keep their body raw;
- // every other text node is escaped (the WHATWG text split).
- const parent = path.parentOf();
- // TODO minify an inline `<script>` too: terser reaches webpack only
- // through `minimizer-webpack-plugin` and its API is async, so it needs
- // a JS-minify hook on the options rather than a call from here.
- if (parent !== 0 && LITERAL_TEXT_PARENTS.has(path.tagName(parent))) {
- if (!minify) return data;
- const literalName = path.tagName(parent);
- if (literalName === "style") {
- if (!_isCssStyleElement(path, parent)) return data;
- if (!_mergeStyles) {
- const minified = _minifyStyleBody(data);
- return minified === null ? data : minified;
- }
- // The run's first sheet prints the whole run; the rest printed
- // into it here, before their own text or tags are reached.
- return _absorbedStyles !== null && _absorbedStyles.has(parent)
- ? ""
- : _mergedStyleRun(path, parent, data);
- }
- if (literalName === "script") {
- const type = _scriptType(path, parent);
- if (JSON_SCRIPT_TYPES.has(type) || _JSON_SUBTYPE_REGEXP.test(type)) {
- return _minifyInlineJson(data);
- }
- // No built-in: webpack ships no JS minifier, so an inline script is
- // touched only by a caller's renderer.
- if (
- (_renderEmbeddedSource !== undefined ||
- _deferEmbeddedSource !== undefined) &&
- JAVASCRIPT_SCRIPT_TYPES.has(type) &&
- data.trim() !== ""
- ) {
- return _renderEmbedded(data, JAVASCRIPT_TYPE);
- }
- }
- return data;
- }
- // Whitespace directly under `<head>` or `<html>` is outside any block
- // formatting context, so nothing ever renders it — the same inert-node
- // reasoning that drops comments. `<body>`'s whitespace does render.
- if (
- minify &&
- parent !== 0 &&
- path.namespace(parent) === NS_HTML &&
- (path.tagName(parent) === "head" || path.tagName(parent) === "html") &&
- isAllWs(data)
- ) {
- return "";
- }
- // A `<` puts the node on the source-passthrough path below, which is what
- // keeps `<%= x %>` off the escaper — worth more than the whitespace.
- if (minify && _collapseWhitespace !== "none" && !data.includes("<")) {
- // Only elements rendering whitespace verbatim without any CSS are
- // excluded — by ancestor, since `white-space` inherits.
- let preformatted = false;
- for (
- let ancestor = parent;
- ancestor !== 0;
- ancestor = path.parentOf(ancestor)
- ) {
- if (
- path.namespace(ancestor) === NS_HTML &&
- LEADING_NEWLINE_ELEMENTS.has(path.tagName(ancestor))
- ) {
- preformatted = true;
- break;
- }
- }
- if (!preformatted) {
- let collapsed = collapseWhitespaceRuns(data);
- // Past `"conservative"`, whitespace against a boundary no line box
- // reaches is dropped rather than kept as one space.
- if (_collapseWhitespace !== "conservative") {
- const all = _collapseWhitespace === "all";
- if (all || _startsOutsideLineBox(path, path.node, parent)) {
- collapsed = collapsed.replace(/^ /, "");
- }
- if (all || _endsOutsideLineBox(path, path.node, parent)) {
- collapsed = collapsed.replace(/ $/, "");
- }
- }
- if (collapsed !== data) return _escapeTextContent(collapsed, minify);
- }
- }
- const raw = path.source();
- // Off-spec workaround: §13.3 escapes `<`/`>` in text, which would rewrite
- // `<%= x %>` to `<%= x %>` in HTML webpack only passes through. Text
- // that decoded to itself re-tokenizes to itself, so emit its source bytes.
- // The guards are what keep that equivalence: every character reference is
- // at least two characters longer than what it decodes to, so equal lengths
- // mean nothing decoded — an unequal one means the node was merged or
- // foster-parented and its range no longer describes `data`. The tail is
- // checked separately because it is the only place concatenation can change
- // tokenization: a `<`, or a `&` still open on a reference, would fuse with
- // the next sibling once a comment between them is dropped. A CR is the one
- // rewrite the length test cannot see — preprocessing maps a lone one to LF
- // without changing length — so it is excluded by name.
- if (
- raw.length === data.length &&
- raw.charCodeAt(raw.length - 1) !== 60 &&
- !(inputHasCr && raw.includes("\r")) &&
- !_hasOpenReference(raw)
- ) {
- return raw;
- }
- return _escapeTextContent(data, minify);
- }
- case NodeType.Doctype: {
- // A leaf with no children, so its source is already tight. Parsing
- // ASCII-lowercases the keyword and the name; the identifiers behind them
- // are case-sensitive strings and stay as written.
- const source = path.source();
- if (!minify) return source;
- const head = /^<!doctype[\t\n\f\r ]+([^\t\n\f\r >]+)/i.exec(source);
- return head === null
- ? source
- : `<!doctype ${_asciiLowerCase(head[1])}${source.slice(
- head[0].length
- )}`;
- }
- case NodeType.Comment:
- return minify && !_keepComment(path.data(), path.source())
- ? ""
- : path.source();
- case NodeType.ProcessingInstruction:
- // A leaf whose source is already tight, and unlike a comment it carries
- // instructions for a consumer, so minification never drops it.
- return path.source();
- default: {
- // Document / DocumentFragment: concatenate children in order.
- if (_fosteredRuns !== null && _fosterTouches(path.node)) {
- return _composeAroundFoster(path, writer);
- }
- let out = "";
- for (let c = path.firstChild(); c !== 0; c = path.nextSibling(c)) {
- out += writer.get(c);
- }
- return out;
- }
- }
- };
- // === Walk ===
- // Walk state, hoisted to module scope so the walk helpers keep one function
- // identity across parses and the streaming path can drive them mid-parse.
- /** @type {CompiledVisitorMap} the between-parses value, so no map is retained */
- const EMPTY_VISITORS = [];
- /** @type {CompiledVisitorMap} */
- let _visitors = EMPTY_VISITORS;
- /** @type {PrintContext | undefined} */
- let _writer;
- // Iterative depth-first walk over the link columns: `firstChild` /
- // `nextSibling` descend, the `parent` column ascends, so arbitrarily deep
- // markup can't overflow the call stack (the old recursive walk died at
- // ~10⁵ nesting). A `<template>`'s content fragment is visited before the
- // element's children; ascending out of it continues with those children
- // (the fragment is never in a sibling chain, `_nodeNextSiblings` = 0).
- /**
- * Fire a node's `enter` visitors; true when the walk may descend.
- * @param {HtmlNodeRef} node node
- * @returns {boolean} false when a visitor called `skipChildren()`
- */
- const _enterNode = (node) => {
- // A holder printing from source composes out of the tree, not out of its
- // children's text, so printing them would only be thrown away.
- if (
- (_nodeFlags[node] & FLAG_FOSTER_REGION) !== 0 &&
- /** @type {Set<HtmlNodeRef>} */ (_fosterVerbatim).has(node)
- ) {
- return false;
- }
- const b = _visitors[_nodeTypes[node]];
- if (b === undefined || b.enter.length === 0) return true;
- _walkSkip = false;
- _currentNode = node;
- const p = _nodeParents[node];
- _currentParent = p === 0 ? null : p;
- const e = b.enter;
- for (let i = 0; i < e.length; i++) e[i](A);
- const skip = _walkSkip;
- _walkSkip = false;
- return !skip;
- };
- // === Printing in pieces ===
- // The printer composes a node from its children's text, so the store holds every
- // node's until the root is taken — several times the output on a document of any
- // size. A node whose text is exactly `open + children + close` instead emits its
- // opener as it opens and its closer as it exits, and each child goes straight out
- // as it finishes; the store then only ever holds the subtrees that cannot (see
- // `_streamTags`). Indexed by print depth, which is the open path.
- /** @type {number} depth of the node the printing walk is on (maintained by `_walkRun`) */
- let _printDepth = 0;
- /** @type {number[]} 1 while the node at that depth prints in pieces */
- const _printOpen = [];
- /** @type {string[]} the closer the node at that depth still owes */
- const _printClose = [];
- // An omitted `<html>` / `<body>` only materializes when the first thing inside it
- // is whitespace, which is known once the first piece is emitted — so its opener
- // waits in the writer's pending stack until then.
- /** @type {string[]} names of the elements whose openers are held back, innermost last */
- const _printPendingNames = [];
- /** @type {number[]} pending-stack slot of the node at that depth, -1 for none */
- const _printPendingSlot = [];
- /**
- * Emit one piece of a node printed in pieces. The omitted openers held back above
- * it decide first: an omitted `<html>` / `<body>` materializes exactly when the
- * first thing inside it is whitespace, which the parser would otherwise drop on
- * the way back in. Only the innermost can — once it does, everything outside it
- * starts with a `<`.
- * @param {string} text output text
- */
- const _emitPiece = (text) => {
- if (text === "") return;
- const w = /** @type {PrintContext} */ (_writer);
- const held = _printPendingNames.length;
- if (held !== 0) {
- const c = text.charCodeAt(0);
- if (c === 0x09 || c === 0x0a || c === 0x0c || c === 0x0d || c === 0x20) {
- w.setPending(held - 1, `<${_printPendingNames[held - 1]}>`);
- }
- _printPendingNames.length = 0;
- w.flushPending();
- }
- w.emitStreamed(text);
- };
- /**
- * Decide whether the node just entered prints in pieces, and emit its opener if
- * so. A node can only do that inside something that already is, so the decision
- * walks down from the root and stops at the first that cannot.
- * @param {HtmlNodeRef} node node
- * @param {number} depth its depth in the walk
- * @param {boolean} descending whether the walk may enter its children
- */
- const _openStreamed = (node, depth, descending) => {
- const w = /** @type {PrintContext} */ (_writer);
- let streams = 0;
- _printPendingSlot[depth] = -1;
- if (descending && (depth === 0 || _printOpen[depth - 1] === 1)) {
- const ty = _nodeTypes[node];
- if (ty === NodeType.Document || ty === NodeType.DocumentFragment) {
- // Nothing of its own to print: its text is its children's, in order.
- streams = 1;
- _printClose[depth] = "";
- } else if (ty === NodeType.Element) {
- _currentNode = node;
- if (_streamTags(A, w.options.mode === "minify")) {
- streams = 1;
- _printClose[depth] =
- (_nodeFlags[node] & FLAG_FOSTER_REGION) !== 0 &&
- !_endTagOutsideSpan(A)
- ? ""
- : _streamClose;
- if (_streamImplied === "") {
- _emitPiece(_streamOpen);
- } else {
- _printPendingSlot[depth] = w.pushPending("");
- _printPendingNames.push(_streamImplied);
- }
- }
- }
- } else if (
- (_nodeFlags[node] & FLAG_FOSTER_REGION) !== 0 &&
- (depth === 0 || _printOpen[depth - 1] === 1) &&
- /** @type {Set<HtmlNodeRef>} */ (_fosterVerbatim).has(node)
- ) {
- // Its children are skipped and its text is `open + that source + close`, so
- // it streams like any other element rather than composing nothing.
- _currentNode = node;
- if (_streamTags(A, w.options.mode === "minify")) {
- streams = 1;
- _printClose[depth] = _endTagOutsideSpan(A) ? _streamClose : "";
- const region = _fosterSourceRegion(A, node);
- if (_streamImplied === "") {
- _emitPiece(_streamOpen + region);
- } else {
- _printPendingSlot[depth] = w.pushPending("");
- _printPendingNames.push(_streamImplied);
- // The tag is still undecided; the content behind it is not, and nothing
- // else will emit it because the children are skipped.
- _emitPiece(region);
- }
- }
- }
- _printOpen[depth] = streams;
- };
- /**
- * Fire a node's `exit` visitors, then — when printing — its printer (the
- * post-order point: its children are already printed). `_writer` undefined =
- * walk only.
- * @param {HtmlNodeRef} node node
- */
- const _exitNode = (node) => {
- const b = _visitors[_nodeTypes[node]];
- if (b === undefined && _writer === undefined) return;
- _currentNode = node;
- const p = _nodeParents[node];
- _currentParent = p === 0 ? null : p;
- if (b !== undefined) {
- const x = b.exit;
- for (let i = 0; i < x.length; i++) x[i](A);
- }
- if (_writer === undefined) return;
- const depth = _printDepth;
- if (_printOpen[depth] === 1) {
- // Printed in pieces: its opener went out as it opened and its children as
- // they finished, so only its closer is left.
- _emitPiece(_printClose[depth]);
- const slot = _printPendingSlot[depth];
- // Nothing inside it printed, so its held-back opener never had to.
- if (slot !== -1 && _writer.isPending(slot)) {
- _printPendingNames.pop();
- _writer.dropPending();
- }
- return;
- }
- // Composed at once, but whatever holds it is being printed in pieces, so it
- // goes straight out rather than into the store for a read-back that will never
- // come.
- if (depth !== 0 && _printOpen[depth - 1] === 1) {
- _emitPiece(_writer.printPiece(A));
- return;
- }
- _writer.printNode(node, A);
- };
- /**
- * First node to visit inside `node` (template content before children).
- * @param {HtmlNodeRef} node node
- * @returns {HtmlNodeRef} first inner node (0 = leaf)
- */
- const _firstInner = (node) => {
- const ty = _nodeTypes[node];
- if (ty === NodeType.Element) {
- const tc = _templateContentOf(node);
- return tc !== 0 ? tc : _nodeFirstChildren[node];
- }
- if (ty === NodeType.Document || ty === NodeType.DocumentFragment) {
- return _nodeFirstChildren[node];
- }
- return 0;
- };
- // === Streaming walk ===
- // The walk follows the parser's open element stack: everything under the
- // deepest *entered* element is finished markup, so it is visited and its node
- // ids recycled, and peak storage is the open subtree rather than the whole
- // document — the HTML analog of the CSS grammar's per-top-level-rule recycling.
- //
- // Streaming starts only once the document passes `_STREAM_MIN_NODES`: below it
- // the walk is a single pass at EOF, which is what a small page wants anyway.
- //
- // Two rules keep it correct and cheap:
- // - An element is entered only when something beneath it has to be flushed
- // (`_FLUSH_BATCH` worth of nodes). Anything that opens and closes in between
- // is never entered and is walked as one completed subtree, so its visitors
- // see a final `end` and live child links.
- // - Nothing is visited unless `_canFlush()` holds. An unsafe stretch (the
- // adoption agency, tables, templates) just buffers, and the next sync sees
- // the final tree.
- //
- // Only walk-only passes stream: printing holds every node's text until its
- // parent prints.
- /** @type {boolean} whether a parse is in progress, so re-entry can be refused */
- let _parsing = false;
- /** @type {boolean} whether this parse may stream its walk (owned by `grammar`) */
- let _streamEligible = false;
- /** set when the root's `enter` called `skipChildren()`, so only its `exit` fires */
- let _docSkipped = false;
- /** @type {boolean} whether the walk has started streaming (see `_STREAM_MIN_NODES`) */
- let _streaming = false;
- /** @type {HtmlElement[]} open elements whose `enter` has already fired */
- const _entered = [];
- /** `_entered` index whose `enter` called `skipChildren()` (-1 = none) */
- let _skipFrom = -1;
- /** set when `open` was spliced mid-stack, so the next sync scans it in full */
- let _openSpliced = false;
- // Set when a mid-stack `open.splice` removes an element the walk already
- // entered (`</form>` with content still open inside it). The element leaves the
- // open stack without its subtree ending, so the open stack no longer describes
- // what has finished and reconciling against it would close and re-enter live
- // elements. Rare enough to simply stop streaming for the rest of the parse: what
- // is already visited stays visited, and the remainder is walked at EOF.
- let _streamHalted = false;
- /** set by `popOpen`: the open stack changed, so the walk has work at token end */
- let _openStackChanged = false;
- // Nodes that must pile up under the deepest entered element before a flush
- // earns its walk setup. Closing an entered element always flushes, so this only
- // bounds how much an un-entered subtree holds — a few KB of columns.
- const _FLUSH_BATCH = 4096;
- // Documents below this many nodes are walked once at EOF instead: recycling
- // only pays once the columns are big enough that not growing them beats the
- // per-token sync, and a page that never reaches it must not carry the cost.
- // Chosen so the streamed cap below stays under `_COLUMN_SHRINK_CAPACITY`,
- // keeping a streamed parse's columns reusable by the next one.
- const _STREAM_MIN_NODES = 49152;
- /**
- * Whether completed subtrees may be visited and recycled at this point in the
- * parse. An empty active-formatting-elements list is the load-bearing check:
- * the adoption agency only moves elements drawn from `open` / `afe` and the
- * direct children of an element that is still open, and table markers live in
- * `afe` too, so with `afe` empty neither it nor foster parenting can reach a
- * node that was already flushed. The rest keep node ids that outlive the open
- * stack out of the recycled range — template contents, pending table text, the
- * `<selectedcontent>` post-pass, the form pointer — and `MODE_IN_BODY` covers
- * the head, which "after head" reopens to insert into.
- * @returns {boolean} true when flushing is safe
- */
- const _canFlush = () =>
- !_streamHalted &&
- afe.length === 0 &&
- templateModes.length === 0 &&
- pendingTableChars === null &&
- !sawSelectedContent &&
- mode === MODE_IN_BODY &&
- open.length !== 0 &&
- (form === 0 || open.includes(form));
- /**
- * Depth-first walk of a whole run of `container`'s children, from `first` up to
- * (not including) `stop` (0 = to the end). One loop for the entire run instead
- * of a call per child — sibling-heavy markup flushes hundreds at a time.
- * @param {HtmlNodeRef} container parent the run hangs off
- * @param {HtmlNodeRef} first first child of the run
- * @param {HtmlNodeRef} stop child to stop before (0 = end of the run)
- */
- const _walkRun = (container, first, stop) => {
- let node = first;
- descend: for (;;) {
- const descending = _enterNode(node);
- // Printing tracks the open path so a node can emit its own pieces; the
- // streamed walk-only runs never print, so they never pay for it.
- if (_writer !== undefined) _openStreamed(node, _printDepth, descending);
- if (descending) {
- const inner = _firstInner(node);
- if (inner !== 0) {
- node = inner;
- _printDepth++;
- continue;
- }
- }
- for (;;) {
- _exitNode(node);
- const parent = _nodeParents[node];
- // Back at the run's own level: step to the next sibling, or stop.
- if (parent === container && _templateContentOf(container) !== node) {
- const next = _nodeNextSiblings[node];
- if (next === 0 || next === stop) return;
- node = next;
- continue descend;
- }
- // Out of a template's content fragment: the element's children follow —
- // they stand in for it, so they share its depth.
- if (_templateContentOf(parent) === node) {
- const inner = _nodeFirstChildren[parent];
- if (inner !== 0) {
- node = inner;
- continue descend;
- }
- } else {
- const next = _nodeNextSiblings[node];
- if (next !== 0) {
- node = next;
- continue descend;
- }
- }
- node = parent;
- _printDepth--;
- }
- }
- };
- /**
- * Rewind the node allocator to `to`, never past the deepest open element. Every
- * still-open element is live and its id must survive, so a flush that empties a
- * parent may not reclaim ids a deeper open element already took.
- * @param {number} to id to rewind to
- */
- const _streamRewind = (to) => {
- const top = open.length !== 0 ? open[open.length - 1] : 0;
- if (to >= top && to < _nodeCount) {
- if (_nodeCount > _nodeHighWater) _nodeHighWater = _nodeCount;
- _nodeCount = to;
- }
- };
- /**
- * Visit and recycle `parent`'s completed children — everything before
- * `openChild`, the child still on the open stack (0 = none, i.e. `parent` is
- * the deepest open element). A trailing text node is held back:
- * `appendTo` merges following text into the last child, so flushing it early
- * would split one text node in two.
- * @param {HtmlNodeRef} parent open element to flush under
- * @param {HtmlElement} openChild still-open child to stop before (0 = none)
- * @param {boolean} visit false to recycle without visiting (under `skipChildren`)
- * @param {boolean} hold true while `parent` can still receive text, so a
- * trailing text node stays put; false once it is closing (nothing follows)
- */
- const _streamFlush = (parent, openChild, visit, hold) => {
- const container = effParent(parent);
- const first = _nodeFirstChildren[container];
- if (first === 0 || first === openChild) return;
- // Where the run ends: before the still-open child, and before a trailing text
- // node while `parent` can still receive text (`appendTo` merges into the last
- // child, so flushing it early would split one text node in two).
- let child = openChild;
- if (hold && openChild === 0) {
- const lastChild = _nodeLastChildren[container];
- if (lastChild !== 0 && _nodeTypes[lastChild] === NodeType.Text) {
- child = lastChild;
- }
- }
- if (first === child) return;
- if (visit) _walkRun(container, first, child);
- _nodeFirstChildren[container] = child;
- if (child === 0) _nodeLastChildren[container] = 0;
- // Ids are bump-allocated in tree order, so the visited run and its
- // descendants fill `[parent + 1, _nodeCount]` whenever nothing below
- // `parent` is still open — rewind the allocator over them.
- if (openChild !== 0 || container !== parent) return;
- if (child === 0) {
- _streamRewind(parent);
- }
- };
- /**
- * Detach a closed element from its parent, releasing its id when it was the
- * parent's last remaining child. Earlier siblings are always flushed before an
- * element is entered, so it is the first child left.
- * @param {HtmlElement} el closed element
- */
- const _streamDetach = (el) => {
- const parent = _nodeParents[el];
- const next = _nodeNextSiblings[el];
- if (parent !== 0) {
- _nodeFirstChildren[parent] = next;
- if (next === 0) _nodeLastChildren[parent] = 0;
- }
- _nodeNextSiblings[el] = 0;
- _nodeParents[el] = 0;
- if (next === 0 && _nodeCount === el) _streamRewind(el - 1);
- };
- /**
- * Bring the streamed walk up to date with the tree built so far: `exit` the
- * elements that closed, `enter` the ones that opened, and flush the completed
- * subtrees in between. Called once per token but does nothing unless
- * `_canFlush()` holds, so an unsafe stretch (incorrectly nested formatting elements,
- * tables, templates) simply buffers and the next catch-up sees the final tree.
- */
- /**
- * Index up to which `_entered` still matches the open stack. Pushes and pops
- * only ever diverge in a suffix; a mid-stack `open.splice` sets `_openSpliced`
- * and forces the full scan.
- * @returns {number} length of the common prefix
- */
- const _enteredPrefix = () => {
- if (_openSpliced) {
- _openSpliced = false;
- let i = 0;
- while (i < _entered.length && i < open.length && _entered[i] === open[i]) {
- i++;
- }
- return i;
- }
- let i = _entered.length < open.length ? _entered.length : open.length;
- while (i > 0 && _entered[i - 1] !== open[i - 1]) i--;
- return i;
- };
- const _streamSync = () => {
- const entered = _entered.length;
- const last = entered !== 0 ? _entered[entered - 1] : 0;
- // Still nothing to do: the entered prefix matches the open stack and too
- // little has piled up under it to be worth a flush.
- if (
- !_openSpliced &&
- open.length >= entered &&
- (entered === 0 || open[entered - 1] === last) &&
- _nodeCount - last < _FLUSH_BATCH
- ) {
- return;
- }
- if (!_canFlush()) return;
- let i = _enteredPrefix();
- // Close: everything entered below the divergence has finished. A descendant
- // of a `skipChildren()` element was tracked without being entered, so it must
- // not exit either — the buffered walk fires `exit` for the skipped element
- // itself and nothing below it.
- for (let k = _entered.length - 1; k >= i; k--) {
- const el = _entered[k];
- const belowSkip = _skipFrom !== -1 && k > _skipFrom;
- _streamFlush(el, 0, _skipFrom === -1 || k < _skipFrom, false);
- if (!belowSkip) _exitNode(el);
- if (_skipFrom === k) _skipFrom = -1;
- _streamDetach(el);
- }
- _entered.length = i;
- // Open: enter the elements now needed, flushing each level's finished
- // children before descending past them.
- for (; i < open.length; i++) {
- const el = open[i];
- const parent = _nodeParents[el];
- const suppressed = _skipFrom !== -1;
- if (parent !== 0) _streamFlush(parent, el, !suppressed, false);
- if (!suppressed && !_enterNode(el)) _skipFrom = i;
- _entered.push(el);
- }
- // Finally the current insertion point's own finished children.
- _streamFlush(open[open.length - 1], 0, _skipFrom === -1, true);
- };
- /**
- * Finish a streamed walk at EOF: close everything still open (their never-
- * entered descendants are walked as completed children), then the root.
- * @param {HtmlNodeRef} root document / fragment root
- */
- const _streamFinish = (root) => {
- for (let k = _entered.length - 1; k >= 0; k--) {
- const el = _entered[k];
- const belowSkip = _skipFrom !== -1 && k > _skipFrom;
- _streamFlush(el, 0, _skipFrom === -1 || k < _skipFrom, false);
- if (!belowSkip) _exitNode(el);
- if (_skipFrom === k) _skipFrom = -1;
- _streamDetach(el);
- }
- _entered.length = 0;
- _streamFlush(root, 0, true, false);
- _exitNode(root);
- };
- /**
- * The HTML `SourceProcessor` grammar: build the document AST (WHATWG tree
- * construction) and walk it, firing `enter` / `exit` in source order. The root
- * document / fragment node is visited too (with a `null` parent). When `writer`
- * is given the same walk also prints: each node's printer builds its text into
- * `writer` as the node finishes (post-order), and the root's text is taken as the
- * result — one parse, no re-tokenization.
- * @param {string} input source text
- * @param {CompiledVisitorMap} visitors compiled visitor map
- * @param {PrintContext | undefined} writer the print context to build output into, or undefined (walk only)
- * @param {HtmlProcessOptions} options process options
- */
- const grammar = (input, visitors, writer, options) => {
- // The walk's state — the visitors, the print context, the open path, the AST
- // columns — is module-scoped so the tokenizer's call sites stay monomorphic,
- // which means one parse at a time: a visitor that started another would take
- // this one's state over and release its columns on the way out. Nesting the
- // *other* language is fine (a `<style>`'s CSS has its own module state), and
- // an `<iframe srcdoc>` is a separate module webpack parses on its own.
- if (_parsing) {
- throw new Error(
- "html syntax: process() is already running; a visitor cannot start another HTML parse"
- );
- }
- _parsing = true;
- _visitors = visitors;
- _writer = writer;
- // Set here, not in `printer`: a tag streams out before the first `printer`
- // call, so these must be known from the start of the walk.
- const redundant =
- writer === undefined ? undefined : writer.options.removeRedundantAttributes;
- _removeRedundantAttributes =
- redundant === true
- ? "smart"
- : redundant === false || redundant === undefined
- ? "none"
- : redundant;
- _removeEmptyAttributes =
- writer !== undefined && writer.options.removeEmptyAttributes === true;
- _removeEmptyElements =
- writer !== undefined && writer.options.removeEmptyElements === true;
- // With the rest rather than in `printer`: they are the print's, not the
- // node's, and that runs once per node.
- _cssEnvironment =
- writer === undefined ? undefined : writer.options.environment;
- _cssConvertLengthUnits =
- writer !== undefined && writer.options.convertLengthUnits === true;
- _cssRewriteCustomProperties =
- writer !== undefined && writer.options.rewriteCustomProperties === true;
- _cssTransforms =
- writer === undefined ? undefined : writer.options.cssTransforms;
- _transforms = _transformsFrom(
- writer === undefined ? undefined : writer.options.transforms
- );
- _commentsKept = _keptComments(_transforms.comments);
- // With the rest, not in `printer`: a tag streams out before the first
- // `printer` call, and both the optional-end-tag tests and
- // `removeEmptyElements` ask what this tier does with a whitespace node.
- const collapse =
- writer === undefined ? undefined : writer.options.collapseWhitespace;
- _collapseWhitespace =
- collapse === true
- ? "conservative"
- : collapse === false || collapse === undefined
- ? "none"
- : collapse;
- _minifying = writer !== undefined && writer.options.mode === "minify";
- _renderEmbeddedSource =
- writer === undefined ? undefined : writer.options.renderEmbeddedSource;
- _deferEmbeddedSource =
- writer === undefined ? undefined : writer.options.deferEmbeddedSource;
- _deferSrcdoc = writer === undefined || writer.options.deferSrcdoc !== false;
- _sortAttributes =
- writer !== undefined && writer.options.sortAttributes === true;
- _sortTokenLists =
- writer !== undefined && writer.options.sortTokenLists === true;
- _mergeStyles = writer !== undefined && writer.options.mergeStyles === true;
- // Node ids are per-parse, so a set kept from the last one would name this
- // one's nodes. Dropped rather than cleared: most documents never fill it.
- _absorbedStyles = null;
- // Counted from the parse this call is about to run, so it cannot be kept.
- _attributeFrequencies = null;
- _attributeRanks = null;
- const implied =
- writer === undefined ? false : writer.options.removeImpliedTags;
- _removeImpliedTags =
- implied === undefined || implied === "smart"
- ? "smart"
- : implied === false
- ? "none"
- : "all";
- // Printing holds every node's text until its parent prints, so only
- // walk-only passes stream; the printing path builds the tree and walks it.
- _streamEligible = writer === undefined;
- _streaming = false;
- _printDepth = 0;
- try {
- const root = parseHtml(input, 0, options);
- let digest = "";
- if (writer !== undefined) {
- if (_fosteredLog !== null) _settleFosteredRuns(A);
- // A shape no arrangement of tags reaches, or foster parenting — whose repair
- // is checked, not trusted. No real document reaches either.
- if (_printFromSource || _fosteredLog !== null) digest = _digest(root);
- _walkRun(_nodeParents[root], root, _nodeNextSiblings[root]);
- // A single root: take its text as the output (the printer read every node's
- // source into the store during the walk, so this needs no source). A root
- // that printed in pieces has already emitted all of it.
- if (_printOpen[0] === 1) writer.dropStore();
- else writer.take(root);
- // Print, then read the output back: the source always parses to its own
- // tree, so it stands in when the printed form does not.
- if (digest !== "") {
- const printed = writer.result();
- if (_digest(parseHtml(printed, 0, options)) !== digest) {
- writer.replaceAll(input);
- }
- }
- }
- } finally {
- // Each held write keeps its offered source — a `<style>`, a `<script>`, a
- // whole `srcdoc` document — and a closure over it, so both go with the parse.
- _renderEmbeddedSource = undefined;
- _deferEmbeddedSource = undefined;
- _parsing = false;
- _removeImpliedTags = "smart";
- _transforms = _DEFAULT_TRANSFORMS;
- _minifying = false;
- // Keyed by this parse's interned names, so it must not outlive it.
- _attributeFrequencies = null;
- _attributeRanks = null;
- // Keyed by this print's options, and holding slices of its input.
- _styleAttributeCache.clear();
- _emptyElements.clear();
- _streamEligible = false;
- _streaming = false;
- _docSkipped = false;
- // Held across the walk, so they outlive it the way the visitors do.
- _printOpen.length = 0;
- _printClose.length = 0;
- _printPendingNames.length = 0;
- _printPendingSlot.length = 0;
- // Hoisting these to module scope outlives the call, so drop the caller's
- // visitors and print context here — they used to fall away with `grammar`.
- _visitors = EMPTY_VISITORS;
- _writer = undefined;
- // The walk consumed the tree: release the side arrays' heap references so
- // the reused columns don't pin this parse's strings until the next parse.
- _releaseAstColumns();
- }
- };
- /**
- * The generic visitor coordinator (`util/SourceProcessor`) bound to the HTML
- * `grammar`. Babel-style usage:
- *
- * ```
- * new SourceProcessor().use({ [NodeType.Element]: (path) => {}, [NodeType.Comment]: { enter, exit } }).process(source, { skip });
- * ```
- * @experimental exposed as `webpack.html.syntax.SourceProcessor`; unstable API
- * @extends {GenericSourceProcessor<HtmlPath, HtmlNodeRef, HtmlProcessOptions>}
- */
- class SourceProcessor extends GenericSourceProcessor {
- constructor() {
- super(grammar, printer);
- }
- }
- /** @typedef {[string, number, number]} ParsedSource */
- // `parseSrcset` is a direct implementation of the WHATWG "parse a srcset
- // attribute" algorithm; it lives here with the other spec-level HTML parsing
- // so it can move into the WASM parser alongside the tokenizer in the future.
- const COMMA = ",".charCodeAt(0);
- const LEFT_PARENTHESIS = "(".charCodeAt(0);
- const RIGHT_PARENTHESIS = ")".charCodeAt(0);
- const SMALL_LETTER_W = "w".charCodeAt(0);
- const SMALL_LETTER_X = "x".charCodeAt(0);
- const SMALL_LETTER_H = "h".charCodeAt(0);
- // (Don't use \s, to avoid matching non-breaking space)
- // Sticky so `collectCharacters` matches at an offset without slicing the
- // whole remaining input per call (which made srcset parsing quadratic).
- // eslint-disable-next-line no-control-regex
- const LEADING_SPACES_REGEXP = /[ \t\n\r\u000C]+/y;
- // eslint-disable-next-line no-control-regex
- const LEADING_COMMAS_OR_SPACES_REGEXP = /[, \t\n\r\u000C]+/y;
- // eslint-disable-next-line no-control-regex
- const LEADING_NOT_SPACES = /[^ \t\n\r\u000C]+/y;
- const TRAILING_COMMAS_REGEXP = /[,]+$/;
- const NON_NEGATIVE_INTEGER_REGEXP = /^\d+$/;
- // ( Positive or negative or unsigned integers or decimals, without or without exponents.
- // Must include at least one digit.
- // According to spec tests any decimal point must be followed by a digit.
- // No leading plus sign is allowed.)
- // https://html.spec.whatwg.org/multipage/infrastructure.html#valid-floating-point-number
- const FLOATING_POINT_REGEXP =
- /^-?(?:[0-9]+|[0-9]*\.[0-9]+)(?:[eE][+-]?[0-9]+)?$/;
- /**
- * @param {string} input input
- * @returns {ParsedSource[]} parsed srcset
- */
- const parseSrcset = (input) => {
- // 1. Let input be the value passed to this algorithm.
- const inputLength = input.length;
- /** @type {string | undefined} */
- let url;
- /** @type {string[]} */
- let descriptors;
- /** @type {number} */
- let descriptorStart;
- /** @type {string} */
- let state;
- /** @type {number} */
- let charCode;
- /** @type {number} */
- let position = 0;
- /** @type {number} */
- let start;
- /** @type {[string, number, number][]} */
- const candidates = [];
- /**
- * @param {RegExp} regExp sticky reg exp to collect characters
- * @returns {string | undefined} characters
- */
- function collectCharacters(regExp) {
- regExp.lastIndex = Math.max(0, position);
- const match = regExp.exec(input);
- if (match) {
- const [chars] = match;
- position += chars.length;
- return chars;
- }
- }
- /**
- * @returns {void}
- */
- function parseDescriptors() {
- // 9. Descriptor parser: Let error be no.
- let pError = false;
- // 10. Let width be absent.
- // 11. Let density be absent.
- // 12. Let future-compat-h be absent. (We're implementing it now as h)
- /** @type {number | undefined} */
- let width;
- /** @type {number | undefined} */
- let density;
- /** @type {number | undefined} */
- let height;
- /** @type {string | undefined} */
- let desc;
- // 13. For each descriptor in descriptors, run the appropriate set of steps
- // from the following list:
- for (let i = 0; i < descriptors.length; i++) {
- desc = descriptors[i];
- const lastChar = desc[desc.length - 1].charCodeAt(0);
- const value = desc.slice(0, Math.max(0, desc.length - 1));
- // If the descriptor consists of a valid non-negative integer followed by
- // a U+0077 LATIN SMALL LETTER W character
- if (
- NON_NEGATIVE_INTEGER_REGEXP.test(value) &&
- lastChar === SMALL_LETTER_W
- ) {
- // If width and density are not both absent, then let error be yes.
- if (width || density) {
- pError = true;
- }
- const intVal = Number.parseInt(value, 10);
- // Apply the rules for parsing non-negative integers to the descriptor.
- // If the result is zero, let error be yes.
- // Otherwise, let width be the result.
- if (intVal === 0) {
- pError = true;
- } else {
- width = intVal;
- }
- }
- // If the descriptor consists of a valid floating-point number followed by
- // a U+0078 LATIN SMALL LETTER X character
- else if (
- FLOATING_POINT_REGEXP.test(value) &&
- lastChar === SMALL_LETTER_X
- ) {
- // If width, density and future-compat-h are not all absent, then let error
- // be yes.
- if (width || density || height) {
- pError = true;
- }
- const floatVal = Number.parseFloat(value);
- // Apply the rules for parsing floating-point number values to the descriptor.
- // If the result is less than zero, let error be yes. Otherwise, let density
- // be the result.
- if (floatVal < 0) {
- pError = true;
- } else {
- density = floatVal;
- }
- }
- // If the descriptor consists of a valid non-negative integer followed by
- // a U+0068 LATIN SMALL LETTER H character
- else if (
- NON_NEGATIVE_INTEGER_REGEXP.test(value) &&
- lastChar === SMALL_LETTER_H
- ) {
- // If height and density are not both absent, then let error be yes.
- if (height || density) {
- pError = true;
- }
- const intVal = Number.parseInt(value, 10);
- // Apply the rules for parsing non-negative integers to the descriptor.
- // If the result is zero, let error be yes. Otherwise, let future-compat-h
- // be the result.
- if (intVal === 0) {
- pError = true;
- } else {
- height = intVal;
- }
- // Anything else, Let error be yes.
- } else {
- pError = true;
- }
- }
- // 15. If error is still no, then append a new image source to candidates whose
- // URL is url, associated with a width width if not absent and a pixel
- // density density if not absent. Otherwise, there is a parse error.
- if (!pError) {
- candidates.push([
- /** @type {string} */ (url),
- start,
- start + /** @type {string} */ (url).length
- ]);
- } else {
- throw new Error(
- `Invalid srcset descriptor found in '${input}' at '${desc}'`
- );
- }
- }
- /**
- * @returns {void}
- */
- function tokenizeDescriptor() {
- // 8.1. Descriptor tokenizer: Skip whitespace
- collectCharacters(LEADING_SPACES_REGEXP);
- // 8.2. Let current descriptor be the empty string.
- // (Tracked as a start offset, `-1` = empty; sliced once per descriptor.)
- descriptorStart = -1;
- // 8.3. Let state be in descriptor.
- state = "in descriptor";
- while (true) {
- // 8.4. Let charCode be the character at position.
- charCode = input.charCodeAt(position);
- // Do the following depending on the value of state.
- // For the purpose of this step, "EOF" is a special character representing
- // that position is past the end of input.
- // In descriptor
- if (state === "in descriptor") {
- // Do the following, depending on the value of charCode:
- // Space character
- // If current descriptor is not empty, append current descriptor to
- // descriptors and let current descriptor be the empty string.
- // Set state to after descriptor.
- if (isSpace(charCode)) {
- if (descriptorStart !== -1) {
- descriptors.push(input.slice(descriptorStart, position));
- descriptorStart = -1;
- state = "after descriptor";
- }
- }
- // U+002C COMMA (,)
- // Advance position to the next character in input. If current descriptor
- // is not empty, append current descriptor to descriptors. Jump to the step
- // labeled descriptor parser.
- else if (charCode === COMMA) {
- position += 1;
- if (descriptorStart !== -1) {
- descriptors.push(input.slice(descriptorStart, position - 1));
- }
- parseDescriptors();
- return;
- }
- // U+0028 LEFT PARENTHESIS (()
- // Append charCode to current descriptor. Set state to in parens.
- else if (charCode === LEFT_PARENTHESIS) {
- if (descriptorStart === -1) descriptorStart = position;
- state = "in parens";
- }
- // EOF
- // If current descriptor is not empty, append current descriptor to
- // descriptors. Jump to the step labeled descriptor parser.
- else if (Number.isNaN(charCode)) {
- if (descriptorStart !== -1) {
- descriptors.push(input.slice(descriptorStart, position));
- }
- parseDescriptors();
- return;
- // Anything else
- // Append charCode to current descriptor.
- } else if (descriptorStart === -1) {
- descriptorStart = position;
- }
- }
- // In parens
- else if (state === "in parens") {
- // U+0029 RIGHT PARENTHESIS ())
- // Append charCode to current descriptor. Set state to in descriptor.
- if (charCode === RIGHT_PARENTHESIS) {
- state = "in descriptor";
- }
- // EOF
- // Append current descriptor to descriptors. Jump to the step labeled
- // descriptor parser.
- else if (Number.isNaN(charCode)) {
- descriptors.push(input.slice(descriptorStart, position));
- parseDescriptors();
- return;
- }
- // Anything else
- // Append charCode to current descriptor. (Covered by the tracked range.)
- }
- // After descriptor
- else if (state === "after descriptor") {
- // Do the following, depending on the value of charCode:
- if (isSpace(charCode)) {
- // Space character: Stay in this state.
- }
- // EOF: Jump to the step labeled descriptor parser.
- else if (Number.isNaN(charCode)) {
- parseDescriptors();
- return;
- }
- // Anything else
- // Set state to in descriptor. Set position to the previous character in input.
- else {
- state = "in descriptor";
- position -= 1;
- }
- }
- // Advance position to the next character in input.
- position += 1;
- }
- }
- // 3. Let candidates be an initially empty source set.
- // const candidates = []; // Moved to top
- // 4. Splitting loop: Collect a sequence of characters that are space
- // characters or U+002C COMMA characters. If any U+002C COMMA characters
- // were collected, that is a parse error.
- while (true) {
- collectCharacters(LEADING_COMMAS_OR_SPACES_REGEXP);
- // 5. If position is past the end of input, return candidates and abort these steps.
- if (position >= inputLength) {
- if (candidates.length === 0) {
- throw new Error("Must contain one or more image candidate strings");
- }
- // (we're done, this is the sole return path)
- return candidates;
- }
- // 6. Collect a sequence of characters that are not space characters,
- // and let that be url.
- start = position;
- url = collectCharacters(LEADING_NOT_SPACES);
- // 7. Let descriptors be a new empty list.
- descriptors = [];
- // 8. If url ends with a U+002C COMMA character (,), follow these sub steps:
- // (1). Remove all trailing U+002C COMMA characters from url. If this removed
- // more than one character, that is a parse error.
- if (url && url.charCodeAt(url.length - 1) === COMMA) {
- url = url.replace(TRAILING_COMMAS_REGEXP, "");
- // (Jump ahead to step 9 to skip tokenization and just push the candidate).
- parseDescriptors();
- }
- // Otherwise, follow these sub steps:
- else {
- tokenizeDescriptor();
- }
- // 16. Return to the step labeled splitting loop.
- }
- };
- // The spec's "URL-potentially-surrounded-by-spaces" cleanup: C0 controls,
- // DEL..C1 and U+00A0 are stripped from the value after the whitespace trim.
- // eslint-disable-next-line no-control-regex
- const IGNORE_CHARS_REGEXP = /[\u0000-\u001F\u007F-\u009F\u00A0]/g;
- /**
- * Parse a `src`-like URL attribute value: trim ASCII whitespace / U+00A0 from
- * both ends (offsets preserved for source rewriting), then strip ignorable
- * control characters. Throws when nothing remains.
- * @param {string} input attribute value
- * @returns {ParsedSource[]} parsed src
- */
- const parseSrc = (input) => {
- const len = input.length;
- if (len === 0) throw new Error("Must be non-empty");
- let start = 0;
- let end = len;
- while (start < end) {
- const code = input.charCodeAt(start);
- if (code > 32 && code !== 160) break;
- start++;
- }
- if (start === end) throw new Error("Must be non-empty");
- while (end > start) {
- const code = input.charCodeAt(end - 1);
- if (code > 32 && code !== 160) break;
- end--;
- }
- let value = input.slice(start, end);
- if (IGNORE_CHARS_REGEXP.test(value)) {
- value = value.replace(IGNORE_CHARS_REGEXP, "");
- if (value.length === 0) throw new Error("Must be non-empty");
- }
- return [[value, start, end]];
- };
- /**
- * Extracts the `icon-uri` value of an `msapplication-task` meta content
- * (`name=…;action-uri=…;icon-uri=…`) — the other parts are page URLs, not assets.
- * @param {string} input input
- * @returns {ParsedSource[]} parsed icon-uri
- */
- const parseMsapplicationTask = (input) => {
- const len = input.length;
- let pos = 0;
- while (pos < len) {
- let sep = input.indexOf(";", pos);
- if (sep === -1) sep = len;
- const eq = input.indexOf("=", pos);
- if (eq !== -1 && eq < sep) {
- const key = input.slice(pos, eq).trim().toLowerCase();
- if (key === "icon-uri") {
- let start = eq + 1;
- let end = sep;
- while (start < end && isSpace(input.charCodeAt(start))) {
- start++;
- }
- while (end > start && isSpace(input.charCodeAt(end - 1))) {
- end--;
- }
- if (start === end) return [];
- return [[input.slice(start, end), start, end]];
- }
- }
- pos = sep + 1;
- }
- return [];
- };
- // CSS syntax is loaded on first `parseCssUrls` call, not at module load —
- // `HtmlGenerator` deliberately defers it the same way, and most documents
- // never carry a URL-bearing SVG presentation attribute.
- /** @type {typeof import("../css/syntax") | undefined} */
- let _cssSyntax;
- /**
- * Extracts `url(...)` references from a CSS value (an SVG presentation
- * attribute). Reuses webpack's CSS lexer so quoting/escaping match the CSS
- * spec; returns the `parseSrc` shape so the shared emit path maps and rewrites
- * the spans. Unquoted `url(path)` rewrites the content span; quoted
- * `url("path")` rewrites the inner string span (quotes preserved).
- * @param {string} input attribute value (a CSS component-value list)
- * @returns {ParsedSource[]} url references
- */
- const parseCssUrls = (input) => {
- const {
- TT_EOF,
- TT_FUNCTION,
- TT_STRING,
- TT_URL,
- TT_WHITESPACE,
- TokenStream,
- equalsLowerCase
- } = _cssSyntax || (_cssSyntax = require("../css/syntax"));
- const ts = new TokenStream(input);
- /** @type {ParsedSource[]} */
- const result = [];
- for (;;) {
- const t = ts.consume();
- if (t.type === TT_EOF) break;
- if (t.type === TT_URL) {
- if (t.contentEnd > t.contentStart) {
- result.push([
- input.slice(t.contentStart, t.contentEnd),
- t.contentStart,
- t.contentEnd
- ]);
- }
- } else if (
- t.type === TT_FUNCTION &&
- equalsLowerCase(input.slice(t.start, t.end - 1), "url")
- ) {
- let s = ts.consume();
- while (s.type === TT_WHITESPACE) s = ts.consume();
- if (s.type === TT_STRING) {
- const quote = input.charCodeAt(s.start);
- const innerStart = s.start + 1;
- // Drop the closing quote, unless the string is unterminated at EOF.
- const innerEnd =
- input.charCodeAt(s.end - 1) === quote ? s.end - 1 : s.end;
- if (innerEnd > innerStart) {
- result.push([
- input.slice(innerStart, innerEnd),
- innerStart,
- innerEnd
- ]);
- }
- }
- }
- }
- return result;
- };
- // Babel's `path.skip()`, children-only: set by `A.skipChildren()` during an
- // `enter` dispatch, consumed by the walk.
- let _walkSkip = false;
- // The walk's current position (`A.node` / `A.parent` read these; module-level
- // so the accessor methods' defaults avoid self-referential `this` typing).
- /** @type {HtmlNodeRef} */
- let _currentNode = 0;
- /** @type {HtmlNodeRef | null} */
- let _currentParent = null;
- /* eslint-disable jsdoc/require-template -- `A` below is the accessor const, not a type parameter */
- /**
- * The HTML path (Babel's `path` shape): the AST accessor with the walk's
- * current position on it — the single argument every visitor receives.
- * @typedef {typeof A} HtmlPath
- */
- /* eslint-enable jsdoc/require-template */
- // AST field-access seam (mirrors the CSS parser's `A`): every AST field a
- // consumer reads goes through one of these accessors, so the node
- // representation can change underneath without touching consumers. `n` is an
- // `HtmlNodeRef`; results are valid until the next `parseHtml` call.
- const A = {
- // === path position (rebound by the walk before every visitor call) ===
- /**
- * @returns {HtmlNodeRef} current node — only valid during a visitor callback
- */
- get node() {
- return _currentNode;
- },
- /**
- * @returns {HtmlNodeRef | null} enclosing node (null = the document root)
- */
- get parent() {
- return _currentParent;
- },
- /** Stop the walk descending into the current node (enter only). */
- skipChildren() {
- _walkSkip = true;
- },
- // === field reads — `n` defaults to the current node ===
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {number} `NodeType`
- */
- type(n = _currentNode) {
- return _nodeTypes[n];
- },
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {number} start offset
- */
- start(n = _currentNode) {
- return _nodeStarts[n];
- },
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {number} end offset
- */
- end(n = _currentNode) {
- return _nodeEnds[n];
- },
- /**
- * Raw source slice `[start, end)` — valid only during the walk (the printer's
- * window), before `parseHtml` releases `_htmlSource`.
- * @param {HtmlNodeRef=} n node
- * @returns {string} raw source slice
- */
- source(n = _currentNode) {
- return _htmlSource.slice(_nodeStarts[n], _nodeEnds[n]);
- },
- /**
- * @param {number} from start offset
- * @param {number} to end offset
- * @returns {string} raw source between the two offsets
- */
- sourceSpanAt(from, to) {
- return _htmlSource.slice(from, to);
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {string} lowercased (foreign-content: adjusted) tag name
- */
- tagName(n = _currentNode) {
- return _nodeStrings[n];
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {number} `NS_*` namespace
- */
- namespace(n = _currentNode) {
- return _nodeFlags[n] & NS_MASK;
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {boolean} true for void elements
- */
- selfClosing(n = _currentNode) {
- return (_nodeFlags[n] & FLAG_SELF_CLOSING) !== 0;
- },
- // materialized attribute list — test/tooling convenience, allocates; the
- // parser reads attributes through the scalar accessors below
- /**
- * @param {HtmlElement=} n element
- * @returns {HtmlAttribute[]} materialized attributes
- */
- attributes(n = _currentNode) {
- const out = [];
- const start = _nodeAttrStarts[n];
- for (let i = start; i < start + _nodeAttrCounts[n]; i++) {
- out.push({
- name: _attrNames[i],
- value: _attrValueOf(i),
- serializedName: _attrSerializedName(i),
- nameStart: _attrNameStarts[i],
- nameEnd: _attrNameEnds[i],
- valueStart: _attrValueStarts[i],
- valueEnd: _attrValueEnds[i]
- });
- }
- return out;
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {number} attribute count
- */
- attributeCount(n = _currentNode) {
- return _nodeAttrCounts[n];
- },
- /**
- * The i-th attribute of an element, as an id for the `attribute*` reads.
- * @param {number} i attribute index
- * @param {HtmlElement=} n element
- * @returns {HtmlAttributeRef} attribute ref
- */
- attributeAt(i, n = _currentNode) {
- return _nodeAttrStarts[n] + i;
- },
- /**
- * Linear lookup by (lowercased) name.
- * @param {string} name attribute name
- * @param {HtmlElement=} n element
- * @returns {HtmlAttributeRef} attribute ref (0 = not present)
- */
- findAttribute(name, n = _currentNode) {
- return _findAttr(_nodeAttrStarts[n], _nodeAttrCounts[n], name);
- },
- /**
- * @param {HtmlAttributeRef} a attribute ref
- * @returns {string} lowercased (foreign-content: adjusted) attribute name
- */
- attributeName(a) {
- return _attrNames[a];
- },
- /**
- * @param {HtmlAttributeRef} a attribute ref
- * @returns {string} raw (undecoded) attribute value ("" when valueless)
- */
- attributeValue(a) {
- return _attrValueOf(a);
- },
- /**
- * @param {HtmlAttributeRef} a attribute ref
- * @returns {number} name start offset
- */
- attributeNameStart(a) {
- return _attrNameStarts[a];
- },
- /**
- * @param {HtmlAttributeRef} a attribute ref
- * @returns {number} name end offset
- */
- attributeNameEnd(a) {
- return _attrNameEnds[a];
- },
- /**
- * @param {HtmlAttributeRef} a attribute ref
- * @returns {number} value start offset (-1 when valueless or on adoption-agency clones)
- */
- attributeValueStart(a) {
- return _attrValueStarts[a];
- },
- /**
- * @param {HtmlAttributeRef} a attribute ref
- * @returns {number} value end offset
- */
- attributeValueEnd(a) {
- return _attrValueEnds[a];
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {number} end offset of the opening tag (after `>`)
- */
- tagEnd(n = _currentNode) {
- return _nodeTagEnds[n];
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {number} end offset of the tag name
- */
- nameEnd(n = _currentNode) {
- return _nodeNameEnds[n];
- },
- /**
- * Raw source of an element's opening tag, `[start, tagEnd)` — attribute quoting
- * / spacing / case preserved byte-for-byte (walk-window only) — or `""` for a
- * parser-inserted element (auto `html`/`head`/`body`/`tbody`, …), which has no
- * real source tag: its offsets are zero-width or borrow the triggering token,
- * so the sliced name doesn't match this element. The empty string lets a printer
- * treat such an element as transparent.
- * @param {HtmlElement=} n element
- * @returns {string} opening-tag source, or `""` when parser-inserted
- */
- /**
- * Whether the source wrote this element's end tag rather than the parser
- * popping it for an implied close. Read back off the range instead of marked
- * during the parse: an element's end spans the token that closed it, so its
- * own end tag is the last thing in it — and only the few elements around a
- * region printed from source ever ask.
- * @param {HtmlElement=} n element
- * @returns {boolean} whether the source wrote its end tag
- */
- sourceClosed(n = _currentNode) {
- const end = _nodeEnds[n];
- if (end === 0 || _htmlSource.charCodeAt(end - 1) !== 62) return false;
- let i = end - 2;
- for (;;) {
- const c = i >= 0 ? _htmlSource.charCodeAt(i) : 0;
- if (c !== 0x20 && c !== 0x09 && c !== 0x0a && c !== 0x0c && c !== 0x0d) {
- break;
- }
- i--;
- }
- const name = _tagNameOf(n);
- const nameEnd = i + 1;
- const nameStart = nameEnd - name.length;
- return (
- nameStart >= 2 &&
- _htmlSource.charCodeAt(nameStart - 1) === 47 &&
- _htmlSource.charCodeAt(nameStart - 2) === 60 &&
- rangeEqualsLowerCase(_htmlSource, nameStart, nameEnd, name)
- );
- },
- openTag(n = _currentNode) {
- const start = _nodeStarts[n];
- const nameEnd = _nodeNameEnds[n];
- const stored = _nodeStrings[n];
- // Fold the raw range against the stored name in place — no name slice, no
- // `toLowerCase` pair. An adjusted foreign name (`foreignObject`) carries
- // upper case the fold cannot match, so a miss falls back to the old compare.
- if (rangeEqualsLowerCase(_htmlSource, start + 1, nameEnd, stored)) {
- return _htmlSource.slice(start, _nodeTagEnds[n]);
- }
- const name = _htmlSource.slice(start + 1, nameEnd);
- return name.toLowerCase() === stored.toLowerCase()
- ? _htmlSource.slice(start, _nodeTagEnds[n])
- : "";
- },
- /**
- * An element's end tag, generated as `</name>` from the opening tag's own name
- * (exact source casing, correct for foreign camelCase elements). Generated, not
- * sliced: element `end` offsets don't span the end tag, and an omitted optional
- * end tag (`<li>`, `<p>`, …) still serializes to the same DOM. `""` when the
- * parser inserted the element, as {@link openTag} does — it has no name in the
- * source to echo, and slicing one would spell `</>`.
- * @param {HtmlElement=} n element
- * @returns {string} closing-tag text, or `""` when parser-inserted
- */
- closeTag(n = _currentNode) {
- const nameEnd = _nodeNameEnds[n];
- return nameEnd === 0
- ? ""
- : `</${_htmlSource.slice(_nodeStarts[n] + 1, nameEnd)}>`;
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {number} under `skip.text`, end offset of a raw-text element's body (`tagEnd` when empty)
- */
- contentEnd(n = _currentNode) {
- return (_nodeFlags[n] & FLAG_HAS_TEMPLATE) !== 0
- ? _nodeTagEnds[n]
- : _nodeContentEnds[n];
- },
- /**
- * @param {HtmlElement=} n element
- * @returns {HtmlDocumentFragment} `<template>` content fragment (0 = none)
- */
- templateContent(n = _currentNode) {
- return _templateContentOf(n);
- },
- /**
- * @param {HtmlText | HtmlComment | HtmlProcessingInstruction=} n text / comment / processing instruction node
- * @returns {string} decoded text / comment / processing instruction data
- */
- data(n = _currentNode) {
- return _nodeStrings[n];
- },
- /**
- * @param {HtmlProcessingInstruction=} n processing instruction node
- * @returns {string} processing instruction target
- */
- piTarget(n = _currentNode) {
- return _piTargets.get(n) || "";
- },
- /**
- * @param {HtmlDoctype=} n doctype node
- * @returns {string} doctype name
- */
- doctypeName(n = _currentNode) {
- return _nodeStrings[n];
- },
- // The doctype ids are per-parse scalars (a document has at most one
- // doctype node); the node parameter is accepted for call-shape uniformity.
- /**
- * @param {HtmlDoctype=} _n doctype node
- * @returns {string | null} doctype public id
- */
- doctypePublicId(_n) {
- return _doctypePublicId;
- },
- /**
- * @param {HtmlDoctype=} _n doctype node
- * @returns {string | null} doctype system id
- */
- doctypeSystemId(_n) {
- return _doctypeSystemId;
- },
- // === tree links (0 = none) ===
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {HtmlNodeRef} first child
- */
- firstChild(n = _currentNode) {
- return _nodeFirstChildren[n];
- },
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {HtmlNodeRef} next sibling
- */
- nextSibling(n = _currentNode) {
- return _nodeNextSiblings[n];
- },
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {HtmlNodeRef} parent node (a `<template>`'s content links to its fragment)
- */
- parentOf(n = _currentNode) {
- return _nodeParents[n];
- },
- /**
- * @param {HtmlNodeRef=} n node
- * @returns {HtmlNodeRef[]} materialized child list — test/tooling convenience, allocates
- */
- children(n = _currentNode) {
- const out = [];
- for (let c = _nodeFirstChildren[n]; c !== 0; c = _nodeNextSiblings[c]) {
- out.push(c);
- }
- return out;
- }
- };
- module.exports.A = A;
- module.exports.EMBEDDED_LANGUAGES = EMBEDDED_LANGUAGES;
- module.exports.NS_HTML = NS_HTML;
- module.exports.NS_MATHML = NS_MATHML;
- module.exports.NS_SVG = NS_SVG;
- module.exports.NodeType = NodeType;
- module.exports.QUOTE_DOUBLE = QUOTE_DOUBLE;
- module.exports.QUOTE_NONE = QUOTE_NONE;
- module.exports.QUOTE_SINGLE = QUOTE_SINGLE;
- // Exposed so HtmlParser can map user-configured (lowercased) tag names to
- // the adjusted camelCase names the AST carries for foreign content.
- module.exports.SVG_TAG_ADJUST = SVG_TAG_ADJUST;
- module.exports.SourceProcessor = SourceProcessor;
- module.exports.askEmbeddedRenderer = askEmbeddedRenderer;
- module.exports.baseTag = baseTag;
- module.exports.buildHeadTags = buildHeadTags;
- module.exports.collectEmbeddedDiagnostics = collectEmbeddedDiagnostics;
- module.exports.decodeEntities = decodeEntities;
- module.exports.embeddedText = embeddedText;
- module.exports.escapeAttribute = escapeAttribute;
- module.exports.escapeText = escapeText;
- module.exports.isAsciiWhitespace = isSpace;
- module.exports.metaTag = metaTag;
- // WHATWG "ASCII whitespace" (tab / LF / FF / CR / space) — the tokenizer's
- // whitespace class, exported under the spec's name.
- module.exports.parseCssUrls = parseCssUrls;
- module.exports.parseHtml = parseHtml;
- module.exports.parseMsapplicationTask = parseMsapplicationTask;
- module.exports.parseSrc = parseSrc;
- module.exports.parseSrcset = parseSrcset;
- module.exports.pickTransforms = pickTransforms;
- module.exports.printer = printer;
- module.exports.tokenize = tokenize;
|