ELF          >            @       P         @ 8  @                                 x       x                     0       0       0                                                                           p     p     p           p                                                                             $       $                                   @       @              Std                  @       @              Ptd   ԃ     ԃ     ԃ                        Ttd   P     P     P     4       4              Qtd                                                  Rtd   p     p     p                                 GNU `y,PCC[b{EݳW    ?   B          	ER@D"  @%nh    B       C               D               E   F       G   H                       I               K   M               O   P               Q   R   T       U                       W   X       Y       Z   [   \   ^   _   `   a           b       c   d   ygk7CҐC\w}nKBa#]s<spyOϛ5|DדvKvJ\4YV#M
٭ZW_MR&9 ;GzVwAt3ƃ5JSvc                                                                    +                                          E                     K                                                                                       A                     -                                          a                     t                     U                      6                     (                                          R                     b                                                                                        p                      ,                       	                                          F   "                                                             $                     /                                                                                      h                                                                                                         s                                                                                                                                                                                              T                     z                                                                                     y                     V                                                                                                          z                                           h                      a                     Z                     r   
                 
 б      V       	   
       ?          
       &         
                 
  e                
 0                
                 
 `a             6   
 0                
                 
 b      )         
 b              	   
        E       M   
 @                
 g               
 Я               
 pg      
          
 g      W       T   
       E          
 f             1   
 f                
                 
 Э      1      /   
 e            m   
 @             :	   
 p                
                  
       	      B   
 Ы             G   
 f             d   
 `                
 b      t          
 d                
               __gmon_start__ _ITM_deregisterTMCloneTable _ITM_registerTMCloneTable __cxa_finalize __cxa_atexit hypot hb_free hb_malloc memcpy hb_realloc __stack_chk_fail hb_draw_funcs_get_empty hb_draw_funcs_destroy memset pthread_mutex_lock pthread_mutex_unlock hb_calloc pthread_mutex_init pthread_mutex_destroy hb_color_line_get_color_stops hb_map_has hb_map_get hb_paint_funcs_get_empty hb_paint_funcs_destroy hb_gpu_shader_source hb_gpu_draw_create_or_fail hb_gpu_draw_reference hb_gpu_draw_destroy hb_blob_destroy hb_gpu_draw_set_user_data hb_gpu_draw_get_user_data hb_gpu_draw_get_funcs hb_draw_funcs_create hb_draw_funcs_set_move_to_func hb_draw_funcs_set_line_to_func hb_draw_funcs_set_quadratic_to_func hb_draw_funcs_set_cubic_to_func hb_draw_funcs_set_close_path_func hb_draw_funcs_make_immutable hb_gpu_draw_set_scale hb_gpu_draw_get_scale hb_gpu_draw_glyph_or_fail hb_font_get_scale hb_font_draw_glyph_or_fail hb_gpu_draw_glyph hb_gpu_draw_clear hb_gpu_draw_encode hb_blob_get_empty round hb_blob_create hb_color_line_get_extend hb_paint_reduce_linear_anchors hb_paint_normalize_color_line hb_gpu_draw_reset hb_gpu_draw_recycle_blob hb_gpu_draw_shader_source hb_gpu_paint_create_or_fail hb_gpu_paint_reference hb_gpu_paint_destroy hb_map_destroy hb_gpu_paint_set_user_data hb_gpu_paint_get_user_data hb_gpu_paint_get_funcs hb_paint_funcs_create hb_paint_funcs_set_push_transform_func hb_paint_funcs_set_pop_transform_func hb_paint_funcs_set_push_clip_glyph_func hb_paint_funcs_set_push_clip_path_start_func hb_paint_funcs_set_push_clip_path_end_func hb_paint_funcs_set_pop_clip_func hb_paint_funcs_set_push_group_func hb_paint_funcs_set_pop_group_func hb_paint_funcs_set_color_func hb_paint_funcs_set_linear_gradient_func hb_paint_funcs_set_radial_gradient_func hb_paint_funcs_set_sweep_gradient_func hb_paint_funcs_set_custom_palette_color_func hb_paint_funcs_set_image_func hb_paint_funcs_make_immutable hb_gpu_paint_set_palette hb_gpu_paint_get_palette hb_gpu_paint_clear_custom_palette_colors hb_map_clear hb_gpu_paint_set_custom_palette_color hb_map_set hb_map_allocation_successful hb_map_create hb_gpu_paint_set_scale hb_gpu_paint_get_scale hb_gpu_paint_glyph_or_fail hb_font_paint_glyph_or_fail hb_gpu_paint_glyph hb_font_paint_glyph hb_gpu_paint_clear hb_gpu_paint_encode hb_blob_get_length hb_blob_get_data hb_gpu_paint_reset hb_gpu_paint_recycle_blob hb_gpu_paint_shader_source libharfbuzz.so.0 libm.so.6 libc.so.6 libharfbuzz-gpu.so.0 GLIBC_2.2.5 GLIBC_2.35 GLIBC_ABI_DT_RELR GLIBC_2.14 GLIBC_2.4 XXXXXXXX                                                                                                                                                               f	     0   ui	   	        	        p	         B    	        	     ii   	     ui	   	              Q                                      R                              ȭ        K           Э                   ح                                                                                                        	                   
                                                                     (                   0                   8                   @                   H                   P                   X                   `                   h                   p                   x                                                                                                                                                                M                               Ȯ        !           Ю        ^           خ        "                   #                   $                   %                   &                    '                   (                   )                   *                    +           (        T           0        b           8        ,           @        Z           H        -           P        .           X        /           `        0           h        1           p        2           x        V                   3                   4                   5                   6                   7                   8                   9                   :                   `           ȯ        ;           Я        <           د        =                   >                   ?                   @                   A           p                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          HH} HtH                                     H= H H9tH~ Ht	        H= H5 H)HH?HHHtH} HtfD      =e  u3UH=}  HtH=. } c< ]f.     @ ff.     gf.     f.     f.     f.     f.     D  H  UHAVAUIATISHHA$M]IKOHHk8DIHk8L1f/1f/@)uE$f(AɉAHHk8D1f/1f/@)   ARLIREJf     GHwIHk8D1f/1f/)ȃt^     ff.     BHHHk8D1f/E1f/AD)tH9s#DABIHk8LHmD  DL)LwHEBI)IL9   LHL]MuLmIH[A\A]A^]ÉE
A$HHk8DIHk8L1f/1f/@)E$f(AɉHLL@ H  UHAVAUIATISHHA$M]IKOHHk8IHk81f/1f/@)uE$f(AɉAHHk81f/1f/@)   ARLIREJff.     GHwIHk81f/1f/)ȃtM@ BHHHk81f/E1f/AD)tH9sDABIHk8HfDL)LwHEBI)IL9   LHL]MLmIH[A\A]A^]ÉE
A$HHk8IHk81f/1f/@)E$f(AɉHLL ff.     H  UHAVAUIATISHHA$M]IKOHHk8DIHk8L1f/1f/@)uE$f(AɉAHHk8D1f/1f/@)   ARLIREJf     GHwIHk8D1f/1f/)ȃtN BHHHk8D1f/E1f/AD)tH9s#DABIHk8LH}D  DL)LwHEBI)IL9   LHL]MLmIH[A\A]A^]ÉE
A$HHk8DIHk8L1f/1f/@)E$f(AɉHLL@ H  UHAVAUIATISHHA$M]IKOHHk8DIHk8L1f/1f/@)uE$f(AɉAHHk8D1f/1f/@)   ARLIREJf     GHwIHk8D1f/1f/)ȃtN BHHHk8D1f/E1f/AD)tH9sDABIHk8LHDL)LwHEBI)IL9   LHL]MLmIH[A\A]A^]ÉE
A$HHk8DIHk8L1f/1f/@)E$f(AɉHLLÐff.     Ɔe  1fUfDoflfDlfofoflflHSHXD)])M)ef(f()Uff(t =HI f(Uf/Y  u1H]@ fD(EH )UH fDXfffDYfDXEfDXEfDYfA(fA(D)Eff(&t f/H wfD(Mf(UfDXMf(Mf(H fD(EDH f(fD\fA(fA(fEffA(f(fD\fDYf(f\fAYfA\D)Mf(ffXff(ffD(EfD( H f(uf(Ef(UD)Ef\fAXfAYfX)Ef     f(mf(f(ff(s 5G f(Uf/uH]    H  wqUHHUoF H~0HMfoG E       MfoMH}foE      oE   H  H  ]fƆe  H  @ ff.     H  t	H       ~~V(~HY~F~X~FY~~XЅuV4    UHV(HH@@,H~.@4Hv H0zLuJ.H8zDuBHO8LG0HtHI HE)UAHE(U@(    H@,    P4f     HO8LGHtHIHEHUHuH})UAHE(UHUHuH}fU(HV(~HSHHH~~^C(H~Y~FHv ~X~Y~~X(((t/HO8HGHtHI)U(((US4H]@ HO8K8C4HGHtH	HUHuH})Ue]HC4C(   (Ue]HUHC,HuH}m@ U(~HV(HSHHH~~~~vDND~~C(AYD~YH~DYY~X~XAYDDY~XDXFHv ~XEX((t.HO8HG HtHI)e(A((ec4H]fHO8K8C4HGHtH	HUHuH})e]UuDEHC4C(   (e]UuHC,DEHUHuH}XfD  UD(~D(HV(HSHHX~D~N>DfY~VA~ًC(D(DnD^EDYH~AYAYEY~X~Y^EXHv ~D(EXEX~X(YAYAY(EY((DX(AXDXXA(AXڅt4HO8HG(HtHI)uA(A((us4H]    HO8K8C4HGHtH	HUHuH})umeDuD}]UHC4C(   (umeDuHC,D}]UHUHuH}<     UHSHH(F(t9C,.C4HV(H~K0Hv z5u3.K8z-u+HO8HG0HtHI fC(    C,H] HO8HGHtHIHUHuH}HUHuH}ff.     UHSHHH?m HH]%m f     H4     UAfHHPGDW@p9  HyHHk8fD(qDf(fDfHm HD fD(fDD@0D@x Hn     )5m )5m )5m H9e  qPy yXfD(fD(A4fD(DfATfDUfDVf(fTfUfVy`DfATfDUfA(fDVAhDfDTfUfAVfATfDUf(fDVf(AfTfUfVf(fTAfAUfVf(fTfAUfVf(fTfUfVf(fTfUfVffTfUQPf(fVfTfUfVfA`҉Q@Hl     )5^l )5gl )5pl A8 f.     G8  xA9	f     DFLA9rA$IwDDEHyHHk8HMDMme]UME   H   Hi DMHMfHDEEHMU]em   QDHk8HtDDMHqHHMj HMDMfDEEHMU]emADHyHDI@pHi DMHMfHDEEHMU]emuA@D9rADHyHpЉA@Jf.     f(8 t^f.O(    Etf.G Eu;fo w0f(f(ff(@tG0 of(f(f(@     ~8 t0~0 u^f.^ Nzuf.N(zuF0ZV D  UZHf(HH Hu)U1Huf(U~8 tF0V      ~8 t"~0 uFf.F Nzuf.N(zu H     UfDofDofDlfofofEofDlflHATflISfDofDoH   D=; dH%(   H]fD(fD fEfE(fE(fE\fE\fA(fA(fA(fff(fW; f(=: AYYXf(fT; f/%  fA(fA(fE(1ffE\D) f(D)\fD)0DYf(f(D)@D)PAYD)`DXD^f(fEfEYfEXfA(fA(D)fA\fA\fY: fY-: fAXfAXfA\fA\f(ff(ffD(`fDoPfD(@fD(0fD(fD( fD(=(: R  
	  fEYfA(˃LfA\fEY׉D)pD)0fAYfEXfEXD)@fAXfE(D)UU]fD\f(fA\fEYfAYfDXxfAXfE(D)PfD\)EemfEYfDXpD)Eu}D)`%fD(`fD(=39 fD(PfD(@fDo0f  HEdH+%(      (H   L[A\]f.     fD( A|$8 t+fAd$ f.f(f(ff(z*u(f.(zuHEdH+%(   ukH   [A\]ÐfA(At$0fA(ff(@u3HEdH+%(   u2(f(LH   [f(A\]1AD$0 Ad$\d ff.     H~8 tgf    Zf.v EЄtOfZf.v(EЄt8.EЄt+.EЄt.EЄt.Dt fDG Zf(ZfA(fff(fD(Zf(A\A\f(f(YYXff/   6 fD(fE^fE(f(A\E\YDYA\D67 YYfD/r1f(A\Yf(fA\Y\YYfD/s6fA(fEf(1f(f(ffffA(X     f(f( H~8 tkf    fZf.n(ZEtf.f Eu8fw w0f(f(ff(@tG0 wf(f(D      H~8 tkfv ffZZf.f(ffD(f(zu	fA.zt4w0ffZZ@tG0 wfA(f(}D      UHf(HH0F<V8p9   Hy@oA H@q<H HQ0HPHsb Hxb     )=ab H9  ~~~i(A Y~y Y~~X~a0~X(YYi0Xa fD    A9<D  DFD A9rA
   DHMHy@H4@DEHme]UMu   H   H}_ DEHMfHuMHU]em   Q<HkHt?DEHq@HM` HMDEfuMHU]em؋A<Hy@DA8p<҉Q8H`     )=` Ɓe  ^ DEHMfHuMHU]emuA8D9rA<Hy@pЉA8ff.     H=-` Hu    1HH` uUHHH}~^ H}H9t%V] fD   ff.     H=_ Hu    1HH_ uUHHH}^ H}H9t%\ fD   ff.        xzUIAHH 9r>AA9s))t IqMLMH<1\ MLMAI   fDFDA9rA?vA1    1D  DƉMIyLMHDEu^HtYH\ DELMHMHtVAQHt DEIqLM!^ LMDEMHIyE!    \ DELMHMHuAD9F ff.     x|xvUIAHH 9rBAA9s&IqM)LMH<F1k[ MLMAI   f     DFDA9rEyA1 1D  DƉMIyLMHDEu_HtZH[ DELMHMHtWAQHt DEIqLM\ LMDEMHIyE'f     R[ DELMHMHuAD9JfD  UHSHHFxFxwB~`  w9FTH{Ppt&CTPHCXHP   C`H]f.     ƃe  H] UHAUIATASHHPHGp;   CTAu AMPHCXHPAU   xXfA AA   fD`fDfxfpfD@fH
fH1ɅHHfPH[A\A]]D  fpA   fA fH
1fD fD`fxfpfxfA E11r@    fD`f1fxfp@    rD  ƃe  H1[A\A]]Ðff.     UHAVIAUIATSDgA4 A    IFHL%u. M. J`-e. Lfff.     /w((TUVY,ЋNf1@    HHfPVމ)كHH'fHAAA)AJAHAH'fHDD)уHH'fH)эQHH'fPL97[DA\A]A^]ff.     FxAхtPVxvfD  H~`  `  DFTEAO  FPD  UADHH D9rqD9s8HqXDDM)DEH<VDU 1HMV DMDEDUHMHAX   DQTfB@fFL@BD@    A`fff.     DF\E9rEyƁe  APDDMHyXHMHD]DUDE   Ht{HW DEDUHD]HMHDMtxQTHt1D]HqXDUDEHM4X HMDEDUD]HDMHyXDYPQTD  Ɓe  V DEDUHD]HMHDMuAPD9     H  UHHPH]HLeIHOLmILu7Eƅ   L}HH  IM   LEtHL	   W AO,I0c  H1     H9	  L;*uH@H4H  EtvHFLvL.LH^LfHuHEV MHutH}AHuAW(L}H!H]LeLmLu    L}1     LV 1 :V AO,  Iw01H H9  L;*uH@HH  QL`HXLHHRHo HRHPAO,	V MtHAL}   71D  HU8      HMS IH$1H1T HEHMIG(    IG0    L9$AW,  HML=U AG,HMȅl  PHMLHH4RIW0HHrHRAG,HUHu:U HUHMHtHMH}HMD  AG(Q9   HIAW,H4L.H^LfHuLT HuAG(      A   DHkI  H@  HJS HH@  AW,HkHtIw0T HAO,I0Ew(Q_     LWT Tx-A9:DEtA9rA
dAG(f1H5T HT )T AG(vHMI0AG,    S HMHMLIG(    IG0    S HMHMLS LS HMAG(vHMI0S HMIG(    IG0    HQ HHAG(D9"AO,I0Q0f     UHH H]S!H_HuH]    C,LuI   HR C,t]LeLm ff.     PHHRHHS0LlLdC,R MtLAHXR C,uLeLmC(w>HC(    HHC0    eR H<R HkR IF    LuH]ÐC,    H{0GR D  C(v
H{0+R HC(    HC0    뜐U11HATSHH dL$%(   LeI1LP E䋃  @  F  A9rQ  9   H    HUHL1O    HUdH+%(   ;  H [A\] DFDA9rAUUU  DMH  H4RDEH   H   HO DE܋MHH     H@HtH  P DE܋MHH  D  @ H  )RH@M܉H<1N H  MD  x
ǃ      1 N DE܋MHHx  D9Љ  1O ff.     H~HtCUHAUIATASHHN tH{DO AE    H[A\A]]1fff.     V<   JLF@HHIIoF HIHN01ɅI9sT~8N<WvFxBUHHH9r;9sf=
c.HULǉMWO MHU19@     =
1HULH4@MHM MHUHtHB@J8f׉z8ËB89sЉB8ff.     H=5O Hu    1HHO uUHHH}L H}H9t%6N fD   ff.     tH1u8   wfHwh  t!    H  HEfD  HY^       tw9H t    H HEf     HV  HDÐ    H! HE@ Ha  U@     HK HHtO@  1HK P  @0Hf@8@P=  f@`    @H@    H]Ðff.     HHtt    H    /  UHSHHH8  L H(  vǃ,      H0  L   vǃ      H   L   vǃ      H  L    vǃ       H   `L    vǃ       H   ;L    vǃ       H   L    vǃ       H   K    vǃ       H   K    vǃ       H   K    vǃ       H   K    vǃ       H   ]K CxvC|    H   >K C@vCD    H{H"K HH]%K @     UH#] ff.     H      HHtxUHHH}HuZJ H}O,tGLG0Hu1LH9t-H;2uH@IHtHPHUUJ HUH@ BJ 1H 1D  UHSHLJ MtLHH] LEH 11H5CHHJ 11H5HI 11H5HH 11H5HkH 11H5pHI H&I H='gV  Ht0HEHI U[H H97H1G )>H HHEHI HD  wpWtD  HtGpHtGt    UHAUATIHuSHLHdL,%(   LmAHUH UЋuHQH HH HLHDG HUdH+%(   uH[A\A]]G  ff.     %FH fD  G@fG0G4    G8GG x* GD    fGP fG`fЉG@f     UHH   G8H]HLmH
  i  D_4LeE(  _Pf.k z  k f(~ f(fTf.
  5M f(fTf.  cXf(f(fTf.v5H,ffUD H*f(fAT\f(fVf(fTf.u  S`f(f(fTf.v8H,fD(fD fUH*DfETAXfVf(f(fTf.  khfD(f(fTf.v7H,f(fD. fDUH*fATXfAVf(fTf.   \\ f(ffff(f(% ffffTfTfUfUfVfVf(f(fffTfUfTfUfVfVffflDcDEu/    E LeI  fD  fDcDEtHCHHECxM     g     B               ]     8              s    N  (    E  Cx  AD9  C|D9   YSP~ Dc|V f(f(fTf.v3H,f5 fUH*f(fT\fVf(51 f/r{D* fD/rk% YcXf(f(fTf.v5H,ffUD H*f(fAT\f(fVf/rfD/   fD  LeE1HB LH]Lm DFl*E9rA$I5  DH   Hk8
  H
  HA HH
  S|Hk8HtH   H,C HH   Dkxf-  Yk`f(f(fTf.  f/3fD/( Y[hfD(f(fTf.v:H,fD(fDa fDUH*DfETAXfAVf(f/fD/LuL}  , LmDe,fHh,LH,)p@*Yډ4Ɖ<8E]f*YH   1I]f*Y]f*Y]D   AYE TA , AYEɉM6A D,i AYE A ,Q AYEED@ D,0 AYEEDM@ , AYE(E@ MEE,DMD9ADNE9EOD9fAnEDME9ELA9fAnAENfbfofY* A9DOA9EMfAnA9DLA9AA9fAnfbD!D9flAG E9fpI8fY !I8AGAGfo% EAg9qm\mAĸ   A9fuf(pAF\uEf/HhA9  ff(E*^p}ff/  ff(DuA*^}uH   `)Phmh  uH   Qw  DpH   fmhJ    f(P`@     %  ,  ;  d  DH   fIH@   "  2  9  .  t UH   1f= D; DJUDB<fD  H@$    x! uf/m  H@,    H8D9  x  uf/vX\]^]\f(fTf.v6H,fH*fD(DfETA\fD(fDUf(fAVP\U^UXfD(fDTfA.v8H,fEL*fE(DfETE\fD(fDUfA(fAV,E1҅AHD,҉P$E9EODP(A9     H9P(}fD  tLe    H,fD(fD fUH*DfETAXf(fVf/f     D)k81Hk8H   : fD  H,f5? fUH*f(fT\f(fVf@ uH   9tcuH   &tPH   I11H   '@ ff.     ff.     HI9  AsLeLuL}E   A   uf     E   ǅp   mD  TTTT@}1҉փ@1T1T1 T109rW     TTTT@r1҉׃@>T>T> T>09rL     \]^]\f(fTf.v6H,fH*fD(DfETA\fD(fDUf(fAVP\U^UXfD(fDTfA.v8H,fEL*fE(DfETE\fD(fDUfA(fAV,E1҅AHD,҉P,E9EODP0A9 H9P0}H   J<    H   1HP*    ff.     ff.     HH9r  AAEsǃ      Љ  ǃ      Љ  vǃ       Љ   Qǃ       Љ   ,ǃ,      Љ(  MC|    Cxǃ       Љ   ǃ       Љ   ǃ       Љ   ǃ       Љ   ǃ       Љ   gǃ       Љ   B     D    H6 HH*CxD9LeCxH   DƉUDhDE:DEH   DDEUH   ։UuH   uH  uH  L   H  1DhDEAHI9uH   L   1HPf     AHH9uH   DeLMHUHr$HMDfHc;V4L   L   f.     D`D$HEE9V}HcV;V8L   L   fff.     AD`E$HEE9V}EAH8A9oLMAAHMH   HULK49H(HhH   HuD0D`ID IDMfff.     EE    E$MID9   D)HhD9IGJ<   HLLEDUHEH}*H}HEDULEH4HWH9szfff.     H9s_Hk8ATHf     fn Hfbf H9s/HfHnHk8AD1f/E1f/AD)ɃtHH9r   D9   D)H   D9IG   N<HLHELeHEIOI4H9sn@ I9s\Hk8AHf.     fn Hfbf I9s,PfHnHk8A1f/1f/@)tHH9rIIL9mDL(HPLLD0H   LD DHuHhH   L DID(ID0MfD  EE    E$MID9   D)D9IG   HhLDMLEJ<HHEH}
H}HELEDMH4HWH9szfff.     H9s_Hk8ATHf     fn Hfbf H9s/HfHnHk8AD1f/E1f/AD)уtHH9r   D9   D)H   D9IG   N<HLHEL5HEIOI4H9so@ I9s]Hk8ATHf     fn Hfbf I9s-PfHnHk8AD1f/1f/@)tHH9rIIL9mCD`D(L D0D(1ECL- At/At$H}Hk8HGhH7        0H8H9uDЃpAt6u1=   L8  EMtHcI9B(3  E1LU}U/ IHk  ELUUȉE@ DUfEu
H(  fAE <LUfAE8fAE4fAEpfAECp9L  9O fAECt9L  9OfAEULU  L0  DeE1H@HHLhL`LELm,  HE2 YDbL</ , YCfA/ , YCfAW/ , YCfAW/ ,fAWHED YC(L,d/ D, YC L/ Av,;u   shAA	E} @uy\ YCH/ D,D YCP. ,fE}AT$IH8fAMD;u   HEJ{0 rAԉ13fAM fE}E11믉QTTTH    HD        D    VTTTH    HD    H@E1XEYp H   LhH]L`LELmD]HhH   LHLL}H`EH`DE  H}EE1EHhH   L   D<L   DAf     CAAHk8Lua  fD  ff.     G  8AHk8f/wA9ACD9sf(AE9uY'  HUD]DE, D]DEE,HUC<DE)@ ff.     ff.     ff.     ff.     EƃGFAD    fA fET E1fET9uE48A)@ EE1҃GFfETAD    fA fEL D9uA f fEDUfADUfA|UfALUHH9UtE3fD  DpIH]EXEH   ILHuH   Y HEHuHPLpEMHHuEHED E  HEI   1EI   ED8M   DAwfD  B?Hk8LuC   7  0AHk8Tf/w9ǉCD9sf(AD9uY L]DUDE* DUDED,L]C4DD)fD  ff.     DFGAD    fA fEL E1fEL9uE0A)@ A<A<A<AD    f fA| 1fA|D9u΋EfA f HEHEHfED fETfAtfATHEȉ]H9EtE` M8  LpLIǃ8      MtHI9D$(O     s( H  uL(LL»H   pu( LeLuL}Iz YEL]DUDE#) DUDE,L]DE@ YEHUD]DE( D]DE,HUDEIJ AL)E;Es@uHMLUȉƉu' UHMHI   HEAE1MuLeLuL}H8  HtHH9P(tYL( Mt$ MtM9t	L( EL}M.AFEMl$LuMAD$LeLH@ L;(u\LE1:( LeLuL}hUHHĀdH%(   HE1d   6  Hx  H]HƆd   G4HH)   Hu% HH/  DClChAPA9;  H{pDSlH( HHH(     H9   H    M}  Uu  9DEAoL  DN9L\  E)EDEADN9LML9HE    foMLHDE   foEM   UoE      H  H]HEdH+%(   N  ЉChH'     fff.     H% ƃe  H]f.     xA9DEL1A9rAwDHMH{pDMH   H   H$ DMHMHH   ClHHtHsp% DMHMHDClH{pDKhAP*ƃe  H  H]ECH$ DMHMHHuChD9DClH{pAPH]$     UHATISHH  wsHx  HtRl$ Hx  SLsH$ oC HC0ƃd  Hx  H\  L  I<$[A\]%>$ fD  "$ Hx  HHuƃe  1I$    [A\]fff.     UHH   L}dH%(   HE1HFH  h  H]HLeL   LmILuE1fD  AD$(C  {l    H      Hx  H  J#  A.T$I|$z]u[fA.D$zOuMA.D$zEuCA.T$z;u9A.D$ z1u/A.D$$z'u%HA$Hx  # A=  fD  H`H\# Hx  `\" AoL$fID$ EHx  EEEHE)p" fHEHx  EHEEELm# M  MLA4$I|$Hp" EAǅtzE.EH}HuMzu.MzHUtAHO8HGHtHIHUH@H8HHH8H@HHHO8HG0HtHI Hx  G4HH)   E  H` IH  KlChQ9  H{pH5" SlHL8Ho"     H9v  `h  dln  9ADh  DNE9EO9L׋p  Dh  Dl  9L9p  NA9DO9LƋt  Dl  9Lt  AL$(ED$,A|$0AT$4AD$8AM    H  AI@I   9GA9H]LeLmLuA       AL$,h  9Oщh  p  AL$49Lщp  l  AL$09Oщl  t  AL$89Lщt  AE \ЉChH      @ ff.     L ƃe  AE H]LeE1LmLuHEdH+%(   B  DL}f     j Hx  HHD  nA9fff.     DEL	A9rA5DDHH{pH  Hz  H0 DHHHz  ClHHtHspg DHHǋKlH{pDKhQEf^LH# 11H5XIH$ 11H5AL 11H5-L 11H5yL 11H5%L$ L; H=ܾg+  Mt5HHL= EH]LeAE LuLm'N IHHL= TM8& I9)L H DHHHChD9pKlH{pQH]LeLmLuJ f.     UHH@dH%(   HE1e   uNDH  H]HE  ~`    HuHLeALuAu*H]LeLuHEdH+%(     f     CTH{Ppa!  CT}1EM@ƍPHCXAAHPUԅ     Af AEE1fxfDfpfD@fDHfH
fH1ɅLuHDfP)ʹHH'fPD҉)֍VDEHH'fP)HH'fPDD)LeHH'fPC`H] xlf E11*D  LeLuƃe  H]H]    f E1fH
fp1fDfD@fxfp 1fpf1fD@fx@    H]LeLu  ff.     UHH   dH%(   HE1e      LeIH     A|$`    HLL}fA~Hx`dhlp  H]HxLmA$  LuM$   HuLAu*H]LeLmLuL}HEdH+%(   s  Ð1    H  fHx`HM dhfAnlpHUHuH}) HMHUL e]EU\uAl$ \ET$,HxD(DY(YUYYAT$(DXXXE((XA(Y(YAXD$0AYXAd$$DYEXL$4DX= /   /	   .v  ,A/f  D/   sE.t  A,A\fP/  /   s.>  ,\fP/  /   s.  ,((fPW YAY\(T /K     9 ^DYDYA/-  D/   sE.  A,YY/  /   s.  ,W% YY/  /%   s.{  ,Y(Y/  /}   r<Gft W% YYYY.  Y111Y.  ,fxH}fp
LfHH      fPHMHE9DuL}B  EE111L4  IH  AL$lAD$hQ9b  AT$lIT$pHHK L H=     H9   1HUL牍xpxH   fHfDhffXAFAD$`Lh  H]LmLuL}AƄ$e  LefD  LeH DxLpHH+  AD$hD9'  AD$hHd     @ L AƄ$e  AFX"@ LeL}% A/  f @ 1xA9DEL	A9rA^DLpI|$pDxHHH DxLpHHAD$lHHtHIt$p DxLpHIL$pEL$hAL$lQf1f     1f     1@f     1f     Y.            Y1]1*1 x  1  J  H]LeLmLuL}V fD  UHH   E|xdH%(   HE1e      LeIH     A|$`     HLHUlptE   H]H}LuA$  L}M$   HuLAxudH]LeLuL}HEdH+%(     H]LmLuL}@ ff.     AƄ$e  Le뽐Lef.     Lm1(   ' HtfHEHMHU LH@    D u|ptUE\\HEDxlEd$,(D(A\YDYYYYXYAD$(DXXAt$ X(D(YE(E(DYAXAXEXL$0EYDXAL$$DYEX\$4EXDi E/  D/S   E.  A,E/f  D/)   sE.  A,A\fPD/  /   s.5  ,\fPD/  /%   s.  ,D/fPj  /   s.  ,D/fPK  /-z   s.  ,((fP
W Y@    AY\(T /8 6  f* 11Y1(Y.   ,fxH}fpLfHH      fPHMHEDmLu  EE111L4  IH  AL$lAD$hQ9  AT$lIT$pHH L H     H9w  HU   LMCMHi  fHfDxffXAEAD$`O  H]LeLmLuL}"fD   ^ DYDYE/  D/%   sE.6  A,YYD/H  /   s.  ,W6 YYD/
  /a   s.  ,YYD/  /54   $H DMLEHH  AD$hD9  AD$hH      L' AƄ$e  AEL fE/ӹ  fq @ xA9DFL	A9rAoDLEI|$pDMH(HH DMLEHHAD$lHHtHIt$p. DMLEHIL$pEL$hAL$lQ_@ 1f     1f     1f     1f     17f     1zf     1111        =  `H]LeLmLuL} UHHP  dH%(   HE1e   m  H]HH    {`    LmI111L	   LeLu"  E1L 1LLH}	 L<
 HH߉     1	 IH~  fL}H HL	 \{(s$(DC,YYYYXXc YXK0XC4XAYX-/ /  / Q  .8  ,/fAU I  /   s./  ,D  fAUAY/(  /   sY ,AYfAU/R  /   sY ,((W=  fAUYAY\((T /F T   4 ^DYDYA/  D/   sE.  A,YY/  /   s.f  ,W5{ (YY/h  /   s.   ,Y(Y/k  /u         ƃe  L H]LeLmLuHEdH+%(   *   H<@H IH  I     f W5 Y(YYY.  Y111Y(.  ,fA}HH      fAu
LfAMfAUHL蒬L  DL_ E111C4$L[ IH   DclChAT$A9z  SlHSpDHH L(H     H9   H   H貪H   DfD`ffHfPC`wtH]LeLmLuL}H
 HHS  Ch9O  ЉChH     D  Lg ƃe  vfL/ ~fLeLmLuƃe  H]@ H]    /  fAM   H]LmS ƃe  L k @9tfD  L19rʉH{pHHH HHClHHtHsp HH{pKhDclAT$     1f     1&f     1f     Y.ɿ            Y(      1f     1f     1[ ;   _ }  pH]LeLmLuL} f     HGp    %F  ff.     UHATISHH8   Hǃ8      Mt I9tL8  [A\]ff.     tH1u8t,wZH  t%    H  HEf.     H           H HGf     HF  HDÐU     H  HH     1H   Hǀ      HH   @HX  @(g H@    H@@H9uL  1foQ ǁH      h  fHǁ\      fd  Hǁx      Hǁ           AHA    H]@ HHtt    H#    /  UHAUATISHH I\$pAD$lL,L9tH;H  I9uI$x  x  I$    LJA$  vAǄ$      I$  h AD$hvAD$l    I|$pG AD$PvAD$T    I|$X& AD$8vAD$<    I|$@ HL[A\A]]%  f.     @ ff.     UH] ff.     H      HHtxUHHH}Hu*  H}O,tGLG0Hu1LH9t-H;2uH@IHtHPHU%  HUH@   1H 1D  UHSHL\  MtLHH] LE 11H5HH 11H5H 11H5(H_ 11H5DH 11H5`Ho 11H5H 11H5H/ 11H5H 11H5`Ho 11H5H 1H1H5% 11H5H 11H5H 11H5<H HJ H=ӯg  Ht0HEH  H9H u HHEH hHP뼐w     G     HHt% D      UHSHHHHt H{H]%r fUu4 uUHHCHuH]1f.     wHWLD  HtGHHtGL    UHAVAUATIHuSHLHdL,%(   LmAHUV UЋuH DsH A   HLHED HUdH+%(   uH[A\A]A^]] fff.     UHAVAUATIHuSHLHdL,%(   LmAHU UЋuH DsHb A   HLHED HEdH+%(   uH[A\A]A^] fff.     UHAUATISHGP   I\$pAD$lAD$T    AD$`    L,I9tff.     H;H I9uAD$h   1(x AD$l    fA$d  AD$8AǄ$H      IǄ$      ID$0    AD$ xFfo> AD$<    A$h  H[A\A]]ÐЉGP5fD  AD$hs@ AD$8    UHH   H]HL}HudH%(   HE1e   R  W`  GTLeLgpLuEGlLmM4M9  E1fI<$I2 AM9uA  CTH  ADEEHMHtH~H9A(  HE    }I IH  DuKlm  E   HE      C`DCTIAG    fA1fAGh  9Lº  9OºfAGl  9Lº  9OºfAG
p  9Lº  9OºfAGt  IG   9Lº  9ODfAGtMHsXn DCTMHE   1Gf  f  VE   WAf AAfDMLMEUHUt7D0D0LGBWE#D	9sLuAAAfF$WfAHL9uD9ZC I|LspClM,M9tWHxL}I@ IH 1HA MtLHLZ IMM9uL}HxHEHtO~h  ~p  fpf~f~f p  )  HMAl  )  HMAL  Hǃ      MtHՉI9F(     X Ht\ML8LLu   HH I?ffw    H  HtHeH9P(tL^ E1}uMLeLmLuH( HEdH+%(   	  LH]L}      ID  H} LeLmLu     H@ L;8txHEH@ L;8  fff.     LeLmLuE1[@ Lq AFM>E;E   M  EuLI IH   IEnHE    EE1HEMn HtI9t	H" EM} AEEM~MAFMKlHE     H  HEHo  HMHH9A(L LeLmLuIHEE1fD  DFD A9rA?wD1DExH HEHHgxDmDEAtDE1M DEMLuDEE1HELxMIA]Af     DLE9seA;vlsEIFpJ<? IIL9euDELxDLA@Ea    H     1f          H 눋ELE1Q LeLmLu   0   LeLmLuz f.     UHSHHHGH    G    H HC    HH]% UHATISHH   Hǃ      Mtd I9tL  [A\]ff.     tH1u8t,wZH  t%    HOZ HEf.     H$          H HGf     H HDÐH 1%-  HH                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                               /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Requires Metal Shading Language 2.0. */


/* Dilate a glyph vertex by half a pixel on screen.
 *
 * position:  object-space vertex position (modified in place)
 * texcoord:  em-space sample coordinates (modified in place)
 * normal:    object-space outward normal at this vertex
 * jac:       inverse of the 2x2 linear part of the em-to-object transform,
 *            stored row-major as (j00, j01, j10, j11).  Maps object-space
 *            displacements back to em-space for texcoord adjustment.
 *            For simple scaling with y-flip (the common case):
 *              em-to-object = [[s, 0], [0, -s]]
 *              jac = (1/s, 0, 0, -1/s)
 * m:         model-view-projection matrix
 * viewport:  viewport size in pixels
 */
void hb_gpu_dilate (thread float2& position, thread float2& texcoord,
		    float2 normal, float4 jac,
		    float4x4 m, float2 viewport)
{
  float2 n = normalize (normal);

  float4 clipPos = m * float4 (position, 0.0, 1.0);
  float4 clipN   = m * float4 (n, 0.0, 0.0);

  float s = clipPos.w;
  float t = clipN.w;

  float u = (s * clipN.x - t * clipPos.x) * viewport.x;
  float v = (s * clipN.y - t * clipPos.y) * viewport.y;

  float s2 = s * s;
  float st = s * t;
  float uv = u * u + v * v;

  float denom = uv - st * st;
  float d = abs (denom) > 1.0 / 16777216.0
	  ? s2 * (st + sqrt (uv)) / denom
	  : 0.0;

  float2 dPos = d * normal;
  position += dPos;
  texcoord += float2 (dot (dPos, jac.xy), dot (dPos, jac.zw));
}
      /*
 * Copyright (C) 2026  Behdad Esfahbod
 * Copyright (C) 2017  Eric Lengyel
 *
 * The _hb_gpu_slug body below is based on the Slug algorithm by
 * Eric Lengyel: https://github.com/EricLengyel/Slug
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Shared fragment-shader helpers for the hb-gpu renderers.
 *
 * Requires GLSL 3.30 or GLSL ES 3.00.
 *
 * For GLSL ES 3.00 / WebGL2, define HB_GPU_ATLAS_2D before
 * including this source.  The atlas is then a 2D isampler2D
 * texture of known width (set via hb_gpu_atlas_width uniform)
 * instead of an isamplerBuffer.
 *
 * Exports:
 *   hb_gpu_fetch (offset)                   atlas lookup
 *   _hb_gpu_slug (rc, pixelsPerEm, glyphLoc) MSAA-aware Slug
 *                                            coverage (caller supplies
 *                                            pixelsPerEm so it can be
 *                                            invoked from non-uniform
 *                                            control flow)
 *   hb_gpu_ppem (rc, glyphLoc)              pixels per em at fragment
 *   _hb_gpu_curve_counts (rc, glyphLoc)     debug: per-pixel curve counts
 *   hb_gpu_stem_darken (cov, brightness, ppem) thin-stroke contrast
 *                                            correction
 */


#ifndef HB_GPU_UNITS_PER_EM
#define HB_GPU_UNITS_PER_EM 4
#endif

#define HB_GPU_INV_UNITS float(1.0 / float(HB_GPU_UNITS_PER_EM))


#ifdef HB_GPU_ATLAS_2D
uniform highp isampler2D hb_gpu_atlas;
uniform int hb_gpu_atlas_width;
ivec4 hb_gpu_fetch (int offset)
{
  return texelFetch (hb_gpu_atlas,
		     ivec2 (offset % hb_gpu_atlas_width,
			    offset / hb_gpu_atlas_width), 0);
}
#else
uniform isamplerBuffer hb_gpu_atlas;
ivec4 hb_gpu_fetch (int offset)
{
  return texelFetch (hb_gpu_atlas, offset);
}
#endif


uint _hb_gpu_calc_root_code (float y1, float y2, float y3)
{
  uint i1 = floatBitsToUint (y1) >> 31U;
  uint i2 = floatBitsToUint (y2) >> 30U;
  uint i3 = floatBitsToUint (y3) >> 29U;

  uint shift = (i2 & 2U) | (i1 & ~2U);
  shift = (i3 & 4U) | (shift & ~4U);

  return (0x2E74U >> shift) & 0x0101U;
}

vec2 _hb_gpu_solve_horiz_poly (vec2 a, vec2 b, vec2 p1)
{
  float ra = 1.0 / a.y;
  float rb = 0.5 / b.y;

  float d = sqrt (max (b.y * b.y - a.y * p1.y, 0.0));
  float t1 = (b.y - d) * ra;
  float t2 = (b.y + d) * ra;

  if (a.y == 0.0)
    t1 = t2 = p1.y * rb;

  return vec2 ((a.x * t1 - b.x * 2.0) * t1 + p1.x,
	       (a.x * t2 - b.x * 2.0) * t2 + p1.x);
}

vec2 _hb_gpu_solve_vert_poly (vec2 a, vec2 b, vec2 p1)
{
  float ra = 1.0 / a.x;
  float rb = 0.5 / b.x;

  float d = sqrt (max (b.x * b.x - a.x * p1.x, 0.0));
  float t1 = (b.x - d) * ra;
  float t2 = (b.x + d) * ra;

  if (a.x == 0.0)
    t1 = t2 = p1.x * rb;

  return vec2 ((a.y * t1 - b.y * 2.0) * t1 + p1.y,
	       (a.y * t2 - b.y * 2.0) * t2 + p1.y);
}

float _hb_gpu_calc_coverage (float xcov, float ycov, float xwgt, float ywgt)
{
  float coverage = max (abs (xcov * xwgt + ycov * ywgt) /
			max (xwgt + ywgt, 1.0 / 65536.0),
			min (abs (xcov), abs (ycov)));

  return clamp (coverage, 0.0, 1.0);
}

/* Decoded glyph band info for a pixel position. */
struct _hb_gpu_glyph_info
{
  int glyphLoc;
  int bandBase;
  ivec2 bandIndex;
  int numHBands;
  int numVBands;
  vec2 scale;
};

_hb_gpu_glyph_info _hb_gpu_decode_glyph (vec2 renderCoord, uint glyphLoc_)
{
  _hb_gpu_glyph_info gi;
  gi.glyphLoc = int (glyphLoc_);

  ivec4 header0 = hb_gpu_fetch (gi.glyphLoc);
  ivec4 header1 = hb_gpu_fetch (gi.glyphLoc + 1);
  vec4 ext = vec4 (header0) * HB_GPU_INV_UNITS;
  gi.numHBands = header1.r;
  gi.numVBands = header1.g;
  gi.scale = vec2 (float (header1.b), float (header1.a));

  vec2 extSize = ext.zw - ext.xy;
  vec2 bandScale = vec2 (float (gi.numVBands), float (gi.numHBands)) / max (extSize, vec2 (1.0 / 65536.0));
  vec2 bandOffset = -ext.xy * bandScale;

  gi.bandIndex = clamp (ivec2 (renderCoord * bandScale + bandOffset),
			ivec2 (0, 0),
			ivec2 (gi.numVBands - 1, gi.numHBands - 1));

  gi.bandBase = gi.glyphLoc + 2;
  return gi;
}

/* Return pixels per em at this fragment. */
float hb_gpu_ppem (vec2 renderCoord, uint glyphLoc_)
{
  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_);
  vec2 emsPerPixel = fwidth (renderCoord);
  return min (gi.scale.x, gi.scale.y) /
	 max (emsPerPixel.x, emsPerPixel.y);
}

/* Return per-pixel curve counts: (horizontal, vertical). */
ivec2 _hb_gpu_curve_counts (vec2 renderCoord, uint glyphLoc_)
{
  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_);
  int hCount = hb_gpu_fetch (gi.bandBase + gi.bandIndex.y).r;
  int vCount = hb_gpu_fetch (gi.bandBase + gi.numHBands + gi.bandIndex.x).r;
  return ivec2 (hCount, vCount);
}

/* Single-sample coverage in [0, 1]. */
float _hb_gpu_slug_single (vec2 renderCoord, vec2 pixelsPerEm, uint glyphLoc_)
{

  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_);
  int glyphLoc = gi.glyphLoc;
  int bandBase = gi.bandBase;
  int numHBands = gi.numHBands;

  float xcov = 0.0;
  float xwgt = 0.0;

  ivec4 hbandData = hb_gpu_fetch (bandBase + gi.bandIndex.y);
  int hCurveCount = hbandData.r;
  /* Symmetric: choose rightward (desc) or leftward (asc) sort */
  float hSplit = float (hbandData.a) * HB_GPU_INV_UNITS;
  bool hLeftRay = (renderCoord.x < hSplit);
  int hDataOffset = (hLeftRay ? hbandData.b : hbandData.g) + 32768;

  for (int ci = 0; ci < hCurveCount; ci++)
  {
    int curveOffset = hb_gpu_fetch (glyphLoc + hDataOffset + ci).r + 32768;

    ivec4 raw12 = hb_gpu_fetch (glyphLoc + curveOffset);
    ivec4 raw3 = hb_gpu_fetch (glyphLoc + curveOffset + 1);

    vec4 q12 = vec4 (raw12) * HB_GPU_INV_UNITS;
    vec2 q3 = vec2 (raw3.rg) * HB_GPU_INV_UNITS;

    vec4 p12 = q12 - vec4 (renderCoord, renderCoord);
    vec2 p3 = q3 - renderCoord;

    if (hLeftRay) {
      if (min (min (p12.x, p12.z), p3.x) * pixelsPerEm.x > 0.5) break;
    } else {
      if (max (max (p12.x, p12.z), p3.x) * pixelsPerEm.x < -0.5) break;
    }

    uint code = _hb_gpu_calc_root_code (p12.y, p12.w, p3.y);
    if (code != 0U)
    {
      vec2 a = q12.xy - q12.zw * 2.0 + q3;
      vec2 b = q12.xy - q12.zw;
      vec2 r = _hb_gpu_solve_horiz_poly (a, b, p12.xy) * pixelsPerEm.x;
      /* For leftward ray: saturate(0.5 - r) counts coverage from the left */
      vec2 cov = hLeftRay ? clamp (vec2 (0.5) - r, 0.0, 1.0)
			  : clamp (r + vec2 (0.5), 0.0, 1.0);

      if ((code & 1U) != 0U)
      {
	xcov += cov.x;
	xwgt = max (xwgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1U)
      {
	xcov -= cov.y;
	xwgt = max (xwgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Crossings over a closed contour sum to zero, so a leftward ray
   * returns the negative of what a rightward ray would; flip it back
   * so that xcov and ycov keep a common sign convention. */
  if (hLeftRay)
    xcov = -xcov;

  float ycov = 0.0;
  float ywgt = 0.0;

  ivec4 vbandData = hb_gpu_fetch (bandBase + numHBands + gi.bandIndex.x);
  int vCurveCount = vbandData.r;
  float vSplit = float (vbandData.a) * HB_GPU_INV_UNITS;
  bool vLeftRay = (renderCoord.y < vSplit);
  int vDataOffset = (vLeftRay ? vbandData.b : vbandData.g) + 32768;

  for (int ci = 0; ci < vCurveCount; ci++)
  {
    int curveOffset = hb_gpu_fetch (glyphLoc + vDataOffset + ci).r + 32768;

    ivec4 raw12 = hb_gpu_fetch (glyphLoc + curveOffset);
    ivec4 raw3 = hb_gpu_fetch (glyphLoc + curveOffset + 1);

    vec4 q12 = vec4 (raw12) * HB_GPU_INV_UNITS;
    vec2 q3 = vec2 (raw3.rg) * HB_GPU_INV_UNITS;

    vec4 p12 = q12 - vec4 (renderCoord, renderCoord);
    vec2 p3 = q3 - renderCoord;

    if (vLeftRay) {
      if (min (min (p12.y, p12.w), p3.y) * pixelsPerEm.y > 0.5) break;
    } else {
      if (max (max (p12.y, p12.w), p3.y) * pixelsPerEm.y < -0.5) break;
    }

    uint code = _hb_gpu_calc_root_code (p12.x, p12.z, p3.x);
    if (code != 0U)
    {
      vec2 a = q12.xy - q12.zw * 2.0 + q3;
      vec2 b = q12.xy - q12.zw;
      vec2 r = _hb_gpu_solve_vert_poly (a, b, p12.xy) * pixelsPerEm.y;
      vec2 cov = vLeftRay ? clamp (vec2 (0.5) - r, 0.0, 1.0)
			  : clamp (r + vec2 (0.5), 0.0, 1.0);

      if ((code & 1U) != 0U)
      {
	ycov -= cov.x;
	ywgt = max (ywgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1U)
      {
	ycov += cov.y;
	ywgt = max (ywgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Ditto, for the vertical ray. */
  if (vLeftRay)
    ycov = -ycov;

  return _hb_gpu_calc_coverage (xcov, ycov, xwgt, ywgt);
}

/* MSAA-aware Slug coverage.  Caller supplies pixelsPerEm so this
 * function can be invoked from non-uniform control flow (for
 * example inside an op-stream branch in hb-gpu-paint-fragment.wgsl,
 * where WGSL would otherwise reject an fwidth call). */
float _hb_gpu_slug (vec2 renderCoord, vec2 pixelsPerEm, uint glyphLoc_)
{
  float c = _hb_gpu_slug_single (renderCoord, pixelsPerEm, glyphLoc_);

#ifndef HB_GPU_NO_MSAA
  float ppem = hb_gpu_ppem (renderCoord, glyphLoc_);

  if (ppem < 16.0)
  {
    vec2 emsPerPixel = 1.0 / pixelsPerEm;
    vec2 d = emsPerPixel * (1.0 / 3.0);
    float msaa = 0.25 *
      (_hb_gpu_slug_single (renderCoord + vec2 (-d.x, -d.y), pixelsPerEm, glyphLoc_) +
       _hb_gpu_slug_single (renderCoord + vec2 ( d.x, -d.y), pixelsPerEm, glyphLoc_) +
       _hb_gpu_slug_single (renderCoord + vec2 (-d.x,  d.y), pixelsPerEm, glyphLoc_) +
       _hb_gpu_slug_single (renderCoord + vec2 ( d.x,  d.y), pixelsPerEm, glyphLoc_));

    c = mix (c, msaa, smoothstep (16.0, 8.0, ppem));
  }
#endif

  return c;
}

/* Stem darkening for small sizes.
 *
 * coverage:    output of hb_gpu_draw / _hb_gpu_slug
 * brightness:  foreground brightness in [0, 1]
 * ppem:        pixels per em at this fragment
 */
float hb_gpu_stem_darken (float coverage, float brightness, float ppem)
{
  return pow (coverage,
	      mix (pow (2.0, brightness - 0.5), 1.0,
		   smoothstep (8.0, 48.0, ppem)));
}
    /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Shared fragment-shader helpers for the hb-gpu renderers.
 *
 * Requires Metal Shading Language 2.0.
 */


#ifndef HB_GPU_UNITS_PER_EM
#define HB_GPU_UNITS_PER_EM 4
#endif

#define HB_GPU_INV_UNITS float(1.0 / float(HB_GPU_UNITS_PER_EM))


int4 hb_gpu_fetch (device const short4* hb_gpu_atlas, int offset)
{
  return int4 (hb_gpu_atlas[offset]);
}

uint _hb_gpu_calc_root_code (float y1, float y2, float y3)
{
  uint i1 = as_type<uint> (y1) >> 31U;
  uint i2 = as_type<uint> (y2) >> 30U;
  uint i3 = as_type<uint> (y3) >> 29U;

  uint shift = (i2 & 2U) | (i1 & ~2U);
  shift = (i3 & 4U) | (shift & ~4U);

  return (0x2E74U >> shift) & 0x0101U;
}

float2 _hb_gpu_solve_horiz_poly (float2 a, float2 b, float2 p1)
{
  float ra = 1.0 / a.y;
  float rb = 0.5 / b.y;

  float d = sqrt (max (b.y * b.y - a.y * p1.y, 0.0));
  float t1 = (b.y - d) * ra;
  float t2 = (b.y + d) * ra;

  if (a.y == 0.0)
    t1 = t2 = p1.y * rb;

  return float2 ((a.x * t1 - b.x * 2.0) * t1 + p1.x,
		 (a.x * t2 - b.x * 2.0) * t2 + p1.x);
}

float2 _hb_gpu_solve_vert_poly (float2 a, float2 b, float2 p1)
{
  float ra = 1.0 / a.x;
  float rb = 0.5 / b.x;

  float d = sqrt (max (b.x * b.x - a.x * p1.x, 0.0));
  float t1 = (b.x - d) * ra;
  float t2 = (b.x + d) * ra;

  if (a.x == 0.0)
    t1 = t2 = p1.x * rb;

  return float2 ((a.y * t1 - b.y * 2.0) * t1 + p1.y,
		 (a.y * t2 - b.y * 2.0) * t2 + p1.y);
}

float _hb_gpu_calc_coverage (float xcov, float ycov, float xwgt, float ywgt)
{
  float coverage = max (abs (xcov * xwgt + ycov * ywgt) /
			max (xwgt + ywgt, 1.0 / 65536.0),
			min (abs (xcov), abs (ycov)));

  return clamp (coverage, 0.0, 1.0);
}

/* Decoded glyph band info for a pixel position. */
struct _hb_gpu_glyph_info
{
  int glyphLoc;
  int bandBase;
  int2 bandIndex;
  int numHBands;
  int numVBands;
  float2 scale;
};

_hb_gpu_glyph_info _hb_gpu_decode_glyph (float2 renderCoord, uint glyphLoc_,
					  device const short4* hb_gpu_atlas)
{
  _hb_gpu_glyph_info gi;
  gi.glyphLoc = int (glyphLoc_);

  int4 header0 = hb_gpu_fetch (hb_gpu_atlas, gi.glyphLoc);
  int4 header1 = hb_gpu_fetch (hb_gpu_atlas, gi.glyphLoc + 1);
  float4 ext = float4 (header0) * HB_GPU_INV_UNITS;
  gi.numHBands = header1.r;
  gi.numVBands = header1.g;
  gi.scale = float2 (float (header1.b), float (header1.a));

  float2 extSize = ext.zw - ext.xy;
  float2 bandScale = float2 (float (gi.numVBands), float (gi.numHBands)) / max (extSize, float2 (1.0 / 65536.0));
  float2 bandOffset = -ext.xy * bandScale;

  gi.bandIndex = clamp (int2 (renderCoord * bandScale + bandOffset),
			int2 (0, 0),
			int2 (gi.numVBands - 1, gi.numHBands - 1));

  gi.bandBase = gi.glyphLoc + 2;
  return gi;
}

/* Return pixels per em at this fragment.
 *
 * renderCoord:  em-space sample position
 * glyphLoc:     texel offset of glyph blob in atlas
 */
float hb_gpu_ppem (float2 renderCoord, uint glyphLoc_,
		   device const short4* hb_gpu_atlas)
{
  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  float2 emsPerPixel = fwidth (renderCoord);
  return min (gi.scale.x, gi.scale.y) /
	 max (emsPerPixel.x, emsPerPixel.y);
}

/* Return per-pixel curve counts: (horizontal, vertical). */
int2 _hb_gpu_curve_counts (float2 renderCoord, uint glyphLoc_,
			   device const short4* hb_gpu_atlas)
{
  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  int hCount = hb_gpu_fetch (hb_gpu_atlas, gi.bandBase + gi.bandIndex.y).r;
  int vCount = hb_gpu_fetch (hb_gpu_atlas, gi.bandBase + gi.numHBands + gi.bandIndex.x).r;
  return int2 (hCount, vCount);
}

/* Single-sample coverage in [0, 1]. */
float _hb_gpu_slug_single (float2 renderCoord, float2 pixelsPerEm, uint glyphLoc_,
			     device const short4* hb_gpu_atlas)
{

  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  int glyphLoc = gi.glyphLoc;
  int bandBase = gi.bandBase;
  int numHBands = gi.numHBands;

  float xcov = 0.0;
  float xwgt = 0.0;

  int4 hbandData = hb_gpu_fetch (hb_gpu_atlas, bandBase + gi.bandIndex.y);
  int hCurveCount = hbandData.r;
  /* Symmetric: choose rightward (desc) or leftward (asc) sort */
  float hSplit = float (hbandData.a) * HB_GPU_INV_UNITS;
  bool hLeftRay = (renderCoord.x < hSplit);
  int hDataOffset = (hLeftRay ? hbandData.b : hbandData.g) + 32768;

  for (int ci = 0; ci < hCurveCount; ci++)
  {
    int curveOffset = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + hDataOffset + ci).r + 32768;

    int4 raw12 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset);
    int4 raw3 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset + 1);

    float4 q12 = float4 (raw12) * HB_GPU_INV_UNITS;
    float2 q3 = float2 (raw3.rg) * HB_GPU_INV_UNITS;

    float4 p12 = q12 - float4 (renderCoord, renderCoord);
    float2 p3 = q3 - renderCoord;

    if (hLeftRay) {
      if (min (min (p12.x, p12.z), p3.x) * pixelsPerEm.x > 0.5) break;
    } else {
      if (max (max (p12.x, p12.z), p3.x) * pixelsPerEm.x < -0.5) break;
    }

    uint code = _hb_gpu_calc_root_code (p12.y, p12.w, p3.y);
    if (code != 0U)
    {
      float2 a = q12.xy - q12.zw * 2.0 + q3;
      float2 b = q12.xy - q12.zw;
      float2 r = _hb_gpu_solve_horiz_poly (a, b, p12.xy) * pixelsPerEm.x;
      /* For leftward ray: saturate(0.5 - r) counts coverage from the left */
      float2 cov = hLeftRay ? clamp (float2 (0.5) - r, 0.0, 1.0)
			    : clamp (r + float2 (0.5), 0.0, 1.0);

      if ((code & 1U) != 0U)
      {
	xcov += cov.x;
	xwgt = max (xwgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1U)
      {
	xcov -= cov.y;
	xwgt = max (xwgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Crossings over a closed contour sum to zero, so a leftward ray
   * returns the negative of what a rightward ray would; flip it back
   * so that xcov and ycov keep a common sign convention. */
  if (hLeftRay)
    xcov = -xcov;

  float ycov = 0.0;
  float ywgt = 0.0;

  int4 vbandData = hb_gpu_fetch (hb_gpu_atlas, bandBase + numHBands + gi.bandIndex.x);
  int vCurveCount = vbandData.r;
  float vSplit = float (vbandData.a) * HB_GPU_INV_UNITS;
  bool vLeftRay = (renderCoord.y < vSplit);
  int vDataOffset = (vLeftRay ? vbandData.b : vbandData.g) + 32768;

  for (int ci = 0; ci < vCurveCount; ci++)
  {
    int curveOffset = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + vDataOffset + ci).r + 32768;

    int4 raw12 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset);
    int4 raw3 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset + 1);

    float4 q12 = float4 (raw12) * HB_GPU_INV_UNITS;
    float2 q3 = float2 (raw3.rg) * HB_GPU_INV_UNITS;

    float4 p12 = q12 - float4 (renderCoord, renderCoord);
    float2 p3 = q3 - renderCoord;

    if (vLeftRay) {
      if (min (min (p12.y, p12.w), p3.y) * pixelsPerEm.y > 0.5) break;
    } else {
      if (max (max (p12.y, p12.w), p3.y) * pixelsPerEm.y < -0.5) break;
    }

    uint code = _hb_gpu_calc_root_code (p12.x, p12.z, p3.x);
    if (code != 0U)
    {
      float2 a = q12.xy - q12.zw * 2.0 + q3;
      float2 b = q12.xy - q12.zw;
      float2 r = _hb_gpu_solve_vert_poly (a, b, p12.xy) * pixelsPerEm.y;
      float2 cov = vLeftRay ? clamp (float2 (0.5) - r, 0.0, 1.0)
			    : clamp (r + float2 (0.5), 0.0, 1.0);

      if ((code & 1U) != 0U)
      {
	ycov -= cov.x;
	ywgt = max (ywgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1U)
      {
	ycov += cov.y;
	ywgt = max (ywgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Ditto, for the vertical ray. */
  if (vLeftRay)
    ycov = -ycov;

  return _hb_gpu_calc_coverage (xcov, ycov, xwgt, ywgt);
}

/* Return coverage in [0, 1].
 *
 * renderCoord:    em-space sample position
 * glyphLoc:       texel offset of glyph blob in atlas
 * hb_gpu_atlas:   device pointer to the atlas buffer
 */
/* The MSAA-aware implementation.  Caller supplies pixelsPerEm so
 * this function can be invoked from non-uniform control flow (for
 * example from a paint op-stream branch where a recomputed fwidth
 * would be rejected by strict derivative-uniformity rules). */
float _hb_gpu_slug (float2 renderCoord, float2 pixelsPerEm, uint glyphLoc_,
			 device const short4* hb_gpu_atlas)
{
  float c = _hb_gpu_slug_single (renderCoord, pixelsPerEm, glyphLoc_, hb_gpu_atlas);

#ifndef HB_GPU_NO_MSAA
  float ppem = hb_gpu_ppem (renderCoord, glyphLoc_, hb_gpu_atlas);

  if (ppem < 16.0)
  {
    float2 emsPerPixel = 1.0 / pixelsPerEm;
    float2 d = emsPerPixel * (1.0 / 3.0);
    float msaa = 0.25 *
      (_hb_gpu_slug_single (renderCoord + float2 (-d.x, -d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas) +
       _hb_gpu_slug_single (renderCoord + float2 ( d.x, -d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas) +
       _hb_gpu_slug_single (renderCoord + float2 (-d.x,  d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas) +
       _hb_gpu_slug_single (renderCoord + float2 ( d.x,  d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas));

    c = mix (c, msaa, smoothstep (16.0, 8.0, ppem));
  }
#endif

  return c;
}

/* Stem darkening for small sizes.
 *
 * coverage:    output of hb_gpu_draw
 * brightness:  foreground brightness in [0, 1]
 * ppem:        pixels per em at this fragment
 */
float hb_gpu_stem_darken (float coverage, float brightness, float ppem)
{
  return pow (coverage,
	      mix (pow (2.0, brightness - 0.5), 1.0,
		   smoothstep (8.0, 48.0, ppem)));
}
  /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Shared fragment-shader helpers for the hb-gpu renderers.
 *
 * Requires WGSL (WebGPU Shading Language).
 */


const HB_GPU_UNITS_PER_EM: f32 = 4.0;
const HB_GPU_INV_UNITS: f32 = 1.0 / 4.0;


fn hb_gpu_fetch (hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>,
                  offset: i32) -> vec4<i32>
{
  return (*hb_gpu_atlas)[offset];
}
fn _hb_gpu_calc_root_code (y1: f32, y2: f32, y3: f32) -> u32
{
  let i1 = bitcast<u32> (y1) >> 31u;
  let i2 = bitcast<u32> (y2) >> 30u;
  let i3 = bitcast<u32> (y3) >> 29u;

  var shift = (i2 & 2u) | (i1 & ~2u);
  shift = (i3 & 4u) | (shift & ~4u);

  return (0x2E74u >> shift) & 0x0101u;
}

fn _hb_gpu_solve_horiz_poly (a: vec2f, b: vec2f, p1: vec2f) -> vec2f
{
  let ra = 1.0 / a.y;
  let rb = 0.5 / b.y;

  let d = sqrt (max (b.y * b.y - a.y * p1.y, 0.0));
  var t1 = (b.y - d) * ra;
  var t2 = (b.y + d) * ra;

  if (a.y == 0.0) {
    t1 = p1.y * rb;
    t2 = t1;
  }

  return vec2f ((a.x * t1 - b.x * 2.0) * t1 + p1.x,
                (a.x * t2 - b.x * 2.0) * t2 + p1.x);
}

fn _hb_gpu_solve_vert_poly (a: vec2f, b: vec2f, p1: vec2f) -> vec2f
{
  let ra = 1.0 / a.x;
  let rb = 0.5 / b.x;

  let d = sqrt (max (b.x * b.x - a.x * p1.x, 0.0));
  var t1 = (b.x - d) * ra;
  var t2 = (b.x + d) * ra;

  if (a.x == 0.0) {
    t1 = p1.x * rb;
    t2 = t1;
  }

  return vec2f ((a.y * t1 - b.y * 2.0) * t1 + p1.y,
                (a.y * t2 - b.y * 2.0) * t2 + p1.y);
}

fn _hb_gpu_calc_coverage (xcov: f32, ycov: f32, xwgt: f32, ywgt: f32) -> f32
{
  let coverage = max (abs (xcov * xwgt + ycov * ywgt) /
                      max (xwgt + ywgt, 1.0 / 65536.0),
                      min (abs (xcov), abs (ycov)));

  return clamp (coverage, 0.0, 1.0);
}

/* Decoded glyph band info for a pixel position. */
struct _hb_gpu_glyph_info
{
  glyphLoc: i32,
  bandBase: i32,
  bandIndex: vec2<i32>,
  numHBands: i32,
  numVBands: i32,
  scale: vec2f,
}

fn _hb_gpu_decode_glyph (renderCoord: vec2f, glyphLoc_: u32,
                          hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> _hb_gpu_glyph_info
{
  var gi: _hb_gpu_glyph_info;
  gi.glyphLoc = i32 (glyphLoc_);

  let header0 = hb_gpu_fetch (hb_gpu_atlas, gi.glyphLoc);
  let header1 = hb_gpu_fetch (hb_gpu_atlas, gi.glyphLoc + 1);
  let ext = vec4f (header0) * HB_GPU_INV_UNITS;
  gi.numHBands = header1.r;
  gi.numVBands = header1.g;
  gi.scale = vec2f (f32 (header1.b), f32 (header1.a));

  let extSize = ext.zw - ext.xy;
  let bandScale = vec2f (f32 (gi.numVBands), f32 (gi.numHBands)) / max (extSize, vec2f (1.0 / 65536.0));
  let bandOffset = -ext.xy * bandScale;

  gi.bandIndex = clamp (vec2<i32> (renderCoord * bandScale + bandOffset),
                        vec2<i32> (0, 0),
                        vec2<i32> (gi.numVBands - 1, gi.numHBands - 1));

  gi.bandBase = gi.glyphLoc + 2;
  return gi;
}

/* Return pixels per em at this fragment.
 *
 * renderCoord:  em-space sample position
 * glyphLoc:     texel offset of glyph blob in atlas
 */
fn hb_gpu_ppem (renderCoord: vec2f, glyphLoc_: u32,
                hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> f32
{
  let gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  let emsPerPixel = fwidth (renderCoord);
  return min (gi.scale.x, gi.scale.y) /
         max (emsPerPixel.x, emsPerPixel.y);
}

/* Return per-pixel curve counts: (horizontal, vertical). */
fn _hb_gpu_curve_counts (renderCoord: vec2f, glyphLoc_: u32,
                         hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> vec2<i32>
{
  let gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  let hCount = hb_gpu_fetch (hb_gpu_atlas, gi.bandBase + gi.bandIndex.y).r;
  let vCount = hb_gpu_fetch (hb_gpu_atlas, gi.bandBase + gi.numHBands + gi.bandIndex.x).r;
  return vec2<i32> (hCount, vCount);
}

/* Single-sample coverage in [0, 1]. */
fn _hb_gpu_slug_single (renderCoord: vec2f, pixelsPerEm: vec2f, glyphLoc_: u32,
                           hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> f32
{

  let gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  let glyphLoc = gi.glyphLoc;
  let bandBase = gi.bandBase;
  let numHBands = gi.numHBands;

  var xcov: f32 = 0.0;
  var xwgt: f32 = 0.0;

  let hbandData = hb_gpu_fetch (hb_gpu_atlas, bandBase + gi.bandIndex.y);
  let hCurveCount = hbandData.r;
  /* Symmetric: choose rightward (desc) or leftward (asc) sort */
  let hSplit = f32 (hbandData.a) * HB_GPU_INV_UNITS;
  let hLeftRay = (renderCoord.x < hSplit);
  var hDataOffset: i32;
  if (hLeftRay) { hDataOffset = hbandData.b + 32768; }
  else          { hDataOffset = hbandData.g + 32768; }

  for (var ci: i32 = 0; ci < hCurveCount; ci++)
  {
    let curveOffset = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + hDataOffset + ci).r + 32768;

    let raw12 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset);
    let raw3 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset + 1);

    let q12 = vec4f (raw12) * HB_GPU_INV_UNITS;
    let q3 = vec2f (vec2<i32> (raw3.r, raw3.g)) * HB_GPU_INV_UNITS;

    let p12 = q12 - vec4f (renderCoord, renderCoord);
    let p3 = q3 - renderCoord;

    if (hLeftRay) {
      if (min (min (p12.x, p12.z), p3.x) * pixelsPerEm.x > 0.5) { break; }
    } else {
      if (max (max (p12.x, p12.z), p3.x) * pixelsPerEm.x < -0.5) { break; }
    }

    let code = _hb_gpu_calc_root_code (p12.y, p12.w, p3.y);
    if (code != 0u)
    {
      let a = q12.xy - q12.zw * 2.0 + q3;
      let b = q12.xy - q12.zw;
      let r = _hb_gpu_solve_horiz_poly (a, b, p12.xy) * pixelsPerEm.x;
      /* For leftward ray: saturate(0.5 - r) counts coverage from the left */
      var cov: vec2f;
      if (hLeftRay) { cov = clamp (vec2f (0.5) - r, vec2f (0.0), vec2f (1.0)); }
      else          { cov = clamp (r + vec2f (0.5), vec2f (0.0), vec2f (1.0)); }

      if ((code & 1u) != 0u)
      {
        xcov += cov.x;
        xwgt = max (xwgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1u)
      {
        xcov -= cov.y;
        xwgt = max (xwgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Crossings over a closed contour sum to zero, so a leftward ray
   * returns the negative of what a rightward ray would; flip it back
   * so that xcov and ycov keep a common sign convention. */
  if (hLeftRay) { xcov = -xcov; }

  var ycov: f32 = 0.0;
  var ywgt: f32 = 0.0;

  let vbandData = hb_gpu_fetch (hb_gpu_atlas, bandBase + numHBands + gi.bandIndex.x);
  let vCurveCount = vbandData.r;
  let vSplit = f32 (vbandData.a) * HB_GPU_INV_UNITS;
  let vLeftRay = (renderCoord.y < vSplit);
  var vDataOffset: i32;
  if (vLeftRay) { vDataOffset = vbandData.b + 32768; }
  else          { vDataOffset = vbandData.g + 32768; }

  for (var ci: i32 = 0; ci < vCurveCount; ci++)
  {
    let curveOffset = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + vDataOffset + ci).r + 32768;

    let raw12 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset);
    let raw3 = hb_gpu_fetch (hb_gpu_atlas, glyphLoc + curveOffset + 1);

    let q12 = vec4f (raw12) * HB_GPU_INV_UNITS;
    let q3 = vec2f (vec2<i32> (raw3.r, raw3.g)) * HB_GPU_INV_UNITS;

    let p12 = q12 - vec4f (renderCoord, renderCoord);
    let p3 = q3 - renderCoord;

    if (vLeftRay) {
      if (min (min (p12.y, p12.w), p3.y) * pixelsPerEm.y > 0.5) { break; }
    } else {
      if (max (max (p12.y, p12.w), p3.y) * pixelsPerEm.y < -0.5) { break; }
    }

    let code = _hb_gpu_calc_root_code (p12.x, p12.z, p3.x);
    if (code != 0u)
    {
      let a = q12.xy - q12.zw * 2.0 + q3;
      let b = q12.xy - q12.zw;
      let r = _hb_gpu_solve_vert_poly (a, b, p12.xy) * pixelsPerEm.y;
      var cov: vec2f;
      if (vLeftRay) { cov = clamp (vec2f (0.5) - r, vec2f (0.0), vec2f (1.0)); }
      else          { cov = clamp (r + vec2f (0.5), vec2f (0.0), vec2f (1.0)); }

      if ((code & 1u) != 0u)
      {
        ycov -= cov.x;
        ywgt = max (ywgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1u)
      {
        ycov += cov.y;
        ywgt = max (ywgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Ditto, for the vertical ray. */
  if (vLeftRay) { ycov = -ycov; }

  return _hb_gpu_calc_coverage (xcov, ycov, xwgt, ywgt);
}

/* Return coverage in [0, 1].
 *
 * renderCoord:    em-space sample position
 * glyphLoc:       texel offset of glyph blob in atlas
 * hb_gpu_atlas:   storage buffer pointer to the atlas
 */
/* The MSAA-aware implementation.  Caller supplies pixelsPerEm so
 * this function can be invoked from non-uniform control flow (for
 * example inside an op-stream branch in hb-gpu-paint-fragment.wgsl,
 * where WGSL would otherwise reject an fwidth call). */
fn _hb_gpu_slug (renderCoord: vec2f, pixelsPerEm: vec2f, glyphLoc_: u32,
                      hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> f32
{
  var c = _hb_gpu_slug_single (renderCoord, pixelsPerEm, glyphLoc_, hb_gpu_atlas);

  /* Inline ppem from pixelsPerEm so we don't re-enter hb_gpu_ppem
   * (which would call fwidth again -- rejected by WGSL when this
   * function is invoked from non-uniform control flow). */
  let gi_pp = _hb_gpu_decode_glyph (renderCoord, glyphLoc_, hb_gpu_atlas);
  let ppem = min (gi_pp.scale.x, gi_pp.scale.y) *
             min (pixelsPerEm.x, pixelsPerEm.y);

  if (ppem < 16.0)
  {
    let emsPerPixel = 1.0 / pixelsPerEm;
    let d = emsPerPixel * (1.0 / 3.0);
    let msaa = 0.25 *
      (_hb_gpu_slug_single (renderCoord + vec2f (-d.x, -d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas) +
       _hb_gpu_slug_single (renderCoord + vec2f ( d.x, -d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas) +
       _hb_gpu_slug_single (renderCoord + vec2f (-d.x,  d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas) +
       _hb_gpu_slug_single (renderCoord + vec2f ( d.x,  d.y), pixelsPerEm, glyphLoc_, hb_gpu_atlas));

    c = mix (c, msaa, smoothstep (16.0, 8.0, ppem));
  }

  return c;
}
/* Stem darkening for small sizes.
 *
 * coverage:    output of hb_gpu_draw
 * brightness:  foreground brightness in [0, 1]
 * ppem:        pixels per em at this fragment
 */
fn hb_gpu_stem_darken (coverage: f32, brightness: f32, ppem: f32) -> f32
{
  return pow (coverage,
	      mix (pow (2.0, brightness - 0.5), 1.0,
		   smoothstep (8.0, 48.0, ppem)));
}
        /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Shared fragment-shader helpers for the hb-gpu renderers.
 *
 * Requires Shader Model 5.0+.
 *
 * The caller must declare:
 *   StructuredBuffer<int4> hb_gpu_atlas : register(t0);
 */


#ifndef HB_GPU_UNITS_PER_EM
#define HB_GPU_UNITS_PER_EM 4
#endif

#define HB_GPU_INV_UNITS (1.0 / (float) HB_GPU_UNITS_PER_EM)


int4 hb_gpu_fetch (int offset)
{
  return hb_gpu_atlas[offset];
}
uint _hb_gpu_calc_root_code (float y1, float y2, float y3)
{
  uint i1 = asuint (y1) >> 31u;
  uint i2 = asuint (y2) >> 30u;
  uint i3 = asuint (y3) >> 29u;

  uint shift = (i2 & 2u) | (i1 & ~2u);
  shift = (i3 & 4u) | (shift & ~4u);

  return (0x2E74u >> shift) & 0x0101u;
}

float2 _hb_gpu_solve_horiz_poly (float2 a, float2 b, float2 p1)
{
  float ra = 1.0 / a.y;
  float rb = 0.5 / b.y;

  float d = sqrt (max (b.y * b.y - a.y * p1.y, 0.0));
  float t1 = (b.y - d) * ra;
  float t2 = (b.y + d) * ra;

  if (a.y == 0.0)
    t1 = t2 = p1.y * rb;

  return float2 ((a.x * t1 - b.x * 2.0) * t1 + p1.x,
                 (a.x * t2 - b.x * 2.0) * t2 + p1.x);
}

float2 _hb_gpu_solve_vert_poly (float2 a, float2 b, float2 p1)
{
  float ra = 1.0 / a.x;
  float rb = 0.5 / b.x;

  float d = sqrt (max (b.x * b.x - a.x * p1.x, 0.0));
  float t1 = (b.x - d) * ra;
  float t2 = (b.x + d) * ra;

  if (a.x == 0.0)
    t1 = t2 = p1.x * rb;

  return float2 ((a.y * t1 - b.y * 2.0) * t1 + p1.y,
                 (a.y * t2 - b.y * 2.0) * t2 + p1.y);
}

float _hb_gpu_calc_coverage (float xcov, float ycov, float xwgt, float ywgt)
{
  float coverage = max (abs (xcov * xwgt + ycov * ywgt) /
                        max (xwgt + ywgt, 1.0 / 65536.0),
                        min (abs (xcov), abs (ycov)));

  return clamp (coverage, 0.0, 1.0);
}


struct _hb_gpu_glyph_info
{
  int glyphLoc;
  int bandBase;
  int2 bandIndex;
  int numHBands;
  int numVBands;
  float2 scale;
};

_hb_gpu_glyph_info _hb_gpu_decode_glyph (float2 renderCoord, uint glyphLoc_)
{
  _hb_gpu_glyph_info gi;
  gi.glyphLoc = (int) glyphLoc_;

  int4 header0 = hb_gpu_fetch (gi.glyphLoc);
  int4 header1 = hb_gpu_fetch (gi.glyphLoc + 1);
  float4 ext = (float4) header0 * HB_GPU_INV_UNITS;
  gi.numHBands = header1.r;
  gi.numVBands = header1.g;
  gi.scale = float2 ((float) header1.b, (float) header1.a);

  float2 extSize = ext.zw - ext.xy;
  float2 bandScale = float2 ((float) gi.numVBands, (float) gi.numHBands) / max (extSize, float2 (1.0 / 65536.0, 1.0 / 65536.0));
  float2 bandOffset = -ext.xy * bandScale;

  gi.bandIndex = clamp ((int2) (renderCoord * bandScale + bandOffset),
                        int2 (0, 0),
                        int2 (gi.numVBands - 1, gi.numHBands - 1));

  gi.bandBase = gi.glyphLoc + 2;
  return gi;
}

/* Return pixels per em at this fragment.
 *
 * renderCoord:  em-space sample position
 * glyphLoc:     texel offset of glyph blob in atlas
 */
float hb_gpu_ppem (float2 renderCoord, uint glyphLoc_)
{
  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_);
  float2 emsPerPixel = fwidth (renderCoord);
  return min (gi.scale.x, gi.scale.y) /
	 max (emsPerPixel.x, emsPerPixel.y);
}

int2 _hb_gpu_curve_counts (float2 renderCoord, uint glyphLoc_)
{
  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_);
  int hCount = hb_gpu_fetch (gi.bandBase + gi.bandIndex.y).r;
  int vCount = hb_gpu_fetch (gi.bandBase + gi.numHBands + gi.bandIndex.x).r;
  return int2 (hCount, vCount);
}


/* Single-sample coverage in [0, 1]. */
float _hb_gpu_slug_single (float2 renderCoord, float2 pixelsPerEm, uint glyphLoc_)
{

  _hb_gpu_glyph_info gi = _hb_gpu_decode_glyph (renderCoord, glyphLoc_);
  int glyphLoc = gi.glyphLoc;
  int bandBase = gi.bandBase;
  int numHBands = gi.numHBands;

  float xcov = 0.0;
  float xwgt = 0.0;

  int4 hbandData = hb_gpu_fetch (bandBase + gi.bandIndex.y);
  int hCurveCount = hbandData.r;
  float hSplit = (float) hbandData.a * HB_GPU_INV_UNITS;
  bool hLeftRay = (renderCoord.x < hSplit);
  int hDataOffset = (hLeftRay ? hbandData.b : hbandData.g) + 32768;

  for (int ci = 0; ci < hCurveCount; ci++)
  {
    int curveOffset = hb_gpu_fetch (glyphLoc + hDataOffset + ci).r + 32768;

    int4 raw12 = hb_gpu_fetch (glyphLoc + curveOffset);
    int4 raw3 = hb_gpu_fetch (glyphLoc + curveOffset + 1);

    float4 q12 = (float4) raw12 * HB_GPU_INV_UNITS;
    float2 q3 = (float2) raw3.rg * HB_GPU_INV_UNITS;

    float4 p12 = q12 - float4 (renderCoord, renderCoord);
    float2 p3 = q3 - renderCoord;

    if (hLeftRay) {
      if (min (min (p12.x, p12.z), p3.x) * pixelsPerEm.x > 0.5) break;
    } else {
      if (max (max (p12.x, p12.z), p3.x) * pixelsPerEm.x < -0.5) break;
    }

    uint code = _hb_gpu_calc_root_code (p12.y, p12.w, p3.y);
    if (code != 0u)
    {
      float2 a = q12.xy - q12.zw * 2.0 + q3;
      float2 b = q12.xy - q12.zw;
      float2 r = _hb_gpu_solve_horiz_poly (a, b, p12.xy) * pixelsPerEm.x;
      float2 cov = hLeftRay ? clamp (float2 (0.5, 0.5) - r, 0.0, 1.0)
                            : clamp (r + float2 (0.5, 0.5), 0.0, 1.0);

      if ((code & 1u) != 0u)
      {
        xcov += cov.x;
        xwgt = max (xwgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1u)
      {
        xcov -= cov.y;
        xwgt = max (xwgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Crossings over a closed contour sum to zero, so a leftward ray
   * returns the negative of what a rightward ray would; flip it back
   * so that xcov and ycov keep a common sign convention. */
  if (hLeftRay)
    xcov = -xcov;

  float ycov = 0.0;
  float ywgt = 0.0;

  int4 vbandData = hb_gpu_fetch (bandBase + numHBands + gi.bandIndex.x);
  int vCurveCount = vbandData.r;
  float vSplit = (float) vbandData.a * HB_GPU_INV_UNITS;
  bool vLeftRay = (renderCoord.y < vSplit);
  int vDataOffset = (vLeftRay ? vbandData.b : vbandData.g) + 32768;

  for (int ci = 0; ci < vCurveCount; ci++)
  {
    int curveOffset = hb_gpu_fetch (glyphLoc + vDataOffset + ci).r + 32768;

    int4 raw12 = hb_gpu_fetch (glyphLoc + curveOffset);
    int4 raw3 = hb_gpu_fetch (glyphLoc + curveOffset + 1);

    float4 q12 = (float4) raw12 * HB_GPU_INV_UNITS;
    float2 q3 = (float2) raw3.rg * HB_GPU_INV_UNITS;

    float4 p12 = q12 - float4 (renderCoord, renderCoord);
    float2 p3 = q3 - renderCoord;

    if (vLeftRay) {
      if (min (min (p12.y, p12.w), p3.y) * pixelsPerEm.y > 0.5) break;
    } else {
      if (max (max (p12.y, p12.w), p3.y) * pixelsPerEm.y < -0.5) break;
    }

    uint code = _hb_gpu_calc_root_code (p12.x, p12.z, p3.x);
    if (code != 0u)
    {
      float2 a = q12.xy - q12.zw * 2.0 + q3;
      float2 b = q12.xy - q12.zw;
      float2 r = _hb_gpu_solve_vert_poly (a, b, p12.xy) * pixelsPerEm.y;
      float2 cov = vLeftRay ? clamp (float2 (0.5, 0.5) - r, 0.0, 1.0)
                            : clamp (r + float2 (0.5, 0.5), 0.0, 1.0);

      if ((code & 1u) != 0u)
      {
        ycov -= cov.x;
        ywgt = max (ywgt, clamp (1.0 - abs (r.x) * 2.0, 0.0, 1.0));
      }

      if (code > 1u)
      {
        ycov += cov.y;
        ywgt = max (ywgt, clamp (1.0 - abs (r.y) * 2.0, 0.0, 1.0));
      }
    }
  }

  /* Ditto, for the vertical ray. */
  if (vLeftRay)
    ycov = -ycov;

  return _hb_gpu_calc_coverage (xcov, ycov, xwgt, ywgt);
}

/* Return coverage in [0, 1].
 *
 * Caller must declare: StructuredBuffer<int4> hb_gpu_atlas
 *
 * renderCoord:  em-space sample position
 * glyphLoc:     texel offset of glyph blob in atlas
 */
/* The MSAA-aware implementation.  Caller supplies pixelsPerEm so
 * this function can be invoked from non-uniform control flow (for
 * example from a paint op-stream branch). */
float _hb_gpu_slug (float2 renderCoord, float2 pixelsPerEm, uint glyphLoc_)
{
  float c = _hb_gpu_slug_single (renderCoord, pixelsPerEm, glyphLoc_);

#ifndef HB_GPU_NO_MSAA
  float ppem = hb_gpu_ppem (renderCoord, glyphLoc_);

  if (ppem < 16.0)
  {
    float2 emsPerPixel = 1.0 / pixelsPerEm;
    float2 d = emsPerPixel * (1.0 / 3.0);
    float msaa = 0.25 *
      (_hb_gpu_slug_single (renderCoord + float2 (-d.x, -d.y), pixelsPerEm, glyphLoc_) +
       _hb_gpu_slug_single (renderCoord + float2 ( d.x, -d.y), pixelsPerEm, glyphLoc_) +
       _hb_gpu_slug_single (renderCoord + float2 (-d.x,  d.y), pixelsPerEm, glyphLoc_) +
       _hb_gpu_slug_single (renderCoord + float2 ( d.x,  d.y), pixelsPerEm, glyphLoc_));

    c = lerp (c, msaa, smoothstep (16.0, 8.0, ppem));
  }
#endif

  return c;
}
/* Stem darkening for small sizes.
 *
 * coverage:    output of hb_gpu_draw
 * brightness:  foreground brightness in [0, 1]
 * ppem:        pixels per em at this fragment
 */
float hb_gpu_stem_darken (float coverage, float brightness, float ppem)
{
  return pow (coverage,
	      lerp (pow (2.0, brightness - 0.5), 1.0,
		    smoothstep (8.0, 48.0, ppem)));
}
        /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Requires GLSL 3.30 or GLSL ES 3.00. */


/* Dilate a glyph vertex by half a pixel on screen.
 *
 * position:  object-space vertex position (modified in place)
 * texcoord:  em-space sample coordinates (modified in place)
 * normal:    object-space outward normal at this vertex
 * jac:       inverse of the 2x2 linear part of the em-to-object transform,
 *            stored row-major as (j00, j01, j10, j11).  Maps object-space
 *            displacements back to em-space for texcoord adjustment.
 *            For simple scaling with y-flip (the common case):
 *              em-to-object = [[s, 0], [0, -s]]
 *              jac = (1/s, 0, 0, -1/s)
 * m:         model-view-projection matrix
 * viewport:  viewport size in pixels
 */
void hb_gpu_dilate (inout vec2 position, inout vec2 texcoord,
		    vec2 normal, vec4 jac,
		    mat4 m, vec2 viewport)
{
  vec2 n = normalize (normal);

  vec4 clipPos = m * vec4 (position, 0.0, 1.0);
  vec4 clipN   = m * vec4 (n, 0.0, 0.0);

  float s = clipPos.w;
  float t = clipN.w;

  float u = (s * clipN.x - t * clipPos.x) * viewport.x;
  float v = (s * clipN.y - t * clipPos.y) * viewport.y;

  float s2 = s * s;
  float st = s * t;
  float uv = u * u + v * v;

  float denom = uv - st * st;
  float d = abs (denom) > 1.0 / 16777216.0
	  ? s2 * (st + sqrt (uv)) / denom
	  : 0.0;

  vec2 dPos = d * normal;
  position += dPos;
  texcoord += vec2 (dot (dPos, jac.xy), dot (dPos, jac.zw));
}
       /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Requires WGSL (WebGPU Shading Language). */


/* Dilate a glyph vertex by half a pixel on screen.
 *
 * position:  object-space vertex position (modified in place)
 * texcoord:  em-space sample coordinates (modified in place)
 * normal:    object-space outward normal at this vertex
 * jac:       inverse of the 2x2 linear part of the em-to-object transform,
 *            stored row-major as (j00, j01, j10, j11).  Maps object-space
 *            displacements back to em-space for texcoord adjustment.
 *            For simple scaling with y-flip (the common case):
 *              em-to-object = [[s, 0], [0, -s]]
 *              jac = (1/s, 0, 0, -1/s)
 * m:         model-view-projection matrix
 * viewport:  viewport size in pixels
 *
 * Returns (new_position, new_texcoord).
 */
fn hb_gpu_dilate (position: vec2f, texcoord: vec2f,
                  normal: vec2f, jac: vec4f,
                  m: mat4x4f, viewport: vec2f) -> array<vec2f, 2>
{
  let n = normalize (normal);

  let clipPos = m * vec4f (position, 0.0, 1.0);
  let clipN   = m * vec4f (n, 0.0, 0.0);

  let s = clipPos.w;
  let t = clipN.w;

  let u = (s * clipN.x - t * clipPos.x) * viewport.x;
  let v = (s * clipN.y - t * clipPos.y) * viewport.y;

  let s2 = s * s;
  let st = s * t;
  let uv = u * u + v * v;

  let denom = uv - st * st;
  var d: f32;
  if (abs (denom) > 1.0 / 16777216.0) {
    d = s2 * (st + sqrt (uv)) / denom;
  } else {
    d = 0.0;
  }

  let dPos = d * normal;
  return array<vec2f, 2> (
    position + dPos,
    texcoord + vec2f (dot (dPos, jac.xy), dot (dPos, jac.zw))
  );
}
  /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Requires Shader Model 5.0+. */


/* Dilate a glyph vertex by half a pixel on screen.
 *
 * position:  object-space vertex position (modified in place)
 * texcoord:  em-space sample coordinates (modified in place)
 * normal:    object-space outward normal at this vertex
 * jac:       inverse of the 2x2 linear part of the em-to-object transform
 * m:         model-view-projection matrix
 * viewport:  viewport size in pixels
 */
void hb_gpu_dilate (inout float2 position, inout float2 texcoord,
                    float2 normal, float4 jac,
                    float4x4 m, float2 viewport)
{
  float2 n = normalize (normal);

  float4 clipPos = mul (m, float4 (position, 0.0, 1.0));
  float4 clipN   = mul (m, float4 (n, 0.0, 0.0));

  float s = clipPos.w;
  float t = clipN.w;

  float u = (s * clipN.x - t * clipPos.x) * viewport.x;
  float v = (s * clipN.y - t * clipPos.y) * viewport.y;

  float s2 = s * s;
  float st = s * t;
  float uv = u * u + v * v;

  float denom = uv - st * st;
  float d = abs (denom) > 1.0 / 16777216.0
          ? s2 * (st + sqrt (uv)) / denom
          : 0.0;

  float2 dPos = d * normal;
  position += dPos;
  texcoord += float2 (dot (dPos, jac.xy), dot (dPos, jac.zw));
}
  /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Draw-renderer fragment shader entry.  The heavy lifting (Slug
 * coverage, MSAA, ppem, stem darkening) lives in the shared
 * hb-gpu-fragment.glsl that must be prepended to this source;
 * this file only adds the thin hb_gpu_draw() wrapper that lifts
 * pixelsPerEm out of fwidth() at uniform control flow before
 * calling the shared _hb_gpu_slug(). */


float hb_gpu_draw (vec2 renderCoord, uint glyphLoc_)
{
  vec2 pixelsPerEm = 1.0 / fwidth (renderCoord);
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_);
}
        /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Draw-renderer fragment shader entry (Metal).  Heavy lifting
 * (Slug coverage, MSAA, ppem, stem darkening) lives in the shared
 * hb-gpu-fragment.msl that must be prepended to this source. */


float hb_gpu_draw (float2 renderCoord, uint glyphLoc_,
		     device const short4* hb_gpu_atlas)
{
  float2 pixelsPerEm = 1.0 / fwidth (renderCoord);
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_, hb_gpu_atlas);
}
      /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Draw-renderer fragment shader entry (WGSL).  Heavy lifting
 * (Slug coverage, MSAA, ppem, stem darkening) lives in the shared
 * hb-gpu-fragment.wgsl that must be prepended to this source. */


fn hb_gpu_draw (renderCoord: vec2f, glyphLoc_: u32,
                  hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> f32
{
  let pixelsPerEm = 1.0 / fwidth (renderCoord);
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_, hb_gpu_atlas);
}
 /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Draw-renderer fragment shader entry (HLSL).  Heavy lifting
 * (Slug coverage, MSAA, ppem, stem darkening) lives in the shared
 * hb-gpu-fragment.hlsl that must be prepended to this source. */


float hb_gpu_draw (float2 renderCoord, uint glyphLoc_)
{
  float2 pixelsPerEm = 1.0 / fwidth (renderCoord);
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_);
}
      /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Paint-renderer fragment shader.
 *
 * Assumes the shared fragment helpers (hb-gpu-fragment.glsl) and
 * the draw-renderer fragment helpers (hb-gpu-draw-fragment.glsl)
 * are prepended to this source.  The draw helper provides
 * hb_gpu_draw() which this interpreter calls to compute clip-glyph
 * coverage.
 */


/* Fetch the i'th stop of a gradient color line starting at @stops_base
 * (2 texels per stop).  Resolves is_foreground to @foreground. */
vec4 _hb_gpu_stop_color (int stops_base, int i, vec4 foreground, out float offset)
{
  ivec4 a = hb_gpu_fetch (stops_base + i * 2);
  offset = float (a.r) / 32767.0;
  ivec4 b = hb_gpu_fetch (stops_base + i * 2 + 1);
  if ((a.g & 1) != 0)
    return vec4 (foreground.rgb, foreground.a * (float (b.a) / 32767.0));
  return vec4 (b) / 32767.0;
}

/* Apply the color-line extend mode to a projected `t` value. */
float _hb_gpu_extend_t (float t, int extend)
{
  if (extend == 1) {           /* HB_PAINT_EXTEND_REPEAT */
    return t - floor (t);
  } else if (extend == 2) {    /* HB_PAINT_EXTEND_REFLECT */
    float u = t - 2.0 * floor (t * 0.5);
    return u > 1.0 ? 2.0 - u : u;
  }
  return clamp (t, 0.0, 1.0);  /* PAD (default) */
}

/* Walk stops starting at @stops_base and return the sampled color
 * at @t.  Same logic reused by all gradient subtypes. */
vec4 _hb_gpu_eval_stops (int stops_base, int stop_count, float t, vec4 foreground)
{
  float off_prev;
  vec4 col_prev = _hb_gpu_stop_color (stops_base, 0, foreground, off_prev);
  if (t <= off_prev)
    return col_prev;
  for (int i = 1; i < stop_count; i++)
  {
    float off;
    vec4 col = _hb_gpu_stop_color (stops_base, i, foreground, off);
    if (t <= off)
    {
      float span = off - off_prev;
      float f = span > 1e-6 ? (t - off_prev) / span : 0.0;
      /* Interpolate in premultiplied space per OpenType COLR spec. */
      vec4 p0 = vec4 (col_prev.rgb * col_prev.a, col_prev.a);
      vec4 p1 = vec4 (col.rgb * col.a, col.a);
      vec4 pm = mix (p0, p1, f);
      return pm.a > 1e-6 ? vec4 (pm.rgb / pm.a, pm.a) : vec4 (0.0);
    }
    col_prev = col;
    off_prev = off;
  }
  return col_prev;
}

/* Apply the stored 2x2 M^-1 (row-major i16 Q10) to @v.  Scaling
 * renderCoord deltas back into canonical gradient space. */
vec2 _hb_gpu_apply_minv (ivec4 m, vec2 v)
{
  vec4 mf = vec4 (m) * (1.0 / 1024.0);
  return vec2 (mf.x * v.x + mf.y * v.y,
	       mf.z * v.x + mf.w * v.y);
}

/* Sample a linear gradient whose param blob starts at @grad_base:
 *   texel 0: (p0_rendered.x, p0_rendered.y, d_canonical.x, d_canonical.y)
 *   texel 1: L^-1 as i16 Q10 (row-major)
 *   texels 2..: stops (2 texels each)
 * Evaluate t in untransformed space. */
vec4 _hb_gpu_sample_linear (vec2 renderCoord, int grad_base,
			    int stop_count, int extend, vec4 foreground)
{
  ivec4 t0 = hb_gpu_fetch (grad_base);
  ivec4 m  = hb_gpu_fetch (grad_base + 1);
  vec2 p0_r = vec2 (float (t0.r), float (t0.g));
  vec2 d    = vec2 (float (t0.b), float (t0.a));
  float denom = dot (d, d);
  if (denom < 1e-6) return vec4 (0.0);
  vec2 p = _hb_gpu_apply_minv (m, renderCoord - p0_r);
  float t = dot (p, d) / denom;
  t = _hb_gpu_extend_t (t, extend);

  return _hb_gpu_eval_stops (grad_base + 2, stop_count, t, foreground);
}

/* Sample a two-circle radial gradient whose param blob starts at
 * @grad_base:
 *   texel 0: (c0_rendered.x, c0_rendered.y, d_canonical.x, d_canonical.y)
 *     d = c1 - c0 in untransformed space
 *   texel 1: (r0, r1, _, _) in untransformed font units
 *   texel 2: L^-1 as i16 Q10 (row-major)
 *   texels 3..: stops (2 texels each)
 * Solves |p - t*cd|^2 = (r0 + t*(r1-r0))^2 with p in untransformed
 * space, so non-uniform scale / shear on the transform becomes a
 * proper ellipse-in-rendered-space instead of a scalar-fudge. */
vec4 _hb_gpu_sample_radial (vec2 renderCoord, int grad_base,
			    int stop_count, int extend, vec4 foreground)
{
  ivec4 t0 = hb_gpu_fetch (grad_base);
  ivec4 t1 = hb_gpu_fetch (grad_base + 1);
  ivec4 m  = hb_gpu_fetch (grad_base + 2);
  vec2 c0_r = vec2 (float (t0.r), float (t0.g));
  vec2 cd   = vec2 (float (t0.b), float (t0.a));
  float r0 = float (t1.r);
  float r1 = float (t1.g);

  float dr = r1 - r0;
  vec2 p  = _hb_gpu_apply_minv (m, renderCoord - c0_r);

  float A = dot (cd, cd) - dr * dr;
  float B = -2.0 * (dot (p, cd) + r0 * dr);
  float C = dot (p, p) - r0 * r0;

  float t;
  if (abs (A) > 1e-6)
  {
    float disc = B * B - 4.0 * A * C;
    if (disc < 0.0) return vec4 (0.0);
    float sq = sqrt (disc);
    /* Prefer the larger root; fall back to the smaller if the
     * larger gives a negative interpolated radius. */
    float t1 = (-B + sq) / (2.0 * A);
    float t2 = (-B - sq) / (2.0 * A);
    t = (r0 + t1 * dr >= 0.0) ? t1 : t2;
  }
  else
  {
    if (abs (B) < 1e-6) return vec4 (0.0);
    t = -C / B;
  }

  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (grad_base + 3, stop_count, t, foreground);
}

/* Sample a sweep gradient whose param blob starts at @grad_base:
 *   texel 0: (center_rendered.x, center_rendered.y, start_q14, end_q14)
 *            start/end are Q14 fractions of pi in untransformed space
 *   texel 1: L^-1 as i16 Q10 (row-major)
 *   texels 2..: stops (2 texels each) */
vec4 _hb_gpu_sample_sweep (vec2 renderCoord, int grad_base,
			   int stop_count, int extend, vec4 foreground)
{
  ivec4 t0 = hb_gpu_fetch (grad_base);
  ivec4 m  = hb_gpu_fetch (grad_base + 1);
  vec2 c_r = vec2 (float (t0.r), float (t0.g));
  float a0 = float (t0.b) / 16384.0;  /* fraction of pi */
  float a1 = float (t0.a) / 16384.0;
  float span = a1 - a0;
  if (abs (span) < 1e-6) return vec4 (0.0);

  vec2 p = _hb_gpu_apply_minv (m, renderCoord - c_r);
  /* atan2 returns (-pi, pi]; normalize to [0, 2) fractions of pi. */
  float ang = atan (p.y, p.x) / 3.14159265358979;
  if (ang < 0.0) ang += 2.0;
  float t = (ang - a0) / span;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (grad_base + 2, stop_count, t, foreground);
}

/* Composite two premultiplied RGBA layers using one of the COLRv1
 * compositing modes.  Unsupported modes fall back to SRC_OVER.
 * Values match hb_paint_composite_mode_t. */
vec4 _hb_gpu_composite (vec4 src, vec4 dst, int mode)
{
  vec4 r = src + dst * (1.0 - src.a);  /* SRC_OVER default */

  /* Approximate unsupported COLRv1 modes with the nearest Porter-Duff
   * mode we do implement.  Better a recognizable rendering than a
   * silent SRC_OVER fallback.  DIFFERENCE / EXCLUSION / HSL_* are
   * not similar enough to anything we have, so they still fall
   * through to SRC_OVER below. */
  if      (mode == 14 || mode == 18 || mode == 19) mode = 23; /* OVERLAY / COLOR_BURN / HARD_LIGHT -> MULTIPLY */
  else if (mode == 17 || mode == 20)               mode = 13; /* COLOR_DODGE / SOFT_LIGHT -> SCREEN */

  if      (mode == 0)  r = vec4 (0.0);                       /* CLEAR */
  else if (mode == 1)  r = src;                              /* SRC */
  else if (mode == 2)  r = dst;                              /* DST */
  else if (mode == 4)  r = dst + src * (1.0 - dst.a);        /* DST_OVER */
  else if (mode == 5)  r = src * dst.a;                      /* SRC_IN */
  else if (mode == 6)  r = dst * src.a;                      /* DST_IN */
  else if (mode == 7)  r = src * (1.0 - dst.a);              /* SRC_OUT */
  else if (mode == 8)  r = dst * (1.0 - src.a);              /* DST_OUT */
  else if (mode == 9)                                        /* SRC_ATOP */
    r = src * dst.a + dst * (1.0 - src.a);
  else if (mode == 10)                                       /* DST_ATOP */
    r = dst * src.a + src * (1.0 - dst.a);
  else if (mode == 11)                                       /* XOR */
    r = src * (1.0 - dst.a) + dst * (1.0 - src.a);
  else if (mode == 12)                                       /* PLUS */
    r = min (src + dst, vec4 (1.0));
  else if (mode == 13) {                                     /* SCREEN (premul) */
    r.rgb = src.rgb + dst.rgb - src.rgb * dst.rgb;
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 15) {                                     /* DARKEN */
    r.rgb = min (src.rgb * dst.a, dst.rgb * src.a)
          + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a);
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 16) {                                     /* LIGHTEN */
    r.rgb = max (src.rgb * dst.a, dst.rgb * src.a)
          + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a);
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 23) {                                     /* MULTIPLY (premul) */
    r.rgb = src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a)
          + src.rgb * dst.rgb;
    r.a = src.a + dst.a - src.a * dst.a;
  }
  /* SRC_OVER (3) and DIFFERENCE / EXCLUSION / HSL_* (21, 22, 24-27)
   * fall through to the SRC_OVER default. */

  return r;
}

/* Wrap _hb_gpu_slug with a sub-glyph extents bail-out.  Many
 * paint layers cover a small region of the outer glyph quad; for
 * fragments outside the layer's bbox (with an AA + MSAA-spread
 * margin) the slug coverage is exactly 0, so we can skip the
 * band/curve walk entirely. */
float _hb_gpu_slug_clipped (vec2 renderCoord, vec2 pixelsPerEm, uint glyphLoc_)
{
  ivec4 header0 = hb_gpu_fetch (int (glyphLoc_));
  vec4 ext = vec4 (header0) * HB_GPU_INV_UNITS;
  vec2 margin = 2.0 / pixelsPerEm;
  if (any (lessThan    (renderCoord, ext.xy - margin)) ||
      any (greaterThan (renderCoord, ext.zw + margin)))
    return 0.0;
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_);
}

/* Combine slug coverages from all clip outlines on the current
 * layer.  Factored out of LAYER_SOLID and LAYER_GRADIENT so the
 * shader has one set of inlined slug walks instead of two.  flags
 * bits: 0x100 = HAS_CLIP2; 0x200 = HAS_CLIP3 (HAS_CLIP3 implies
 * HAS_CLIP2). */
float _hb_gpu_layer_coverage (vec2 renderCoord, vec2 pixelsPerEm,
			      int base, int flags,
			      int clip1_payload, int clip2_payload, int clip3_payload)
{
  float cov = _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				    uint (base + clip1_payload));
  if ((flags & 0x100) != 0)
  {
    cov *= _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				 uint (base + clip2_payload));
    if ((flags & 0x200) != 0)
      cov *= _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				   uint (base + clip3_payload));
  }
  return cov;
}

/* Walks the paint blob's flat op stream and returns a
 * premultiplied RGBA coverage value for the current fragment.
 *
 * glyphLoc: atlas texel offset of the paint-blob header.
 * foreground: caller-supplied foreground color, used when an op
 *             sets the is_foreground flag.
 */
#define HB_GPU_PAINT_GROUP_DEPTH 4

vec4 hb_gpu_paint (vec2 renderCoord, uint glyphLoc, vec4 foreground,
		   out float coverage)
{
  /* fwidth once, at uniform control flow: every per-layer
   * coverage sample below uses this pre-computed pixelsPerEm via
   * _hb_gpu_slug. */
  vec2 pixelsPerEm = 1.0 / fwidth (renderCoord);

  int base    = int (glyphLoc);
  ivec4 h0    = hb_gpu_fetch (base);      /* (num_ops, _, _, _) */
  ivec4 h2    = hb_gpu_fetch (base + 2);  /* (ops_offset, _, _, _) */
  int num_ops = h0.r;
  int cursor  = base + h2.r;

  vec4 acc = vec4 (0.0);
  vec4 group_stack[HB_GPU_PAINT_GROUP_DEPTH];
  int sp = 0;
  coverage = 0.0;

  for (int i = 0; i < num_ops; i++)
  {
    ivec4 op    = hb_gpu_fetch (cursor);
    int op_type = op.r;
    int aux     = op.g;
    int payload = (op.b << 16) | (op.a & 0xffff);

    if (op_type == 0)  /* LAYER_SOLID */
    {
      /* texel 1: (clip2_hi, clip2_lo, clip3_hi, clip3_lo) -- valid
       *           per HAS_CLIP2 / HAS_CLIP3 flag bits.
       * texel 2: RGBA as signed Q15. */
      ivec4 op2 = hb_gpu_fetch (cursor + 1);
      int clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      int clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      ivec4 ct = hb_gpu_fetch (cursor + 2);
      vec4 col = ((aux & 1) != 0)
	       ? vec4 (foreground.rgb, foreground.a * (float (ct.a) / 32767.0))
	       : vec4 (ct) / 32767.0;

      float cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
					  base, aux,
					  payload, clip2_payload, clip3_payload);
      coverage = max (coverage, cov);
      vec4 src = vec4 (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor += 3;
    }
    else if (op_type == 1)  /* LAYER_GRADIENT */
    {
      /* texel 1: (clip2_hi, clip2_lo, clip3_hi, clip3_lo) -- valid
       *           per HAS_CLIP2 / HAS_CLIP3 flag bits.
       * texel 2: (grad_payload_hi, grad_payload_lo, extend, stop_count) */
      ivec4 op2 = hb_gpu_fetch (cursor + 1);
      int clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      int clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      ivec4 op3 = hb_gpu_fetch (cursor + 2);
      int grad_payload = (op3.r << 16) | (op3.g & 0xffff);
      int extend       = op3.b;
      int stop_count   = op3.a;
      int subtype      = aux & 0xff;

      vec4 col = vec4 (0.0);
      if (subtype == 0)       /* linear */
        col = _hb_gpu_sample_linear (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground);
      else if (subtype == 1)  /* radial */
        col = _hb_gpu_sample_radial (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground);
      else if (subtype == 2)  /* sweep */
        col = _hb_gpu_sample_sweep  (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground);

      float cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
					  base, aux,
					  payload, clip2_payload, clip3_payload);
      coverage = max (coverage, cov);
      vec4 src = vec4 (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor += 3;
    }
    else if (op_type == 2)  /* PUSH_GROUP */
    {
      if (sp < HB_GPU_PAINT_GROUP_DEPTH) {
        group_stack[sp] = acc;
        sp++;
      }
      acc = vec4 (0.0);
      cursor += 1;
    }
    else if (op_type == 3)  /* POP_GROUP */
    {
      if (sp > 0) {
        sp--;
        vec4 src = acc;
        vec4 dst = group_stack[sp];
        acc = _hb_gpu_composite (src, dst, aux);
      }
      cursor += 1;
    }
    else
    {
      break;
    }
  }

  return acc;
}
     /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Paint-renderer fragment shader (Metal).
 *
 * Assumes the shared fragment helpers (hb-gpu-fragment.msl) and
 * the draw-renderer fragment helpers (hb-gpu-draw-fragment.msl)
 * are prepended to this source.
 */


/* Fetch the i'th stop (2 texels per stop), resolving is_foreground. */
static float4 _hb_gpu_stop_color (device const short4* hb_gpu_atlas,
				  int stops_base, int i, float4 foreground,
				  thread float &offset)
{
  int4 a = int4 (hb_gpu_atlas[stops_base + i * 2]);
  offset = float (a.r) / 32767.0;
  int4 b = int4 (hb_gpu_atlas[stops_base + i * 2 + 1]);
  if ((a.g & 1) != 0)
    return float4 (foreground.rgb, foreground.a * (float (b.a) / 32767.0));
  return float4 (b) / 32767.0;
}

static float _hb_gpu_extend_t (float t, int extend)
{
  if (extend == 1) return t - floor (t);
  if (extend == 2) {
    float u = t - 2.0 * floor (t * 0.5);
    return u > 1.0 ? 2.0 - u : u;
  }
  return clamp (t, 0.0, 1.0);
}

static float4 _hb_gpu_eval_stops (device const short4* hb_gpu_atlas,
				  int stops_base, int stop_count,
				  float t, float4 foreground)
{
  float off_prev;
  float4 col_prev = _hb_gpu_stop_color (hb_gpu_atlas, stops_base, 0, foreground, off_prev);
  if (t <= off_prev) return col_prev;
  for (int i = 1; i < stop_count; i++)
  {
    float off;
    float4 col = _hb_gpu_stop_color (hb_gpu_atlas, stops_base, i, foreground, off);
    if (t <= off)
    {
      float span = off - off_prev;
      float f = span > 1e-6 ? (t - off_prev) / span : 0.0;
      float4 p0 = float4 (col_prev.rgb * col_prev.a, col_prev.a);
      float4 p1 = float4 (col.rgb * col.a, col.a);
      float4 pm = mix (p0, p1, f);
      return pm.a > 1e-6 ? float4 (pm.rgb / pm.a, pm.a) : float4 (0.0);
    }
    col_prev = col;
    off_prev = off;
  }
  return col_prev;
}

/* Apply the stored 2x2 M^-1 (row-major i16 Q10) to a vector. */
static float2 _hb_gpu_apply_minv (int4 m, float2 v)
{
  float4 mf = float4 (m) * (1.0 / 1024.0);
  return float2 (mf.x * v.x + mf.y * v.y,
		 mf.z * v.x + mf.w * v.y);
}

static float4 _hb_gpu_sample_linear (float2 renderCoord, int grad_base,
				     int stop_count, int extend,
				     float4 foreground,
				     device const short4* hb_gpu_atlas)
{
  int4 t0 = int4 (hb_gpu_atlas[grad_base]);
  int4 m  = int4 (hb_gpu_atlas[grad_base + 1]);
  float2 p0_r = float2 (float (t0.r), float (t0.g));
  float2 d    = float2 (float (t0.b), float (t0.a));
  float denom = dot (d, d);
  if (denom < 1e-6) return float4 (0.0);
  float2 p = _hb_gpu_apply_minv (m, renderCoord - p0_r);
  float t = dot (p, d) / denom;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (hb_gpu_atlas, grad_base + 2, stop_count, t, foreground);
}

static float4 _hb_gpu_sample_radial (float2 renderCoord, int grad_base,
				     int stop_count, int extend,
				     float4 foreground,
				     device const short4* hb_gpu_atlas)
{
  int4 t0 = int4 (hb_gpu_atlas[grad_base]);
  int4 t1 = int4 (hb_gpu_atlas[grad_base + 1]);
  int4 m  = int4 (hb_gpu_atlas[grad_base + 2]);
  float2 c0_r = float2 (float (t0.r), float (t0.g));
  float2 cd   = float2 (float (t0.b), float (t0.a));
  float r0 = float (t1.r);
  float r1 = float (t1.g);

  float dr = r1 - r0;
  float2 p  = _hb_gpu_apply_minv (m, renderCoord - c0_r);

  float A = dot (cd, cd) - dr * dr;
  float B = -2.0 * (dot (p, cd) + r0 * dr);
  float C = dot (p, p) - r0 * r0;

  float t;
  if (abs (A) > 1e-6)
  {
    float disc = B * B - 4.0 * A * C;
    if (disc < 0.0) return float4 (0.0);
    float sq = sqrt (disc);
    float t1r = (-B + sq) / (2.0 * A);
    float t2r = (-B - sq) / (2.0 * A);
    t = (r0 + t1r * dr >= 0.0) ? t1r : t2r;
  }
  else
  {
    if (abs (B) < 1e-6) return float4 (0.0);
    t = -C / B;
  }
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (hb_gpu_atlas, grad_base + 3, stop_count, t, foreground);
}

static float4 _hb_gpu_sample_sweep (float2 renderCoord, int grad_base,
				    int stop_count, int extend,
				    float4 foreground,
				    device const short4* hb_gpu_atlas)
{
  int4 t0 = int4 (hb_gpu_atlas[grad_base]);
  int4 m  = int4 (hb_gpu_atlas[grad_base + 1]);
  float2 c_r = float2 (float (t0.r), float (t0.g));
  float a0 = float (t0.b) / 16384.0;
  float a1 = float (t0.a) / 16384.0;
  float span = a1 - a0;
  if (abs (span) < 1e-6) return float4 (0.0);

  float2 p = _hb_gpu_apply_minv (m, renderCoord - c_r);
  float ang = atan2 (p.y, p.x) / 3.14159265358979;
  if (ang < 0.0) ang += 2.0;
  float t = (ang - a0) / span;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (hb_gpu_atlas, grad_base + 2, stop_count, t, foreground);
}

static float4 _hb_gpu_composite (float4 src, float4 dst, int mode)
{
  float4 r = src + dst * (1.0 - src.a);  /* SRC_OVER default */

  /* Mode numbers match hb_paint_composite_mode_t.  SRC_OVER (3) is
   * the default `r` above. */

  /* Approximate unsupported modes with the nearest Porter-Duff mode
   * we do implement.  DIFFERENCE / EXCLUSION / HSL_* still fall
   * through to SRC_OVER below. */
  if      (mode == 14 || mode == 18 || mode == 19) mode = 23; /* OVERLAY / COLOR_BURN / HARD_LIGHT -> MULTIPLY */
  else if (mode == 17 || mode == 20)               mode = 13; /* COLOR_DODGE / SOFT_LIGHT -> SCREEN */

  if      (mode == 0)  r = float4 (0.0);                       /* CLEAR */
  else if (mode == 1)  r = src;                                /* SRC */
  else if (mode == 2)  r = dst;                                /* DST */
  else if (mode == 4)  r = dst + src * (1.0 - dst.a);          /* DST_OVER */
  else if (mode == 5)  r = src * dst.a;                        /* SRC_IN */
  else if (mode == 6)  r = dst * src.a;                        /* DST_IN */
  else if (mode == 7)  r = src * (1.0 - dst.a);                /* SRC_OUT */
  else if (mode == 8)  r = dst * (1.0 - src.a);                /* DST_OUT */
  else if (mode == 9)                                          /* SRC_ATOP */
    r = src * dst.a + dst * (1.0 - src.a);
  else if (mode == 10)                                         /* DST_ATOP */
    r = dst * src.a + src * (1.0 - dst.a);
  else if (mode == 11)                                         /* XOR */
    r = src * (1.0 - dst.a) + dst * (1.0 - src.a);
  else if (mode == 12)                                         /* PLUS */
    r = min (src + dst, float4 (1.0));
  else if (mode == 13) {                                       /* SCREEN (premul) */
    r.rgb = src.rgb + dst.rgb - src.rgb * dst.rgb;
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 15) {                                       /* DARKEN */
    r.rgb = min (src.rgb * dst.a, dst.rgb * src.a)
          + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a);
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 16) {                                       /* LIGHTEN */
    r.rgb = max (src.rgb * dst.a, dst.rgb * src.a)
          + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a);
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 23) {                                       /* MULTIPLY (premul) */
    r.rgb = src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a)
          + src.rgb * dst.rgb;
    r.a = src.a + dst.a - src.a * dst.a;
  }
  /* SRC_OVER (3) and DIFFERENCE / EXCLUSION / HSL_* (21, 22, 24-27)
   * fall through to the SRC_OVER default. */

  return r;
}

/* Wrap _hb_gpu_slug with a sub-glyph extents bail-out.  Many
 * paint layers cover a small region of the outer glyph quad; for
 * fragments outside the layer's bbox (with an AA + MSAA-spread
 * margin) the slug coverage is exactly 0, so we can skip the
 * band/curve walk entirely. */
float _hb_gpu_slug_clipped (float2 renderCoord, float2 pixelsPerEm, uint glyphLoc_,
			    device const short4* hb_gpu_atlas)
{
  int4 header0 = hb_gpu_fetch (hb_gpu_atlas, int (glyphLoc_));
  float4 ext = float4 (header0) * HB_GPU_INV_UNITS;
  float2 margin = 2.0 / pixelsPerEm;
  if (any (renderCoord < ext.xy - margin) ||
      any (renderCoord > ext.zw + margin))
    return 0.0;
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_, hb_gpu_atlas);
}

/* Combine slug coverages from all clip outlines on the layer.
 * Factored out so the shader has one set of inlined slug walks
 * instead of two (one per LAYER op type).  flags bits: 0x100 =
 * HAS_CLIP2; 0x200 = HAS_CLIP3 (HAS_CLIP3 implies HAS_CLIP2). */
static float
_hb_gpu_layer_coverage (float2 renderCoord, float2 pixelsPerEm,
			int base, int flags,
			int clip1_payload, int clip2_payload, int clip3_payload,
			device const short4* hb_gpu_atlas)
{
  float cov = _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				    uint (base + clip1_payload), hb_gpu_atlas);
  if ((flags & 0x100) != 0)
  {
    cov *= _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				 uint (base + clip2_payload), hb_gpu_atlas);
    if ((flags & 0x200) != 0)
      cov *= _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				   uint (base + clip3_payload), hb_gpu_atlas);
  }
  return cov;
}

#define HB_GPU_PAINT_GROUP_DEPTH 4

float4 hb_gpu_paint (float2 renderCoord, uint glyphLoc_, float4 foreground,
		     device const short4* hb_gpu_atlas,
		     thread float &coverage)
{
  /* fwidth once, at uniform control flow. */
  float2 pixelsPerEm = 1.0 / fwidth (renderCoord);

  int base    = int (glyphLoc_);
  int4 h0     = int4 (hb_gpu_atlas[base]);       /* (num_ops, _, _, _) */
  int4 h2     = int4 (hb_gpu_atlas[base + 2]);   /* (ops_offset, _, _, _) */
  int num_ops = h0.r;
  int cursor  = base + h2.r;

  float4 acc = float4 (0.0);
  float4 group_stack[HB_GPU_PAINT_GROUP_DEPTH];
  int sp = 0;
  coverage = 0.0;

  for (int i = 0; i < num_ops; i++)
  {
    int4 op     = int4 (hb_gpu_atlas[cursor]);
    int op_type = op.r;
    int aux     = op.g;
    int payload = (op.b << 16) | (op.a & 0xffff);

    if (op_type == 0)  /* LAYER_SOLID */
    {
      /* texel 1: (clip2_hi, clip2_lo, clip3_hi, clip3_lo) -- valid
       *           per HAS_CLIP2 / HAS_CLIP3 flag bits.
       * texel 2: RGBA as signed Q15. */
      int4 op2 = int4 (hb_gpu_atlas[cursor + 1]);
      int clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      int clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      int4 ct = int4 (hb_gpu_atlas[cursor + 2]);
      float4 col = ((aux & 1) != 0)
		 ? float4 (foreground.rgb, foreground.a * (float (ct.a) / 32767.0))
		 : float4 (ct) / 32767.0;

      float cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
					  base, aux,
					  payload, clip2_payload, clip3_payload,
					  hb_gpu_atlas);
      coverage = max (coverage, cov);
      float4 src = float4 (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor += 3;
    }
    else if (op_type == 1)  /* LAYER_GRADIENT */
    {
      int4 op2 = int4 (hb_gpu_atlas[cursor + 1]);
      int clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      int clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      int4 op3 = int4 (hb_gpu_atlas[cursor + 2]);
      int grad_payload = (op3.r << 16) | (op3.g & 0xffff);
      int extend       = op3.b;
      int stop_count   = op3.a;
      int subtype      = aux & 0xff;

      float4 col = float4 (0.0);
      if (subtype == 0)
        col = _hb_gpu_sample_linear (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground,
                                     hb_gpu_atlas);
      else if (subtype == 1)
        col = _hb_gpu_sample_radial (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground,
                                     hb_gpu_atlas);
      else if (subtype == 2)
        col = _hb_gpu_sample_sweep  (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground,
                                     hb_gpu_atlas);

      float cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
					  base, aux,
					  payload, clip2_payload, clip3_payload,
					  hb_gpu_atlas);
      coverage = max (coverage, cov);
      float4 src = float4 (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor += 3;
    }
    else if (op_type == 2)  /* PUSH_GROUP */
    {
      if (sp < HB_GPU_PAINT_GROUP_DEPTH) {
        group_stack[sp] = acc;
        sp++;
      }
      acc = float4 (0.0);
      cursor += 1;
    }
    else if (op_type == 3)  /* POP_GROUP */
    {
      if (sp > 0) {
        sp--;
        float4 src = acc;
        float4 dst = group_stack[sp];
        acc = _hb_gpu_composite (src, dst, aux);
      }
      cursor += 1;
    }
    else
    {
      break;
    }
  }

  return acc;
}
     /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Paint-renderer fragment shader (WGSL).
 *
 * Assumes the shared fragment helpers (hb-gpu-fragment.wgsl) and
 * the draw-renderer fragment helpers (hb-gpu-draw-fragment.wgsl)
 * are prepended to this source.
 *
 * atlas is passed as an explicit storage-buffer pointer parameter,
 * matching WGSL's scoping rules.
 */


fn _hb_gpu_stop_color (hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>,
                       stops_base: i32, i: i32, foreground: vec4f,
                       offset: ptr<function, f32>) -> vec4f
{
  let a = hb_gpu_fetch (hb_gpu_atlas, stops_base + i * 2);
  *offset = f32 (a.r) / 32767.0;
  let b = hb_gpu_fetch (hb_gpu_atlas, stops_base + i * 2 + 1);
  if ((a.g & 1) != 0) {
    return vec4f (foreground.rgb, foreground.a * (f32 (b.a) / 32767.0));
  }
  return vec4f (b) / 32767.0;
}

fn _hb_gpu_extend_t (t: f32, extend: i32) -> f32
{
  if (extend == 1) { return t - floor (t); }
  if (extend == 2) {
    let u = t - 2.0 * floor (t * 0.5);
    if (u > 1.0) { return 2.0 - u; }
    return u;
  }
  return clamp (t, 0.0, 1.0);
}

fn _hb_gpu_eval_stops (hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>,
                       stops_base: i32, stop_count: i32,
                       t: f32, foreground: vec4f) -> vec4f
{
  var off_prev: f32;
  var col_prev = _hb_gpu_stop_color (hb_gpu_atlas, stops_base, 0, foreground, &off_prev);
  if (t <= off_prev) { return col_prev; }
  for (var i: i32 = 1; i < stop_count; i = i + 1)
  {
    var off: f32;
    let col = _hb_gpu_stop_color (hb_gpu_atlas, stops_base, i, foreground, &off);
    if (t <= off)
    {
      let span = off - off_prev;
      var f: f32 = 0.0;
      if (span > 1e-6) { f = (t - off_prev) / span; }
      let p0 = vec4f (col_prev.rgb * col_prev.a, col_prev.a);
      let p1 = vec4f (col.rgb * col.a, col.a);
      let pm = mix (p0, p1, f);
      if (pm.a > 1e-6) { return vec4f (pm.rgb / pm.a, pm.a); }
      return vec4f (0.0);
    }
    col_prev = col;
    off_prev = off;
  }
  return col_prev;
}

/* Apply the stored 2x2 M^-1 (row-major i16 Q10) to a vector. */
fn _hb_gpu_apply_minv (m: vec4<i32>, v: vec2f) -> vec2f
{
  let mf = vec4f (m) * (1.0 / 1024.0);
  return vec2f (mf.x * v.x + mf.y * v.y,
                mf.z * v.x + mf.w * v.y);
}

fn _hb_gpu_sample_linear (renderCoord: vec2f, grad_base: i32,
                          stop_count: i32, extend: i32, foreground: vec4f,
                          hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> vec4f
{
  let t0 = hb_gpu_fetch (hb_gpu_atlas, grad_base);
  let m  = hb_gpu_fetch (hb_gpu_atlas, grad_base + 1);
  let p0_r = vec2f (f32 (t0.r), f32 (t0.g));
  let d    = vec2f (f32 (t0.b), f32 (t0.a));
  let denom = dot (d, d);
  if (denom < 1e-6) { return vec4f (0.0); }
  let p = _hb_gpu_apply_minv (m, renderCoord - p0_r);
  var t = dot (p, d) / denom;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (hb_gpu_atlas, grad_base + 2, stop_count, t, foreground);
}

fn _hb_gpu_sample_radial (renderCoord: vec2f, grad_base: i32,
                          stop_count: i32, extend: i32, foreground: vec4f,
                          hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> vec4f
{
  let t0 = hb_gpu_fetch (hb_gpu_atlas, grad_base);
  let t1 = hb_gpu_fetch (hb_gpu_atlas, grad_base + 1);
  let m  = hb_gpu_fetch (hb_gpu_atlas, grad_base + 2);
  let c0_r = vec2f (f32 (t0.r), f32 (t0.g));
  let cd   = vec2f (f32 (t0.b), f32 (t0.a));
  let r0 = f32 (t1.r);
  let r1 = f32 (t1.g);

  let dr = r1 - r0;
  let p  = _hb_gpu_apply_minv (m, renderCoord - c0_r);

  let A = dot (cd, cd) - dr * dr;
  let B = -2.0 * (dot (p, cd) + r0 * dr);
  let C = dot (p, p) - r0 * r0;

  var t: f32;
  if (abs (A) > 1e-6)
  {
    let disc = B * B - 4.0 * A * C;
    if (disc < 0.0) { return vec4f (0.0); }
    let sq = sqrt (disc);
    let t1r = (-B + sq) / (2.0 * A);
    let t2r = (-B - sq) / (2.0 * A);
    if (r0 + t1r * dr >= 0.0) { t = t1r; } else { t = t2r; }
  }
  else
  {
    if (abs (B) < 1e-6) { return vec4f (0.0); }
    t = -C / B;
  }
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (hb_gpu_atlas, grad_base + 3, stop_count, t, foreground);
}

fn _hb_gpu_sample_sweep (renderCoord: vec2f, grad_base: i32,
                         stop_count: i32, extend: i32, foreground: vec4f,
                         hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> vec4f
{
  let t0 = hb_gpu_fetch (hb_gpu_atlas, grad_base);
  let m  = hb_gpu_fetch (hb_gpu_atlas, grad_base + 1);
  let c_r = vec2f (f32 (t0.r), f32 (t0.g));
  let a0 = f32 (t0.b) / 16384.0;
  let a1 = f32 (t0.a) / 16384.0;
  let span = a1 - a0;
  if (abs (span) < 1e-6) { return vec4f (0.0); }

  let p = _hb_gpu_apply_minv (m, renderCoord - c_r);
  var ang = atan2 (p.y, p.x) / 3.14159265358979;
  if (ang < 0.0) { ang = ang + 2.0; }
  var t = (ang - a0) / span;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (hb_gpu_atlas, grad_base + 2, stop_count, t, foreground);
}

fn _hb_gpu_composite (src: vec4f, dst: vec4f, mode_in: i32) -> vec4f
{
  var r = src + dst * (1.0 - src.a);  /* SRC_OVER default */

  /* Mode numbers match hb_paint_composite_mode_t.  Approximate
   * unsupported modes with the nearest Porter-Duff mode we do
   * implement; DIFFERENCE / EXCLUSION / HSL_* still fall through to
   * SRC_OVER below. */
  var mode = mode_in;
  if      (mode == 14 || mode == 18 || mode == 19) { mode = 23; } /* OVERLAY / COLOR_BURN / HARD_LIGHT -> MULTIPLY */
  else if (mode == 17 || mode == 20)               { mode = 13; } /* COLOR_DODGE / SOFT_LIGHT -> SCREEN */

  if      (mode == 0)  { r = vec4f (0.0); }                    /* CLEAR */
  else if (mode == 1)  { r = src; }                            /* SRC */
  else if (mode == 2)  { r = dst; }                            /* DST */
  else if (mode == 4)  { r = dst + src * (1.0 - dst.a); }      /* DST_OVER */
  else if (mode == 5)  { r = src * dst.a; }                    /* SRC_IN */
  else if (mode == 6)  { r = dst * src.a; }                    /* DST_IN */
  else if (mode == 7)  { r = src * (1.0 - dst.a); }            /* SRC_OUT */
  else if (mode == 8)  { r = dst * (1.0 - src.a); }            /* DST_OUT */
  else if (mode == 9)  { r = src * dst.a + dst * (1.0 - src.a); }  /* SRC_ATOP */
  else if (mode == 10) { r = dst * src.a + src * (1.0 - dst.a); }  /* DST_ATOP */
  else if (mode == 11) { r = src * (1.0 - dst.a) + dst * (1.0 - src.a); }  /* XOR */
  else if (mode == 12) { r = min (src + dst, vec4f (1.0)); }   /* PLUS */
  else if (mode == 13) {                                       /* SCREEN (premul) */
    r = vec4f (src.rgb + dst.rgb - src.rgb * dst.rgb,
               src.a + dst.a - src.a * dst.a);
  }
  else if (mode == 15) {                                       /* DARKEN */
    r = vec4f (min (src.rgb * dst.a, dst.rgb * src.a)
             + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a),
               src.a + dst.a - src.a * dst.a);
  }
  else if (mode == 16) {                                       /* LIGHTEN */
    r = vec4f (max (src.rgb * dst.a, dst.rgb * src.a)
             + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a),
               src.a + dst.a - src.a * dst.a);
  }
  else if (mode == 23) {                                       /* MULTIPLY (premul) */
    r = vec4f (src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a)
             + src.rgb * dst.rgb,
               src.a + dst.a - src.a * dst.a);
  }
  /* SRC_OVER (3) and DIFFERENCE / EXCLUSION / HSL_* (21, 22, 24-27)
   * fall through to the SRC_OVER default. */
  return r;
}

/* Wrap _hb_gpu_slug with a sub-glyph extents bail-out.  Many
 * paint layers cover a small region of the outer glyph quad; for
 * fragments outside the layer's bbox (with an AA + MSAA-spread
 * margin) the slug coverage is exactly 0, so we can skip the
 * band/curve walk entirely. */
fn _hb_gpu_slug_clipped (renderCoord: vec2f, pixelsPerEm: vec2f, glyphLoc_: u32,
                         hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> f32
{
  let header0 = hb_gpu_fetch (hb_gpu_atlas, i32 (glyphLoc_));
  let ext = vec4f (header0) * HB_GPU_INV_UNITS;
  let margin = 2.0 / pixelsPerEm;
  if (any (renderCoord < ext.xy - margin) ||
      any (renderCoord > ext.zw + margin)) {
    return 0.0;
  }
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_, hb_gpu_atlas);
}

/* Combine slug coverages from all clip outlines on the layer.
 * Factored so the shader has one set of inlined slug walks
 * instead of two (one per LAYER op type).  flags bits: 0x100 =
 * HAS_CLIP2; 0x200 = HAS_CLIP3 (HAS_CLIP3 implies HAS_CLIP2). */
fn _hb_gpu_layer_coverage (renderCoord: vec2f, pixelsPerEm: vec2f,
                           base: i32, flags: i32,
                           clip1_payload: i32, clip2_payload: i32, clip3_payload: i32,
                           hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>) -> f32
{
  var cov = _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
                                       u32 (base + clip1_payload), hb_gpu_atlas);
  if ((flags & 0x100) != 0) {
    cov = cov * _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
                                          u32 (base + clip2_payload), hb_gpu_atlas);
    if ((flags & 0x200) != 0) {
      cov = cov * _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
                                            u32 (base + clip3_payload), hb_gpu_atlas);
    }
  }
  return cov;
}

const HB_GPU_PAINT_GROUP_DEPTH: i32 = 4;

fn hb_gpu_paint (renderCoord: vec2f, glyphLoc_: u32, foreground: vec4f,
                 hb_gpu_atlas: ptr<storage, array<vec4<i32>>, read>,
                 coverage: ptr<function, f32>) -> vec4f
{
  /* Compute pixelsPerEm once here at uniform control flow.  WGSL
   * rejects fwidth inside a loop-conditional branch, so we call
   * _hb_gpu_slug (the MSAA-aware implementation that takes a
   * pre-computed pixelsPerEm) instead of the top-level
   * hb_gpu_draw() which would re-call fwidth. */
  let emsPerPixel = fwidth (renderCoord);
  let pixelsPerEm = 1.0 / emsPerPixel;

  let base    = i32 (glyphLoc_);
  let h0      = hb_gpu_fetch (hb_gpu_atlas, base);     // (num_ops, _, _, _)
  let h2      = hb_gpu_fetch (hb_gpu_atlas, base + 2); // (ops_offset, _, _, _)
  let num_ops = h0.r;
  var cursor  = base + h2.r;

  var acc = vec4f (0.0);
  var group_stack: array<vec4f, HB_GPU_PAINT_GROUP_DEPTH>;
  var sp: i32 = 0;
  *coverage = 0.0;

  for (var i: i32 = 0; i < num_ops; i = i + 1)
  {
    let op      = hb_gpu_fetch (hb_gpu_atlas, cursor);
    let op_type = op.r;
    let aux     = op.g;
    let payload = (op.b << 16) | (op.a & 0xffff);

    if (op_type == 0) {  // LAYER_SOLID
      // texel 1: (clip2_hi, clip2_lo, clip3_hi, clip3_lo) -- valid
      //          per HAS_CLIP2 / HAS_CLIP3 flag bits.
      // texel 2: RGBA as signed Q15.
      let op2 = hb_gpu_fetch (hb_gpu_atlas, cursor + 1);
      let clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      let clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      let ct = hb_gpu_fetch (hb_gpu_atlas, cursor + 2);
      var col: vec4f;
      if ((aux & 1) != 0) {
        col = vec4f (foreground.rgb, foreground.a * (f32 (ct.a) / 32767.0));
      } else {
        col = vec4f (ct) / 32767.0;
      }

      let cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
                                        base, aux,
                                        payload, clip2_payload, clip3_payload,
                                        hb_gpu_atlas);
      *coverage = max (*coverage, cov);
      let src = vec4f (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor = cursor + 3;
    } else if (op_type == 1) {  // LAYER_GRADIENT
      let op2 = hb_gpu_fetch (hb_gpu_atlas, cursor + 1);
      let clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      let clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      let op3 = hb_gpu_fetch (hb_gpu_atlas, cursor + 2);
      let grad_payload = (op3.r << 16) | (op3.g & 0xffff);
      let extend = op3.b;
      let stop_count = op3.a;
      let subtype = aux & 0xff;

      var col = vec4f (0.0);
      if (subtype == 0) {
        col = _hb_gpu_sample_linear (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground,
                                     hb_gpu_atlas);
      } else if (subtype == 1) {
        col = _hb_gpu_sample_radial (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground,
                                     hb_gpu_atlas);
      } else if (subtype == 2) {
        col = _hb_gpu_sample_sweep  (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground,
                                     hb_gpu_atlas);
      }

      let cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
                                        base, aux,
                                        payload, clip2_payload, clip3_payload,
                                        hb_gpu_atlas);
      *coverage = max (*coverage, cov);
      let src = vec4f (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor = cursor + 3;
    } else if (op_type == 2) {  // PUSH_GROUP
      if (sp < HB_GPU_PAINT_GROUP_DEPTH) {
        group_stack[sp] = acc;
        sp = sp + 1;
      }
      acc = vec4f (0.0);
      cursor = cursor + 1;
    } else if (op_type == 3) {  // POP_GROUP
      if (sp > 0) {
        sp = sp - 1;
        let src = acc;
        let dst = group_stack[sp];
        acc = _hb_gpu_composite (src, dst, aux);
      }
      cursor = cursor + 1;
    } else {
      break;
    }
  }

  return acc;
}
       /*
 * Copyright (C) 2026  Behdad Esfahbod
 *
 *  This is part of HarfBuzz, a text shaping library.
 *
 * Permission is hereby granted, without written agreement and without
 * license or royalty fees, to use, copy, modify, and distribute this
 * software and its documentation for any purpose, provided that the
 * above copyright notice and the following two paragraphs appear in
 * all copies of this software.
 *
 * IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE TO ANY PARTY FOR
 * DIRECT, INDIRECT, SPECIAL, INCIDENTAL, OR CONSEQUENTIAL DAMAGES
 * ARISING OUT OF THE USE OF THIS SOFTWARE AND ITS DOCUMENTATION, EVEN
 * IF THE COPYRIGHT HOLDER HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
 * DAMAGE.
 *
 * THE COPYRIGHT HOLDER SPECIFICALLY DISCLAIMS ANY WARRANTIES, INCLUDING,
 * BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
 * FITNESS FOR A PARTICULAR PURPOSE.  THE SOFTWARE PROVIDED HEREUNDER IS
 * ON AN "AS IS" BASIS, AND THE COPYRIGHT HOLDER HAS NO OBLIGATION TO
 * PROVIDE MAINTENANCE, SUPPORT, UPDATES, ENHANCEMENTS, OR MODIFICATIONS.
 */


/* Paint-renderer fragment shader (HLSL).
 *
 * Assumes the shared fragment helpers (hb-gpu-fragment.hlsl) and
 * the draw-renderer fragment helpers (hb-gpu-draw-fragment.hlsl)
 * are prepended to this source.
 */


float4 _hb_gpu_stop_color (int stops_base, int i, float4 foreground,
			   out float offset)
{
  int4 a = hb_gpu_fetch (stops_base + i * 2);
  offset = (float) a.r / 32767.0;
  int4 b = hb_gpu_fetch (stops_base + i * 2 + 1);
  if ((a.g & 1) != 0)
    return float4 (foreground.rgb, foreground.a * ((float) b.a / 32767.0));
  return (float4) b / 32767.0;
}

float _hb_gpu_extend_t (float t, int extend)
{
  if (extend == 1) return t - floor (t);
  if (extend == 2) {
    float u = t - 2.0 * floor (t * 0.5);
    return u > 1.0 ? 2.0 - u : u;
  }
  return clamp (t, 0.0, 1.0);
}

float4 _hb_gpu_eval_stops (int stops_base, int stop_count, float t, float4 foreground)
{
  float off_prev;
  float4 col_prev = _hb_gpu_stop_color (stops_base, 0, foreground, off_prev);
  if (t <= off_prev) return col_prev;
  for (int i = 1; i < stop_count; i++)
  {
    float off;
    float4 col = _hb_gpu_stop_color (stops_base, i, foreground, off);
    if (t <= off)
    {
      float span = off - off_prev;
      float f = span > 1e-6 ? (t - off_prev) / span : 0.0;
      float4 p0 = float4 (col_prev.rgb * col_prev.a, col_prev.a);
      float4 p1 = float4 (col.rgb * col.a, col.a);
      float4 pm = lerp (p0, p1, f);
      return pm.a > 1e-6 ? float4 (pm.rgb / pm.a, pm.a) : float4 (0.0, 0.0, 0.0, 0.0);
    }
    col_prev = col;
    off_prev = off;
  }
  return col_prev;
}

/* Apply the stored 2x2 M^-1 (row-major i16 Q10) to a vector. */
float2 _hb_gpu_apply_minv (int4 m, float2 v)
{
  float4 mf = (float4) m * (1.0 / 1024.0);
  return float2 (mf.x * v.x + mf.y * v.y,
		 mf.z * v.x + mf.w * v.y);
}

float4 _hb_gpu_sample_linear (float2 renderCoord, int grad_base,
			      int stop_count, int extend, float4 foreground)
{
  int4 t0 = hb_gpu_fetch (grad_base);
  int4 m  = hb_gpu_fetch (grad_base + 1);
  float2 p0_r = float2 ((float) t0.r, (float) t0.g);
  float2 d    = float2 ((float) t0.b, (float) t0.a);
  float denom = dot (d, d);
  if (denom < 1e-6) return float4 (0.0, 0.0, 0.0, 0.0);
  float2 p = _hb_gpu_apply_minv (m, renderCoord - p0_r);
  float t = dot (p, d) / denom;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (grad_base + 2, stop_count, t, foreground);
}

float4 _hb_gpu_sample_radial (float2 renderCoord, int grad_base,
			      int stop_count, int extend, float4 foreground)
{
  int4 t0 = hb_gpu_fetch (grad_base);
  int4 t1 = hb_gpu_fetch (grad_base + 1);
  int4 m  = hb_gpu_fetch (grad_base + 2);
  float2 c0_r = float2 ((float) t0.r, (float) t0.g);
  float2 cd   = float2 ((float) t0.b, (float) t0.a);
  float r0 = (float) t1.r;
  float r1 = (float) t1.g;

  float dr = r1 - r0;
  float2 p  = _hb_gpu_apply_minv (m, renderCoord - c0_r);

  float A = dot (cd, cd) - dr * dr;
  float B = -2.0 * (dot (p, cd) + r0 * dr);
  float C = dot (p, p) - r0 * r0;

  float t;
  if (abs (A) > 1e-6)
  {
    float disc = B * B - 4.0 * A * C;
    if (disc < 0.0) return float4 (0.0, 0.0, 0.0, 0.0);
    float sq = sqrt (disc);
    float t1r = (-B + sq) / (2.0 * A);
    float t2r = (-B - sq) / (2.0 * A);
    t = (r0 + t1r * dr >= 0.0) ? t1r : t2r;
  }
  else
  {
    if (abs (B) < 1e-6) return float4 (0.0, 0.0, 0.0, 0.0);
    t = -C / B;
  }
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (grad_base + 3, stop_count, t, foreground);
}

float4 _hb_gpu_sample_sweep (float2 renderCoord, int grad_base,
			     int stop_count, int extend, float4 foreground)
{
  int4 t0 = hb_gpu_fetch (grad_base);
  int4 m  = hb_gpu_fetch (grad_base + 1);
  float2 c_r = float2 ((float) t0.r, (float) t0.g);
  float a0 = (float) t0.b / 16384.0;
  float a1 = (float) t0.a / 16384.0;
  float span = a1 - a0;
  if (abs (span) < 1e-6) return float4 (0.0, 0.0, 0.0, 0.0);

  float2 p = _hb_gpu_apply_minv (m, renderCoord - c_r);
  float ang = atan2 (p.y, p.x) / 3.14159265358979;
  if (ang < 0.0) ang += 2.0;
  float t = (ang - a0) / span;
  t = _hb_gpu_extend_t (t, extend);
  return _hb_gpu_eval_stops (grad_base + 2, stop_count, t, foreground);
}

float4 _hb_gpu_composite (float4 src, float4 dst, int mode)
{
  float4 r = src + dst * (1.0 - src.a);  /* SRC_OVER default */

  /* Mode numbers match hb_paint_composite_mode_t.  Approximate
   * unsupported modes with the nearest Porter-Duff mode we do
   * implement; DIFFERENCE / EXCLUSION / HSL_* still fall through to
   * SRC_OVER below. */
  if      (mode == 14 || mode == 18 || mode == 19) mode = 23; /* OVERLAY / COLOR_BURN / HARD_LIGHT -> MULTIPLY */
  else if (mode == 17 || mode == 20)               mode = 13; /* COLOR_DODGE / SOFT_LIGHT -> SCREEN */

  if      (mode == 0)  r = float4 (0.0, 0.0, 0.0, 0.0);        /* CLEAR */
  else if (mode == 1)  r = src;                                /* SRC */
  else if (mode == 2)  r = dst;                                /* DST */
  else if (mode == 4)  r = dst + src * (1.0 - dst.a);          /* DST_OVER */
  else if (mode == 5)  r = src * dst.a;                        /* SRC_IN */
  else if (mode == 6)  r = dst * src.a;                        /* DST_IN */
  else if (mode == 7)  r = src * (1.0 - dst.a);                /* SRC_OUT */
  else if (mode == 8)  r = dst * (1.0 - src.a);                /* DST_OUT */
  else if (mode == 9)  r = src * dst.a + dst * (1.0 - src.a);  /* SRC_ATOP */
  else if (mode == 10) r = dst * src.a + src * (1.0 - dst.a);  /* DST_ATOP */
  else if (mode == 11) r = src * (1.0 - dst.a) + dst * (1.0 - src.a);  /* XOR */
  else if (mode == 12) r = min (src + dst, float4 (1.0, 1.0, 1.0, 1.0));  /* PLUS */
  else if (mode == 13) {                                       /* SCREEN (premul) */
    r.rgb = src.rgb + dst.rgb - src.rgb * dst.rgb;
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 15) {                                       /* DARKEN */
    r.rgb = min (src.rgb * dst.a, dst.rgb * src.a)
          + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a);
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 16) {                                       /* LIGHTEN */
    r.rgb = max (src.rgb * dst.a, dst.rgb * src.a)
          + src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a);
    r.a = src.a + dst.a - src.a * dst.a;
  }
  else if (mode == 23) {                                       /* MULTIPLY (premul) */
    r.rgb = src.rgb * (1.0 - dst.a) + dst.rgb * (1.0 - src.a)
          + src.rgb * dst.rgb;
    r.a = src.a + dst.a - src.a * dst.a;
  }
  /* SRC_OVER (3) and DIFFERENCE / EXCLUSION / HSL_* (21, 22, 24-27)
   * fall through to the SRC_OVER default. */
  return r;
}

/* Wrap _hb_gpu_slug with a sub-glyph extents bail-out.  Many
 * paint layers cover a small region of the outer glyph quad; for
 * fragments outside the layer's bbox (with an AA + MSAA-spread
 * margin) the slug coverage is exactly 0, so we can skip the
 * band/curve walk entirely. */
float _hb_gpu_slug_clipped (float2 renderCoord, float2 pixelsPerEm, uint glyphLoc_)
{
  int4 header0 = hb_gpu_fetch ((int) glyphLoc_);
  float4 ext = (float4) header0 * HB_GPU_INV_UNITS;
  float2 margin = 2.0 / pixelsPerEm;
  if (any (renderCoord < ext.xy - margin) ||
      any (renderCoord > ext.zw + margin))
    return 0.0;
  return _hb_gpu_slug (renderCoord, pixelsPerEm, glyphLoc_);
}

/* Combine slug coverages from all clip outlines on the layer.
 * Factored so the shader has one set of inlined slug walks
 * instead of two (one per LAYER op type).  flags bits: 0x100 =
 * HAS_CLIP2; 0x200 = HAS_CLIP3 (HAS_CLIP3 implies HAS_CLIP2). */
float _hb_gpu_layer_coverage (float2 renderCoord, float2 pixelsPerEm,
			      int base, int flags,
			      int clip1_payload, int clip2_payload, int clip3_payload)
{
  float cov = _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				    (uint) (base + clip1_payload));
  if ((flags & 0x100) != 0)
  {
    cov *= _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				 (uint) (base + clip2_payload));
    if ((flags & 0x200) != 0)
      cov *= _hb_gpu_slug_clipped (renderCoord, pixelsPerEm,
				   (uint) (base + clip3_payload));
  }
  return cov;
}

#define HB_GPU_PAINT_GROUP_DEPTH 4

float4 hb_gpu_paint (float2 renderCoord, uint glyphLoc_, float4 foreground,
		     out float coverage)
{
  /* fwidth once, at uniform control flow. */
  float2 pixelsPerEm = 1.0 / fwidth (renderCoord);

  int base    = (int) glyphLoc_;
  int4 h0     = hb_gpu_fetch (base);       /* (num_ops, _, _, _) */
  int4 h2     = hb_gpu_fetch (base + 2);   /* (ops_offset, _, _, _) */
  int num_ops = h0.r;
  int cursor  = base + h2.r;

  float4 acc = float4 (0.0, 0.0, 0.0, 0.0);
  float4 group_stack[HB_GPU_PAINT_GROUP_DEPTH];
  coverage = 0.0;
  int sp = 0;

  for (int i = 0; i < num_ops; i++)
  {
    int4 op     = hb_gpu_fetch (cursor);
    int op_type = op.r;
    int aux     = op.g;
    int payload = (op.b << 16) | (op.a & 0xffff);

    if (op_type == 0)  /* LAYER_SOLID */
    {
      /* texel 1: (clip2_hi, clip2_lo, clip3_hi, clip3_lo) -- valid
       *           per HAS_CLIP2 / HAS_CLIP3 flag bits.
       * texel 2: RGBA as signed Q15. */
      int4 op2 = hb_gpu_fetch (cursor + 1);
      int clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      int clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      int4 ct = hb_gpu_fetch (cursor + 2);
      float4 col = ((aux & 1) != 0)
		 ? float4 (foreground.rgb, foreground.a * ((float) ct.a / 32767.0))
		 : (float4) ct / 32767.0;

      float cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
					  base, aux,
					  payload, clip2_payload, clip3_payload);
      coverage = max (coverage, cov);
      float4 src = float4 (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor += 3;
    }
    else if (op_type == 1)  /* LAYER_GRADIENT */
    {
      int4 op2 = hb_gpu_fetch (cursor + 1);
      int clip2_payload = (op2.r << 16) | (op2.g & 0xffff);
      int clip3_payload = (op2.b << 16) | (op2.a & 0xffff);
      int4 op3 = hb_gpu_fetch (cursor + 2);
      int grad_payload = (op3.r << 16) | (op3.g & 0xffff);
      int extend       = op3.b;
      int stop_count   = op3.a;
      int subtype      = aux & 0xff;

      float4 col = float4 (0.0, 0.0, 0.0, 0.0);
      if (subtype == 0)
        col = _hb_gpu_sample_linear (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground);
      else if (subtype == 1)
        col = _hb_gpu_sample_radial (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground);
      else if (subtype == 2)
        col = _hb_gpu_sample_sweep  (renderCoord,
                                     base + grad_payload,
                                     stop_count, extend, foreground);

      float cov = _hb_gpu_layer_coverage (renderCoord, pixelsPerEm,
					  base, aux,
					  payload, clip2_payload, clip3_payload);
      coverage = max (coverage, cov);
      float4 src = float4 (col.rgb * col.a, col.a) * cov;
      acc = src + acc * (1.0 - src.a);

      cursor += 3;
    }
    else if (op_type == 2)
    {
      if (sp < HB_GPU_PAINT_GROUP_DEPTH) {
        group_stack[sp] = acc;
        sp++;
      }
      acc = float4 (0.0, 0.0, 0.0, 0.0);
      cursor += 1;
    }
    else if (op_type == 3)
    {
      if (sp > 0) {
        sp--;
        float4 src = acc;
        float4 dst = group_stack[sp];
        acc = _hb_gpu_composite (src, dst, aux);
      }
      cursor += 1;
    }
    else
    {
      break;
    }
  }

  return acc;
}
          @-q=      ?                  0C        A      @          @      P?              ?      ?      ?      ?                                   UUUUUU?UUUUUU?      ?      ?                                     ?          ?       F   ̼+  D>      @  F;  P   l  |  |  |L  |      ,  L  L  L@  d  l    <    L  0  D  ,|    L    <    4  <d  l      l  <P        <H  ,    <      	  ,4	  \	  	  	  	  	  	  	  
  
  l
  <l    ,T    'h  (|  l(  (  )  )  <+  \+8  +`  -  -  -  -  \.  l.  .  ,/<  /h  0  8  L8(  8L  9d         zR x  4         KCDED
Ae 4   T      KCDEDr
Ac 4      `   KCDEDz
Ae 4      (   KCDEDr
Ae              (        AaAP
E     <      PGfC   \  @           p  L    QG^
J        (    EVAw
E      Z   EVA
C      @   E[A
H       ܸ    ECA[
D   $  X'    ECAX,   D  h   QJ
AjKH     t  ؼi            4x    AMi       8       4        AhFU
O^
BJ
F                4y            y       (   0  e   EM
G6
A  $   \  PR    iCX
LA   $     R    iCX
LA   ,     2   OKC
C`HH   ,     *   KK@
J\DH   $     m    ECAL
KL,   4  4   ACBED
F (   d  (E   ACBEF3  4     L   MIl
Na
AH   d        JCHGKIO^HDLJH   8   0  /   ACNN
HGcQDB
EB  $   l      AGC
D  (     S    NCBEDsA 4         OF{Hu
CF
AM $     R    iCX
LA                   4  xt    EMa     T             h  )   dCA         ECF   $         aC\
HIG       h   ECEW
D      T            P       (      \    ECDHg
A      L  
          `  W          t     ECOJY 
NXD)?LPz
DDF5]gH
DDH
DDEF
SLEfDD8     P   ECk+TAq
L
EVD (   H      ECBD^
Lj l   t  h   ACpJGKG?DDDNDGD[J
DLDE|P\     R   ECgdGPDDTJ
tWH\DK
EDLPL    D	     ECgjPLeDDDDTBDDDM
KDEoDDHT x   	     EC|[T]DDDTADDDD\
CDLKj
DDDDKT   H
     ECbw\TDDDTDHp
DDDDEkDDDK
IDL`DDHT   
             
  E    ECBDv                ,  0    EM     L          ,   `  1   dCDHBBA            ECF   $         aC\
HIG          ECEW
D      0            ,          $  (       $   8  4V    ECA_
He    `  l          t  h       (     t    ECFHv
A (         ECFHv
A (     \ 	   ECDH
B   p     @!&   ECKaXRDDh
IPN
DDJtDDL
DDEP
DDE\L      '?    ECAp      (E    ECBDv      H(                (                             (                   0      GNU                 	                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                           1      0             U	             f	             p	             z	              0                          p                          x                   o                 8                   
       	                           X                   	                            o           o          o           o          $       `       #              %                                      	                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         GCC: (GNU) 16.2.1 20260810  libharfbuzz-gpu.so.0.61440.0.debug    .shstrtab .note.gnu.build-id .gnu.hash .dynsym .dynstr .gnu.version .gnu.version_r .rela.dyn .relr.dyn .init .text .fini .rodata .eh_frame_hdr .eh_frame .sframe .note.gnu.property .init_array .fini_array .dynamic .got .data .bss .comment .gnu_debuglink                                                                                                $                                 o                                               (                         x	                          0             8      8      	                             8   o                                               E   o                                               T             X      X                                 ^             `       `                                    h              0       0                                    n             @0      @0      Ì              @               t                                                       z                                                                    ԃ     ԃ                                               `     `                                     o       P     P     4                                                     @                                           p     p                                               x     x                                                                                                           p                                                                                                                                               0                                                                       $     (                                                    L                                   