From 523f4f62148129f76a90e417b01117b5545d7d92 Mon Sep 17 00:00:00 2001 From: badbl0cks <4161747+badbl0cks@users.noreply.github.com> Date: Fri, 7 Aug 2026 17:18:34 -0700 Subject: [PATCH] Add ats-tool --- bun.lockb | Bin 212693 -> 226596 bytes package.json | 3 + public/theme.css | 91 ++++ src/layouts/BaseLayout.astro | 5 +- src/lib/RateLimiter.ts | 70 +++ src/lib/ats/dates.ts | 158 +++++++ src/lib/ats/diagnostics.ts | 605 +++++++++++++++++++++++++ src/lib/ats/extractDocx.ts | 170 +++++++ src/lib/ats/extractPdf.ts | 205 +++++++++ src/lib/ats/fields.ts | 304 +++++++++++++ src/lib/ats/groupLines.ts | 172 +++++++ src/lib/ats/index.ts | 281 ++++++++++++ src/lib/ats/limits.ts | 20 + src/lib/ats/readingOrder.ts | 222 +++++++++ src/lib/ats/sections.ts | 212 +++++++++ src/lib/ats/subsections.ts | 103 +++++ src/lib/ats/types.ts | 214 +++++++++ src/pages/index.astro | 2 +- src/pages/tools.astro | 33 +- src/pages/tools/ats.astro | 846 +++++++++++++++++++++++++++++++++++ 20 files changed, 3704 insertions(+), 12 deletions(-) create mode 100644 src/lib/RateLimiter.ts create mode 100644 src/lib/ats/dates.ts create mode 100644 src/lib/ats/diagnostics.ts create mode 100644 src/lib/ats/extractDocx.ts create mode 100644 src/lib/ats/extractPdf.ts create mode 100644 src/lib/ats/fields.ts create mode 100644 src/lib/ats/groupLines.ts create mode 100644 src/lib/ats/index.ts create mode 100644 src/lib/ats/limits.ts create mode 100644 src/lib/ats/readingOrder.ts create mode 100644 src/lib/ats/sections.ts create mode 100644 src/lib/ats/subsections.ts create mode 100644 src/lib/ats/types.ts create mode 100644 src/pages/tools/ats.astro diff --git a/bun.lockb b/bun.lockb index 03e6e41e4a5e4f1e474d8debc35bda2a7c77b5cc..e7256a3705d8b2219ff9039c147545477e2442d5 100755 GIT binary patch delta 47887 zcmeFa2Urx@+BMo$(n_PKfB_H{Fo6mZlprb~+6pKN1_VV#Vhaco6aj58f*DKQX2l4m z5hLb7$1#sO<~W8iM;sHzf4x=J!kmB3eBb}wbMJlbbEh6wul=t5?i{MRx{JQ}rN-(f zhO_N$mu6ZyC-r$>>!|(YZP9D5e9o_7d_b`#r@(QM)9ewvcf#D%DhfKB@`D>|Gdl93 zQSria3WZ9c$bJdQ0DNEO>oOZ4K`rRxkxLVBIM@p84YmYZ$;^Wr!~Pxqjle(4Tm)_i zdpfZefg}Vh!2QAI;KbCVc+D_{;%PO7q8{uBeTBjdd>VEQZ~*MO;H$8W!HLmPsS&XX zMOt(ss;E$$LhX&9M@FTjCc{n$Pe@2ijaAgBE;Wdk`88UntqsTc#3V6cVp6ImafCt< znHCX8!tj&~4aKZSsnlX8LxrLaxGA_MxSq`45RW{w4Pa9RH6vo9k~OI*nw0S5KW z%5Se0uax03Fii@wY0?Y_!za5B7?sUViAoq1g~99&o4UHA%)gjOJ?8?Z3izOO>bc02 zs8n=>B0^)Q8G(7R1MxJ63Ncfu$C8pYDG9?<6z!MD5osed(VD18#i(daJdIA%CJF_H zGdl%L6}yIRpe{#8r^R5hJ+Ky%`R3Z8NKWOY#A@O+Nao#KDyJEk%E>@7s_*4yQvIA| zy&jmR)Mx0_(QY=9y;SDcGABpHM5QC&8L%-4vR}fVs#T+<*lun1CtGPqGe&E~MbM;` z)FG$A)ZiUp>KH>XjYv5XqVKXFN*QEdv6C|HXfOFWfT@5Z(9yQ+y=ob!U_A$E zU|zN%)}mzyPyuxvrHu37NFz2$?y!q;L8WrR8^BbNh=|0r5g4O<_|e=S>n!!aNHBGD zu-vo5Q|!isC&VjsYat~(H5IE%@va>zMB9gMGH1C+2^FqVMPgvn6j}qDy6zGZSb|Hy zG$)sUY3hv!(+~`kDj^}#d)-nW%{x}JyBL-R0|`coH)OB7B~BT_Y~ny8eG zaHtENb&%#jp3GV>b)}}G)MNM2Y;lexMq-2&irL7FT0Y4{c%s)_>(N>27#lDZQwK~V z$IJR7Z>ioFz@#4lQ^#%qlmC1$RbwKUIyebTexYFUQ_FT6Fb$ZQtXBmiy*B%`mz3Z( zm@+se+dILO!74BnG!sk(WytZ7a(r*u?gXavb3CQBaL-q2XLL%`DD+v12ANR5L`Q0p z6*#1k$J9zrjE|2@96iFLr*wufm-$N%X+&;;X@vHHt-%YySgzTlWjh2+BkcyZ0-J!b zezM>BOX)9xEn)8hQ$9?#Gr<;G2+;^o%ld+;1xbl1sSznDiuHY@DX;=eH6IZ^B2j}u z9EldzfPJ)|G&DbeX>H5})8ZKorl}MRrh+}d7GP^IEmox*|D>-%p{4|95umxb8%$%A zg|4FQMt7=f6DrxU&eU-VMW;d1lyZ={#ULT5S|@|t!IB(oA}p%bT$>{2lByXoCMrED zf;b{xlaw+lhR!02--b%fvJaOy6-*Pg8JJ2RmY5hnEIcJDGBF`7RTH0*G)#)uEdX?p z-RRix)Jw>R{9%jvlAf9zC3xzaX#?asB-xEji%N@9JV*DD?`^OlILR(vlb}ge6h=vt zCUIC=4Ard_(p5)%LUMv80dWc8F`5XR|E@+$UEdyFG#46!sRvI%r~Dct9SzN$dm$r@Q#1X0C5mAbRuxYH7agtwjJT@Yn$Pd837W7C>G>XE- zL)55noW>OqiHUJ^o?kEL9}%CDPVq}%*QAYe4g!?n6fm`@I6*2f6*lT7&i=F{T&9hb zCQDr~O`dXSG--YV)7&@&CVj1}&j8b8qKg`u{edz&fURKH23ykXeuY7?1pf*q#~;A7 zd}e^D3rB!yS@e?G5lnt{!Bl~FBc%9iV6qQ_smptTsmmkLklNt>;F@5U45^{5!CGp0 zBLpbI5KNaPoin8jx*!4mvK{b3L(@4+YSF{73I)!C**}6?fEUPiD!3`^USL|^&A=_e zpOMZM{41F3jo{|seN=Di;+Gf?sz`Wzl4eBI@RTj%rLj~@5MCQJ*XB)>IwBoR^^2P% zxeo%@f*lE+>gPRKith`X8tfzcqo1S>m?CW>DVo%%gzzNAzixzlu9R;|xHv=TRELxb zfmJus>Mn3DTvR~IZQaV>KAKs)$MrM=xNWwv!XI2Gb4^86|B6!EmAi2hNo;I0L3>hb!8Yk?0|#`SLWKC(Wu` zVA_n2f@v=;2GeP5vCNafbzvuiX;g=TjcJ|qL4Y#WU3>nFf~X=}!Q}X4p_I^fk))57 zTfP?Yw4?L}Q$>3$mg2j?ru4eAhA$dIJun92PD#s_O4I2screW~-G#v=2$Z4Ca;XIk z!IZHHn7X*u3W=jqBEsXtBcrgVr)iMkLFiN!!EPE?*^eaL`(b)nKaTL+G@OUBJ}gJJ(73 zhkK!ve-Lc+pEldNNJ_XEOzkiP(?}@5RL?g^KpBP)AB8I7EVn`G_*P(A#ui}er)Zo9 z?9!4o@Uzo6x{VO88{vNd6|hD8S}-+q*CuIZ%!XZ$w&HOJP)*~&l;MCMqylv3h^T}F z@$E#%Ez(A`4qO|4Hejl#5B!N8z?A-8u~fimFio*gzy3pu>0pTQ`>n2cZkawb^21B16~EeK}!^0Uw%Bn*LrcW|Djty3;ow| z-PdsY#w>}>oj%TY&9XU~091t1c8sR9<#?Sfcw2WfJH=sP zp~I`vFyi1PQp5}S&D*HVV0J_-CveuDsx?)xKYzCj^c0Hl?^YbFA>XYNf3j?q3PqS4 z=cNy0;CIJ6f3iBD79ro`=KaZf0W0`>T-Ry}MIW8TFQ_I=wN@)ReTAZ@FvHqMnT1dv z3aQQ@?uRJ{Z)lK)~HbRn>zTnVQUsVt5iMq-_unO@s>LXja>P9u10ZCouOhp#KV9pYqT4x z8mevxtB7T)v8;NNDxgf&d{|vzVL;n@DzA|xGz{@kHNy5l(?BVdZS>>^)f6&Dbkl?8 zCK$T==wCp{1+hZl_7-Tck=S4Z?7d!MR_GKXKUh>zU4s=uS(vLT6Dl>aijd#dQ?(UV zd&!#Qsd@&hhnPNRlUHr*KG^lXw@6R8Gu%~~1hu`;v8j*QdW2{gzi0ow z6{e|?Ew)Y?3E4%VF9a@YBUU*YF;wL$)Oyu+SRSx=;aqo5{;QGjq?uZ2fZfMg7~9;3 z-&Rk^Xs%X%!}b(Fg0g3Q^rRSCijZ1#dR<>w)IzOt!1+a*AO(q@%3-kjQncXJR*#Di zHnsLqnwcpS1H{B55sD#=zhWkq(9|3&LUht1B$aRmp&_DX-$2exIUXS?v-=21e(fxD zesd9$wBL0a@6%8$U{*u90RBlsVSQ`0vKy}F!^PYS5t3T>4xu2iP5l}RC+yU!yvEX$ z;Au+eAB9B&@O=u@ZmZI_R49DJ?hFs|T2^@LH5_vD8(6|Om{Rm){jnh57*ixciRd8(>4 z(jNCni z+Fp9f-mnbBrIye__EHui)KN@%A0g~gZGHG|Ho^%PwQ8e{w1TUO^OLic*R!~Iy1}9; zhK%*VgD`~OmQPorrV1leRD=4X>&n7B?l8D!d66 z4N6sUt9c140G8z0v5mAuu@hiW#=shkIB8QdbCAYdvf}<^?S~~*4%@2I7F$ejv2tS& z>V+IQv2%{Wq9K=>$DyBSXQ?9Au_r8QhGZ>;)fW~oPL48IwAo4}H^ZJv>8jFHQB8+M z+o?E_RlmZb5~Ot1kw}{4bVBF@i?&BS+WJ*_u-rsXv{YG2al%+{AKu7C$naJxbK&VJ z7)#ehpCH!D8$fghCPOqK^QrUrs!&HNje><;V$T%{*0u5v0+R z+FJ?>4Z&2Pef%vfYN0gFcG!7?VChTO2(Y?{>1doE%a&4%v%{_o0W*{0a$pUBCH3!B zSW-I7D`mq@IBy7>X8G{`orDwJ)vD_5>-2=!H~ygd~POb?VC?(4~)@)Vx< zs+HBeun2{*em;E9&O(NtTA7c2#Afa1qr8kzcQLB5j~Eq!)}zM|wF@CJO8E_;9-`)t z`bY`$5%LogUhg6%Y}{4n9)pnNzU{mA2_bB%$krDVNQ#|@kd)bVgrr(D=_Y0t(@mGz z4+u%_pAeGVy)ie$vX#>ilInOKA<3_?uP(C~U&>6i0RoMU8TGi)OIY5ps>$=HwovTq z$~*W8p+Ras)=yXoTH_}?2~w*rV$YDKH0_!e{!%uakkiIfHK-@%li2%Y?fa8ufmuSs zD0!yB8uZ;d3v1YS%L(f>^1C%3Rz=(gSfStJg0NmI%G(Mn_IsRpVCC|XVbMk+m3|gh zMd__ED=Xrr!s`Fs^ENDWBGO?h8rk%bx=HRO#Z3xOuA!_BFez?FDHih^oJE4dWF~ktU864U>gmB&t@ez_@^f-;E9SRoD zz=&;v&coRsp>YU_e#$!tNol=8bU80YD46{CHz9&SnA&Lez<<4B#TKDD02RwKySm`r z6tx{F+znH!dZ1HiadG1AKl{5y4Z1K$*fSVs=rC!(72;Y_hY5Z|)T%6qwCu4Z;Hq#- zm~ai^ZP`=W*GR@z{SxQQ21Lx&3OBh;$35NJQpqe0~# z4;4;8bPJa@9cc?14~wdV4IdlX&#E=YixSmJ=XkVF z^o>ENx9Cg5w+oik8FXp!5LOrArjw6Rs|0z*$lE~`!D^Kommv6!RI5r6NBf3CoUf`8 zQrSux6y;P{?SzD6A3kWLFg#hU+%}SycEi*zddajaAVwL3P(LxW521l#sD6qrHVL6V zqIMXeFfnAAstb)pD1tQpb}Hr1d#8!TtW48+d_yQkOcgx}Z`H-nNrZ-op_Zd{p*)17 z+#et$)yy+p7rP80Nqdh_u$VSr45pZP;WZl}FZ6(TCRUz>)sbeck#~mtu2}9h>D_EP zRA~(9LTNj!E@JP|Wc&b&E>xw9@UTp2x0bHj_rju5#ES#vCs^HtgiIfPV3sgEQ?0Bu zPN7I7K{*bg1Tpkf*1G6v#a=*UDMA`CvSl`IU5lZm2#pj&HFI=Y4nlFHsh%N(la{n) z8MPn(=b4}qu*LGD#tSFLsg(y2=`Dn5efa7Vgix*8sLKTDEK)`80YiEV%ukviJkhFE zhY(2}p(kFIf1My?WUKiN6NN({(@BCsj#`yANpg|)!eUr7uf;Qz@;xj+!7vANYqID( ze=;s35iQ;(;4e-V_Ka7n>Q9kGY*u!jsyJ9U&0%HGY(E7{+Ca*@ytrK9+61*x*W5o} z9vIUd2!3*|Fnpp~wHJ~0Vg-t4;Yv=nu#I9d*3Xmnrt0Ebw;_4L)JbZie26|splg2> z@qWe~sGY@ZXd`SgRoZK?rm*7PrME*QS>CsZ#w3M&Qk7zl_;E^(nWs;=7vsI&47h137LzVDl9BJ+0x9!Jo#>2 zgH;jNa~9^y_qY;RU0{hLrTPqub`|k*K^Z@r&QbJkYBfSMqeWfyJ1p9kai&Y~R9feg z7u|;mMX0r?(+PAkESfap3gHjs3)iNrRpk(+q_p=3&e3H{Rz57LD)bKKA}p$kGy#-? z?j8kA2Ul1W#|gM&p^AY;dCK?sV9^MO>GYPCQCD0Gnz=&z*=p4d zMA9Jhg4KFYzR^74#B8pCnFRrhHfGhC>RP|^cdd1iPBhvdWxYF-$V5m z(2Y7uIt(FSF|+}pu43pFLhdYZI%gtSx;rRs7ozgi$9&8}q2*k)Q2_*B)J9iT1G=13 zeS_*pst~UAGzwZIO+-c`v?0OxF3dzE$2=kyHEH}g< z9p=FdSWaR(iaQ32Mz)Wtsw&3<4-W%3lDdnd} zJ*nypi^{@n9Hh(mZqX&QlE3)C9 z%9pTQrJ+5$)IsI4M%NC^Qhvl5!Ec#bwdK2*x*v+Iu;4$ z1!~orB3*Y=2b-;zc3Sa`tx+5->PTH{m2@9ge-l(MsFn9RqvsIYK-A4bBbtT!K^vsY zl*+Ekl@J5Oo#YilG<~FtY?qC?F0z{JrALNvI^M@<8$#mRmusO<%T;*~YGy@84Bmtl z`rTBWhdGogE}V1mR61|QrONk6b~+5#>8rzCu`#kIJ)HTC?K)9Az%7uEO&KAJsR6g2g(~>7)Nv z>MdGZH#2q!kEjXpx8Ijb8sjPQKCt==hw zZd9wHc7C54xUGC{r|@K>+Q@vDbmG!YChprQXdP0t$r}6@&KD^_8;q^A8-7mD?uIA6}7si8H z9uQ7|MjQ}~cXYNj$2=y`@mEY8+W@F8Y}!#@*tbLfzk-%7Q)~g^)&LLK19T8mj#!&w zJa$i!+kxr$D>i_RnJpfF#Z(TaoOpn>VgWud^?+_bH6Q?>qcT%PdWq`)8B@C6a=QPW zJ){_lAV=(U;!&AN50Z6aS0E2)AcXCzFU0TC*OCWzV=)J8x#B@gI<{8vASN5zsdx~R zjU7}xh{?uQC?3RQW0w;TVzRMyi3c&+*sQdofPg4$7A=M1ub2{Ga}xcw$bNstG#1zl z#N+Q+36I@WK4pw`ARffDB^&~1FpmQ?t4_*%8chDb0CfBvliwN9>#v#0xs1gt7I0PO z>tITF1E7PL?3*&*0@F0SBlGX13uz^dwbY)cvibsC9ViFrAf{$~A_K?Yu?q2eD3VI4 z2Bvf>F#Z$Ms~0TQ$G;3Gl!c?XkSGAFD%e#nz)dcIm`2=7 zwuz}?oxwD8-M|#ponHP6CKq447=s7O{#pvqwwfjj{|~eF|JH)3O{3(7Rc7kJbm-#z z8zVaq8zEw)A|_N&Y!9n!6T1jJeWlk8ZTsRxeA@yF%({|?hWd=~z+N-lya=_NTmF$FKv z`lsVEEGp=l%r_|($6ql;-NFk^qT90H-!b{!mHjF+N%wT?|DNpdS4@ry!e_kB#g|Op~ycY*%K=xV5YklWiy4#FWlKwuz~y+Jeci9WS5%5fB9z*|9QH zg7(nq8gwL>*7GRYkC-wVEpxi86H~e@Ssx3g9-b`6SFj1xe~I9P{iSW_OOlE?vU_Ew z;sjZ*%%snibz-XKLfQTwu=FNj5i+2&|4KQ-b#jKpRKcyX{hw!fU!;twjO{Y-1XD|Q z$@Xs9-UGIPeo3}3gXyTu)FU^b)6BXj$Nv?J+nrW+cuj%msn%6^rZq}q6)_&Q)Jv!3inOeb+` zF!fLiN!N-2s(_8`NKCdZm@=}H^}l2Cx0n5hsbB{%4S_3|;@!abQ+VKo^7RBOX^ z1Gr?QSRBNZfj^iM_K)w{BwVVZj^`@5|!_Z&;~;`T_XH*f8?M0Bmarp7&NE;oDZ&p|1A8^Y`XgW=l;k) z_eW@gN%u+UI`MySUqnogbpPBR5qm=M&;5~q?vMQc@%~67ar*z&eG!V7yUq4e7N?Cc z8d%=cqTu|f&1qi(#@=k0H+<}Y(=!MCk#(=vYBo->-e5=fpVFS5AG>Ay^ILsJp8CUW z^UkOJH>-WVIUmkxbS6>y6!~kduei#h?Mip(wGV3laOdu@%oC>lKFz+Idun+rkFE80 z{c>vk^C_=){B&=_UH;zkJ&DJ)RrYNha&TEyi=M6A&zIL;IAuoGqR~-F{oa2#yQFumHJ6`tyBT>hu*+G)UccLa3o`P17qGq8-dpA; z&M%wh=vyP@*%!s2!xO7t8ajK@)81j(1BZ+GiVyVDEk5xcmYBDm!3E3C?d$d&w|7RH z;F6}#Tg)(K`;1y2vtJSN(0uga1*HCuXdt$VP5-=S_F$&AI4bVy6FC8L)VElE>S{sqv0E4>u_eg+6clpxX&;&hQ;a z&fa%vTljK@f3K#hQ72E1J5^`UsfF5$eb&6IzUXm`0e#-a;ocba1ieGn9eOR6#!D7jt$mv-n6C#iHPraX5=-B=GkDA>B7FzWD@!a;tKm0i6KrPcHWi1M~yjR-3 zo!l*IO7x4Z`4;UcU#*-meLqKc=$Z9hwd{spdaZ*e-}z|+$0bKOT#U(`9T4hh@?`0} zW=e-d<%yR$F&0ggn}=G@8ZuGkQ?+~R@eY%#d5svHcB0er)lwJhXCp(p1Ft*uy2nQO z{Cs%&{?2EHt=qE7eCxsTVy&w~gEtFjH=3`%#D8n#+Ha%gTxqjwopxZAl7-j5rGL0u zw^q}IV@oC%tX|kX@CseZ<4>P@g6;?F4!t8CuJyK_+t9@=-fB_W!~AId23`CA9N1Bl zam6F%#23ZfQ6D_L*S}cjrQcF-=$=|LpI6ymea6~FcI}V9)K(dAvC{|o+*_A1ebqsC z=*_(Rhhn7WL*LMD`+n|~`&#v4OOT@Kmx}{`uc=t`n=<0ciQ&7S8@^S{o%byHk#*yp zT_1E_xA)+Ei;#|XE6yYrPwXe=iwiq&OjsBM4oNM=|dtIm4Zaunqj*rGWb=ABJ)B~JC+9&7> z1iC|5eYpo`ER%f}KI?%a+N8ZU?h5YZ@%W#!Ys!>hIrzzdp|{K& zT2{)qp)Oajo?)iX&Ba}OTs@qRmyFD?JC-`Gj^2(}_9erA91yYiNz}6;!}iT^NxE@9 zs$^oJ>!u6#g%8ZP4fwtI{+o}Zf@idQxSu{3)pe<~)l0eSas6N2>^v&-(vzpte$}+` zTHNhkW}j=X=l@*P_SwSG*Q}e^E<4~e*KJ7Cr_*L#SpI%~uf7g-GPe9?+^d5_`K-(x zeCJ9PZd|#-T^}B^>wWgNQ(@P|s@oslUDbBzaw~6w$IRFD7V1~cKJkm+tS%Eb-tReR zNwuNhmSx?X@cV#w)2woSu{Ejf_uO0*MfPQ$&c+u5#<*GSxB2v{PN(*#T*|z1-)Hbabx$8IPaoi>ed~N?R*IK0S^8XF znp{?uGd|SKZr-MV$H8a*xKjJW^4Eq>Yn-z$eV%-Idcn3KCRO?l@F?;9biU{=e`CSQ z*NabVitTmD!SHzU%LHv+*TR&;b=8$BY+X6uBS|mEoD#IFhdnI}eK}``@hYr^nO==| zbU0&?(C@gW*uRItW|dE3s@*KbK;~htdirJ7qgD^jI_{Zzt!R^w(UZR7pc}d-l{0?5 zR1p+V-_1=wQ*Y(3L-f{qbTSILl)L%#3MW0Yp4BD>e2r<|p;l0hh&>JWE)3Le?DzA8 z{r;EdOfEAxv*>Zj{W{N6D`niYa>lo&&22x|YNx?<_f45C4sQPG_s3Q)^txi8yPkSGHO6|0N--vkcjL@-L*B&)#X`NRmt?t)o<1A7Bf3*)^X==i`x#NdpNp&X;C>}vzv1xjeYXR40slO{KzN0KgwUW?5fu% za97!B=lI-Pvpv`D@95cMK~k1^M*K+I>CIXyUv10X{G0vg&$V{uzTTFz=VPUeZ4_+L z6s|vSt6)z;x&GRgl?&=yVr3Q4<5B0czs&sOP>ki-t3q(Islz;urJr#<(LS}b_W6ar zPn|b^9$WQ>jd}a!L-T5fwf-gN;I~#o%bnK5o$<1(R8FhP`8wzwPg|6=WzPimb)Vgv z=JDU0ZC;;kH#o;?QFHzAdoAj;?W8_=;(COQ;(N))UUD9&)*J|T7 zJ<1&<^t^AXy=z?h>T}6y*66`6PNOgWv2xLt{@-4kUg>wiEUTUS_ziaw!uB)MLL(~G&aQIp0t=_NaGkFXw4dc*e|XNL{2BiD z9~yd3_8S+!@Xdi&HAa0~$2Qq+3^tpb>z%x>t%im7f|MKQ>yBNim_78cx@bxX@e%q#0Z?pK}-Kpc>q#xK@<@LLln@VOk zU+j^!;`|)sM#Y_f2aVdoOg^03qz5sdm-e*pnN2Hkdf= zbD!1vRWiD{-5stDx<1!`Xy^^6>%K+1!d}(gu%y=V2Awik00!k*ule)4QbQ~R>)H~Nvc-?s@?HgN40WWGPt(N?Ksj|_!`>;9apup3tTV?&|g4S&vE(A)^-+6$$4^$^T%26G*Rk$CMW9KRVX z~8sUp2C=0!JLA{+E?C$NEBy`H@z9?O5i}2j1)H~q!f99o1+z!N z!WCF!ANg}(!f{x+@6eXV{@f5@%;R9e?Y*IJ6IQrj`y^O+1Z&0x>OnxHB}>tH36 z`E#R%-DSaCx?uD=m>VN#@R}i%;x$t+`y-gk5=P>6tZ*E!;{>ZW!JJkY^Cp(5OPuEJbU1!l*0{@f%X?_DtKt_Slu%qfEV`(XBg%%$)Bxjf-9 z%t9s1o*(?VX~Mz}!E8WPm|tMd5PZvnnSM2xo67yUSwcC?5;DU+`g8e0(Z^tJj=+5i z<^&-GuXBZByv`F;pM$yiLO5QTup6%n1fws(+(JQv*F{1pUKa~yUxT?N!brR>6^`R| znPBxTm|HH4!Rrd)99|3N*>b_$%6U0>T{Z72URN_m4zz~l5v^r+h}JQ89#qKkiHg`` zqV>$X3TOjc2x0?_I79ZO3g^#lV!nFNH#6*z!Q2n59K=eD5gn#PbTKPZLWnhiU|1Ex zHWpG9j@wx=(GI4n2HMHOiFUEwL?z5fAGDiki1x5jqP@&a1=`0(676Tli4HI;1JFS> zhUiCjj;NH`RtFtoIYfuqRidAmV-3&|mPhn6yF+x8xf_Cxv3#QA>@m>^=3NtXk}V`U z#mb0IGv8XEUswUr8CFhomIc-Zonu8rzcS7Ube@HPm}z4)c)t-Ee37Y)Asizi!5G42 zwwr`>O9%~3AY5e{6QsPxN{OyBvpS#~Y$VZ5cAV%Iv#JaFjg29?&CU_sVYc-^cUcb6 zJ$9Aocjj0hbf4uBJz#f;9x``R&?A;l^q4&cvBKu4KuZ24O%82wzBe&V0=w z=-WWpWDemaD~G_nVu1}nWvqzkHRCKmf3OguH>{ZGEmJiFy<_1-@7Zpm56q|$sGMnt zKC)7xPt2?_3c1o6g^X>CLcXx$B;?vbu(O2ljg7H{;ARiuCIp`2SW7FeKYK*V3@a#A zIChPc1#O^ow1%SO*i>sM-5sDjC#4$4+Bbpnft009pr|XloC?HnnN+-*!t#BV%tJ7YyriDW1%gen6`tmpOm^B zGq8bjjFbc$DD^qEhm>>|C=G3)m~m{lEfgD9D5ptjz%lcdP_B?Nwk4E?96Lcut{W7) zR!|yqETa{!3oSYJE3p-Nr#0A`W8;aNpl^tqa?HsN+zdTK+?-=~!CVXGZjW)WVfjS1 z>@iVG=G_L=iY+8+&B}=En6CrKo)r+aVdX>)EYK0;$cl)Z80Q3XW+6mvSus&NrgFvr z9`iz#6Pz)ku5336>D~|;wuRu%G;JZ+bcS%61P^A`4#E`@#v+^LLuBFA&1%agz$)j89gCPU{^_4 zFc3n=00@&pk;OM5}cV~_KH-u?yVQ;ixIx8cZ!F&Tj zGg$%AELKi5n*|1e@>vnl9LDtl2`q$YE-NOQ$5eem^I13%W4nnKFr$8;g-k=Vh?Np8 zW@i0COV~)FrR+F}Xb2rc zAgp0|ArQL9KzL5VI_4e<;R6XvLm?C~?|~2sVX(NItc(-7@ur9^v} zSvY7f8wp}pMj+dZ;mCGBvl<2=Hxa_5VGs_ob0oMWL1-5Np_JuBKzKyLeG(2c$4CeZ zMnaeu3E>F4Lqhjt2wkHf9A)`Y5I&IbmW1QXI~qb^3WPP$5Kgi(5(cC~=obUwG%JXK zpq~ao84KYI3yg(OLc$Ib&M{5{A$AmmC=G=3te6DT(Gcnmhj5XF4~KA!gd-$eW=3%k z($gWN#X-2rN=dL81EERI&lqnEivcMDwB_!-1p^R~<5Mpy6M5RLbgB6otIvzsZGzf25 zcp8LbBpe~(Ju?~wA$>LShlOePlgVP_!v+-lF(>@~QJ}Fgr=9B?t!4xR-GN34Vc9)dyxlp=hLaD~HIhjyC zkn)xk70)_nK`G3GvL*{kb)LN@Wx!M@{l-Euh zwP0hOg%X?aY%6gco*86=>+)InT_;gB$QHnb?A7 zCx{#J%z6U25zjJ+8}sZ}VoRR2oCvn!*)?Ko^uQ!=6P`^aZi@DUxn|6LGFD-8mQU1z zJtneY-cvxftc<87^UVddVg*F4Svip%3(N!Avm&B4jGGE_U?D_~yNaiBvp80JA$J4+ zk5$i5_Y+N1@YziDUHFF{O?U&{j~cx++7&RJv*-EljoG-FTqy5jx$D|Y?v|diD;`0w zN#1A^vg%D)F^Bn_KJVhm+Rx{jD?fY6`AHs;qDN8zr^j9`;PjNW{P6{S%0S@6nAwXs zePz48f68@*u)r}wC z=QBr8ORV`_sH_)-K6ciR1TAVTc$y4>OsB7xH^F}%!Dms~_F!YGj#jF3hfS#S)|GhDh~>=0dRRer^ELSJ zgTbl3nAxATS)>`t|EE2BS|J|MV%`N@8-qJV;+M;_KNaoTyqR0en~B-Tm?B_4)yV*1o`3TjOi z|NgBxgvUV;KfILv=)=Y@lnIXSPZ=V-M%IjFKYGZ`nVzVkR*H`ug6V0e6V$&r>d20D z5!O8nTvyiWL0bwxs(d|JtB>$A)JFs)2FRkGg+jERJ^Ha zILu{@KJ9dpwFa_ApH$YAH4A7Iqz@UJ$_48lYoyNy={a3$kfofKKHHlj`_a>k2sTz& zLZk9Cd^)4dSDq36rSHQ}vUdINrtmYPS+q3Y9DOz455;Q-A-nso^P&9p>-W*0p) zRu!O0bp^NztN;ptWxx_(F|ZU^4$xzLZh#Bm3ed!}2dGj_c>LF!rU*0xngcBWYAVe{ zdgf#jFd66r^aE%%1_J|t5Fiv72n+(kfWg2JV5owr8}WW2sR)k(MgufM#{d~XCXfY; z1!$7eq|63#0D9)d2j~K*fv!M1zy)vx+yHmn)ZtCEP6*On`nCXld}agC$6Yk)^mHIS z`!o}n1LGAjw|f08k1{ftCy8 z0aJl#z;s{+Fq22G%|c){kPkQk^wg^(K;QnPCueN|ny1r|*$iMNFbkLsp76FTaCBRZ(8L%8!0Tj@vtprv9dB9YFR(uI~2e1>^3eeU-k7B~$vCjdR@{yRV)7vBc%0OwG#Ux7=&Mc_Pe0XPGkM*J_pN#GPfTNiz`Ksi8Pvg4qC11rEhKqp-~$oc|^8h^;aTwSS>;+1I1HeIG53mnd49o-afkglV z<^v0WIlw|-E+7Dtf$=~lparsl9AE;F2jtTCR>mSQ4wwkgIcO4)1=D$ zDMU60zXF#5uYo^+H^4jKJ@5gz1Kb7f0T8Upo!7KpnAOb2cN z4!+0%$HZXq3P4L+HEr3n6l(*Nc`fiU@DIRLfaU-VXF4z%pp7U6ppA~shjebV2529t z4;TS@0E(x%O`Rovqm-_T=wgqqndtfoSB+ZnYlU=~sRHPNtr}1jr~}jkjOB1$un9n! zQk)JtozA#sfGN-xZ~`0w2cQi=6>JR9#-`i&tWdtfQWi+G16l*LJ2nAmr?ds=luoC1 zI>pmYNjoHI_JA|s0z{*kbgmcz3JMppu*bZz1wgQ8Jfk1!24WKtN{lI;JK0pu<2=oT%oD%@hImQp@4yb`HKxe=U zpmUKwKJFMxu;&hehxeX`T`$;a^M5-o)*+Q1l|H~fIooOKpF4~(C4I6 zt2x47fiHj=P#^eAt?Q-zj3AwX$wd!b1>gZX`O&`q4McV|zyMGIH37;(cOIm=Swd?F zkk$aO089X?JDvMTGX?4ab%446bgjZnHgqDfInW4b3{d9Hz)b-<*R=s!1Ekx7?PS{# z>;kv}?f{+9X&QxrX<*c{O(Q%A;ekLXUEhQtFaQV!`UCxdzCa%!2nYmv1HAyc81D)6 z0O$hE@ub9tlw8Vt^zd5g0+IhXf!Vhy#WL8Xy*+X^;#~1=4^V zARC~Tjsr3QIt|c87};ZiEP(Rdg|rjF69C%r^T4^l6krlC8K8NtU53C?U@(kn;5k4P zFoZOqE8qik2E2h#_<4anflfe2paT#B-2+?#{y#X!Y5yZ9IaoowKxq#FbxQF<6M?#i4vJv;wtUmtM-Y;< zZ{Mg`QafnL_XezMz*XQlK$G+ea31&-I1QWvP68)@v%neP7vLOl0U(Y1h%d_FOW@1E zbsz`OO?JAiF%EWhx*zfg0lLRQyEVO;`U3ki@ChgfJ^&Aa-+?UP9&i)53)}&20lxva z0g8VB+y^pb`ze^ppaLHQI{gX4_i4VAA@Ck}2TFQcHKllZOG$4nzap&jQ^5Wv z`%&dc7jG;>r?;m%(A!jcOG|H4E9Xf)r6oa^n7W7x%LJ&lbjyiaU%A_K@pRu;ml@5M zQ9uencaT~F%>nucaP%gfZt&0z9hzJ8hF*96!#hK*_->xw)zdq9-Mf5x$4~F@akEjp znzsbgJw{vf61_QV2DSz_rLe5GAeO^6h_^ww4M3Hlo0oO~-QuDNO%V8C(=LS35J381q;uqc%5)xBm6}q2gMU11pqm#V@ zn-R=6VA0OJwTbTMVG-wI?}T)2_gRiJz8v+EEe2_MR6>OXk-WsJQRhb!{dkV90Z5^x zPz$X4VOc%O-ND`g=Q70Lte!3ZG;En3M-k3QmxLG##OQuJwjyRaVj3bw_w%xH4DC`y za+K~DXe$!^jF?79r2CE9ikRQ!7~QYfR>bgC#5Zf%y5G62h_OP9C31Tp{}L|6^q1;Q zcbOwXj$MLF2J*}LFkBRM81JUJyw_w%_tH%=bu z%Sc0$-pB5G?*Vf(%yBe~qjqG$kJ}hU4g81s#$PuIad15if5sOAqLypn?fHB2r zF{e?z$h3$!DoFR6#bQCUoD_*kshUJw+UtJR7%g(3rSx|N{rf7!Yoe$ah3=<|@f8Pj zkTZ2qQg~bB z=+V?rnn$`{b(RZ6S|`NRp-RfX`z)7-s{EVs5L1L!(G1i5v@_D6lkDBm=6kiJ@zwpB zvs@7RX|0h|1*clpinRTK4bV@t;E0K6eB$ z4$`EbW6Wx};43Lhj9GIRz6ZbFn5DY#+57=x_RfWmwz!HsX=Llx^bL)Oz1*JXPRq)D zW7Z>qH?{afw$rDj?l*b%aR$#(KhSRb&6sU;<(+sF6L#AbWi&QnA6-#KD-+h;jgRJC zOxPAT=m93I=16d`30o1so3hVtd|Q5~30vX?!_S=$;58=fMF3wrgl%_6o>?aB zkvreRViFom6XMSOr4wEpa+_JvE3-{l@AmLqV#2afPkx69Th<=@iwP@f&$l$Vjg+W* z_G=UN47x=}%mB)%fugtdnH4(?#hk>R?_P&ldhkvbec*w2f7wNQhdx;~K6;cMhjzBL z$A}NF!$x>u1hnvI3=f~3pKhMpw>*XRTqpas=!<2Dq4P^Gy#@^vsviokh}m0*ZSI8f zFT;aQCf!G@?(#KG?`DO^+dAwOa_3FzvZ3VZjE$E(cb2Z)>SBCX85s zYj_Op8)W*d@7{n4k6DPZM2wYQtrfrZExub3^Fv)W9ce8N!-Mwz%xA-1)tFGUvZ9q& z>#`py_m}W6g~y^ARw}&^{3iGJeyE0ol=kB^?GF_f!CgYs$jooijHh>57j1|h9Qs+{}Yd-E=}o^4oC;W4!yn@t|e;el?< z_8-mp1^#iPc7?~i`s}BUypzknZ~wpVGr3p)rr+g};pLI~o3Yf5_P?JG+B)oRC#(`5 zOXlE-QP-^!dA;%Ttk^iuO7ptVik%`m!+Ji}3qQ%4{pgEznw|B+3VdMA+`LhiWSjC$o3It$$V9d+Qg8u5v$pN{*;XEBN}FR%$h(p3!K!xV z%`GnDGLwcb+t1qa`F?G8S~^%cI15UZvUmvx?6KKT)~s;Zc)e47o@*;@Op4FVSW;(< zT)pPfd1C9N>m`jFn5S2GG;7XocR`(;o3jJ(bn%8KZslYfF7ZFSvYKUFh3CJohMdQL zSy5{EzV0YYSE$S`|2}IsTtd?cu8HPORl{-v7qN@P6PzxO88+;l54Mk`vggd#F^;9S z6KlZJMeHQSYa2G0nke&sUCmFnY$Fm_)MzQ~VOwk`hV2Uo38?6c|4xgUmk%zq5x8|9c$cy2W|g2 zD?OBBVusZocIsiqUc}%UhoZ_`vad*MVc1H_J*@8om(@p4;@K-*?oC=TCpEfk2RvwE zg~oABot(xGt?)SAiqX&RVb)94mZp7OB?`{7W4}=16?Wq1HrWYr8D*ok1)r=)yw{Fd zbVCJ>z=KYhFFe~FPgm{;hetc9@>lFw?`~Ky@-*ii?AbC((a~P2(6;9F>rP)Bu@Na~ z3{mR8cz#bQ>n@kuNGVG{JS<)Pbrl6a;S(I-;NDJT8ky{p!aEWJ$7%p-FxoX3XhU@ zY&PY7NzNnR$v33rWR*D;9`6v-0BMgs4b?kmZPcnFCdP$X_CxMjF48s4t#A1^W15|5 z$>S{TC|)AsEMe@AVaE@csr_+vvBp)phWDwR(Bbx@360TxczZ-wH(4%hvp;$WDUMN! zdLHaE+42A|`yS}VjULRe2ewyvTV-2F)D01a412IC`wG!Q+d;Zy+<3mX{)ItL0#T5= zG_T~v!g}=N&0X@~M%RF*Pq)_WmDFn>+~f(%m9Z4GsPY;@=Dc}vmF`e3U~SU z3tQ7sny0ObBJRISTeS)q$us7^Ubec%4-eptn||#i-hRv;=_&R0xRSJ#V{R9nkfEc! zGb*Q$&Q3k#t2-x)Iq;+_u>3ZmGn(({=w62#3#BfUrJk%qFTN$e$&(;@g8=UsL(^xyX(c6LEh@qu&tXglTew1*iA|}#{&FqaClp%YVe*EYd zbiyU7!eb_4sMKW*9QK@l91&U(v&xIzKw68P@SuIsuzr`v%QlZXTH$fZi`5Q9?ziDV zSC^XI9SYl4O?Xh@@menT`h;k;R0>Y@NjiS%m!~}AB40A%4`$I9HFN7B-LZKezxaK1^$*=DJVJV~-tcf44G(MNQNyRx0>dEZnH3%j z5kvQMy6)eSxY*Prx*}#@54Mcb-hv0c15vy^S#ZYZO6LlXFNmR}}(WqCo!j+X|2W*3Q|tAI4!tAGT#O_QCai*xPq&_ZB60|A}C5rWD!lI zp#g1h86<9T7o$W4R2sV3gl?cg5CKs^aElCrM@8I75MzQ66^t5Ol1$j}afxPFL%WAJ)-S=p| zOmPZie)vI!b_ftOmlYo2U;dC6hv#_>(x-z$bJ;_EdTROp;{7ruC6KKIuc{CTYB%H5 z^D`+4-$luYU4iT{{aoXlK=#x7*b{pMv5@JwMqz5+bgT}(c&b9#N&CL(0l(Un{boav z4Hkyz6U6F))XoBed`)$vsB&DP&6kAW#S{}Lv@plzE>T<^w`VOV1Kg2Hp!^=h+-3-N zF-1U-H=N1dG_(BH(0jz|C&AERd zHQd6^#=;L~kCjC6hlF6WSD2gw>I!Wv>Tm~{M^?LFRzdAK1aldG>)~tv=$yNYI2vqk zy%lbtbOiDbrDws&ghi=bsu${CV?jYlBP%C)q^|P#_5(H+zS5m2e6|NAG9_zdyFncz zZ7hOthx(Wp%rrA$DGP#`o<4Je*}9p+9-(gtn-qlk(1dW@NAp+bK8>AbBd!->Im<7E zEe^tR3LifSq>wio??%DAz}8}e@p(3s-3^9>OQF1N%O71;3vaPnNT4DSbSqgW8yq4` z()Q8u3XaBn>Z_^G0v`iLOC}Z%cTlMF&(C^H6>hBe69v$AMv_yENB*XVtfpC!q0kk5e=yB=y~VLFFm+^f8RY% z$oAn0LAilh!Lzh=5qwW&Ib!q$%Rk-0iwuD98ie^e4+_m!x31SF^vfG_q=BMQNM5OZ zGm_hUpRdcN_O*{`(}3_%3%G3;l$J`}_5 zg$cHALOX5e@`i8sIUjL%mZ(vw9$OX1+~=Uv-G=Z^`^^>(e*cd*Ej&k=0fhX6+L4f$5(93=UNaBYw412)~PXRThpxD9re1HGvx z^OKQ1qb46p98)&jz3ac_6ryopUaffCU$~&?*T!v$!fLB z-KdrC2r|Mdo>P4Ke(LM>$4RWj4Sm+cGnEcKn*s!7*Xxhf*JN%!G)_jC;#mL?+EqZ% zK6%*dQz8=cj_GAY87LGj7mpfy>c47_ddrlf@oW?EUIL;gA;xc>duvDg2{Pib#A|u^ zl&9O1zdeyDwh8Pu^~@`Q_kDP+OR38?MTv~)l*W38LApa4_nYO8m%kr7)A2s08WT-M z;06jg-nLt5{gXl->niH`^&UkNf258kr)k)7pPP4(V}opDefjIfgK+#(hHNV!rNto7qqlH*B-uOc;@ z)$0+-mu0hi>8R3rDfHRQ%>dqz9QI@(KIi7JP`Z9+DN8dTg1?{3zBM4~33*Htfw4Bv z{K@$TVBPXkTp7zlN;}igzucU%0ab)dtUI*G77SyV5r_a&mT_yc8Q^g9x9Y40vEN0{TW$YQi+J@gq zj{39yzW_;hOAj)5dehAw=uWOO!qtOl72{08WQ0W|rml$EPpau|6^{cV0VzJFZP{ z;%eP_iE=%IZGt>Oa#eQH2z9vSv)goYU_N`zgv`=EpYMbxuCMjE(tc$H`2!*qYz?+O-DCZL#F#M}}gJ+z4D)l;n1^DR%#^`pr^KL_DN zP7wWZFno)QC*zDVUNYk5dZsaBjxqjmX3Wdj z4J?zcc^tD-?cTs+c9gl|`F~sPj23qU;+%i6fz?Tz*UZ8VVMH-InS_2$DCRS|;CAHk zJnLIdHWn~$@h=b*(xAKLo!qlmuQJNmJ;gFkXfbn025<8j{p>to_Yp$vb^irE&Dz77a0$e1_Ht!B&As z!P13n`j=uhbRK&45WF<&+a4J;ZNBXpC-=R>Mivi*)_EhJ!LgQ#f&ka51TLM%Jp>fW zmzE#;-0H%`D>+0NAdYIyp%}5e1KHhL5h6DUVB; zY5{bHj7htIR?Tddy#Tt4{E+R#wQ6n-*Ol$_L!+yr&Vds-tf=kLWvq8PEbF~8p5+!C zJ$l|Ypv!rV@PTQ``I(fDX&YV;Ftc3u#s>D|c4!X|tdSHen(Nhu&TU}EaoT|AcB<;_ zdV`4= z6CAXIeIw=fQ#rX|qaa&K3luw8=S=9f(+=+GHx$$ zY9|Xtj;rdwlhWQ06K;``_Ej&}G0&U5gqly8CabZ> zcB*2Gjq>_~MLwgBTRG*T4J>;!E0Rn>TgujD!CqdtzLq7t?UGl)NA?(&N7@-)GR$vf zRj7+7fhJtCj>cJQ;aRc;Xm;_DUbYB-Z! z$+ke6b~_MOXuqiAHNQ>gdm<``_JeT#Dhj(xyvM7!Ut7I*te>mt!yTe@QO8$6Q9)|X z)JLN%#`+Bh1tFGncl$FobV=hV8g=-J8I~-)(H6ZVU$A?Hhc|d$bGxDZNZg3gPRu_W z{gC!uD*CAq#lSya@A|cWhdr-}Jr;AUPCM8KK>i=R_-Rg>c^gNgL z_1ItY$kV779PUiX5hlhoyhp^e?mNrOya@~pG!9Jlvc23~y$izAj+AV{Ry`eeXedPz zC2n8uUR&4PUk@G%4G7s=L1_m>w>=-N`StS&ABoaX2gV_^z07?njBU?e7Pb^F+pI{f z+FH%cWzB)MMbG!@@a{{oFaXY~YW9rE*3J*`HF#@3`l&4b%J;I-%Y~6Ff4QK+oBWO}7l!p5W5G=%+jn4{ zMQZ0~=Ok=;Qccelcxz<)czr~SS*?#YC)L=m6d+W-;i2}BFVxb(WA~)UG=HNmK3bh< z_6XM*=jqrHJ7uqhO>UXND$>kqGCS{5TeBQ?y>) zoVXAjF{@*vjmfDbQXOebZZ1|P#wD9$n^VyjZ!eKG&O9%?IhFH@jkYd&4mHf_FPARS zUn)N_+8jQw_0F>+Hp;H-9VVzOn`&AuQrgQ37q11iC#yUlIQz9ekQW+~8TlArAW1Sh zHL;O3zr*ia$DkkrRT`3lBaQ%bGy(q@3HKSE%>Ov zoU5kSD&c#fLs+!YL;BSM>tdx;MetuRfKj>&rx%aNQ>aLdi`ARrJorbL&TN3R7_&aw z7#XHZQYV?z$;Rk#lim=oONvrQo7K87vne(?$&hT+8xqaorbGi+!;+(8_3X$cp_i)R z%8-cm zbxDRqgHEpti#4dtN%$JaruZuVG?_9rUlFN`FsQ?lBO(lmqziR&QgkfT1S~W-M{gk2 zC5ktp!bC$j1jZ+t!VPA#+Gt2kibj8M{-2#H6I5=k5GYW$c3aIf9|$h&gs(CJud!X! zS?R>W1}W|FowUFlhA44fp^MUo%^sq(W{WooF1=nH6QsU)3G-PbD4jde(RgYQAGj4? zFSzu;L3-T5e`tpuGe9|9GdT3Dbm-BjC4d^g1FB*7+?0;4EzJnH7B|Jo?B4K`1|S-~ zLeKzq@0#G)qtP&l+F+zGnSn4VC#KmgbhQ@qC$WR(cql(9LXDR^D|_{5xQZcN$`_K$ z^lIt=fN)QMl^}y-M^*~GYdYo&rbRWoyC_GvvbMg;0`@Oo;-#? zJw=U4G|@z(CK_X6kH$x%CYo=pDIA^@-{<|_|NZ~#zpk9ib=JPuzIU13X3w0#dFQUn z#aG1_dV9{;(R130t2L7+{k;BNx!I1M8`kaq+;4S*^}e&Mz4DHH_A zJ~U2}GGdd|%%7oDSLnlH($Z6*d8)} zV33|cE!d0?)#FR(b`9Nr31)^4U}kv7QIaZy&w#6f_kdaPmEiK=w3x)vF*tMuHtL=| z(^1M&0~H{!$tF8!M=-yP)&;}TV$#tN($M&Z@uS#`@MDudL8)x9e9SOubZmS=4ALc`_AbzmfLXEHV0IPSIb(R7B>9$8Hrdt5+Kl8lwCuFF_>u8Q zHWzs@yJUQ^m=ACetZxn2Y~V?{9tCCxv<0)Fr&iMJ+d2oMQ5c^ZGdyN2^1TGT6zpd3 zXSE_wCmeQGwiN;P=(sWQ>W~?Q-e7yUqZQ$s&DLbWeW9~MT7%i}e!5@H>RQG#!Sovs zW&x$ZXv%B{Fo*KJYLbLHWOuEpbxbpr2jIvMV8S7_w2Uvnkv(<}N6rSTrWaIPFZdZM z%!&*hnv^jLU33+GtnjHi+7Z78X1X;FikG=cR+y(&t}$Xig3aiTURrTGVIzm^N3hvI z)nTIs*>+%*nEhitZGe3OW@yTEgxSnA4b1N|0Gs;_ShyK~u(OPVXNN9v_qT3mjPRca5~3zM%6NFdJ;QkJc_e7$F=9#`QHGSh5Z?r`5Xeff!BfQm!sR`!0eV` zy51d(^0Kmnb%*+3W>8+YO<-p5u(4LqWiSi+0PF_d4rct@x;+cb^w<2f5$_$W9cOG> z%xJV(+HM>-+a-2bd@9>v4QvZ0adv7_Lc*}5F{9oJl_U(d>=>Qff!QOTVD^wBm}}AZ zA(DhKmwf_E`yDWQaw?eV2Z3vY!@x{m9b648gR6qSYpK~Ef<3Yz6d=GOTLtD3Bqyb% z4^2yxo?5l3a1YFykBSbQKiJ9yl@MSb+k@FhAE`})HA!Fo$92{gnK(2tKeM`OLu$Ovsa=)XPA#3k=&s3_ zIiEY#k&ieiFP*A%@am=2K2ztTV2;ORumv2Ql#~!1l@>EBDKR5GJ|XQ;Z!O*!=V*zB zW8$LHtM^i(E#6tVdfk#6rewrq#7IGXwQTBxxo#vkOo&g6PnVv+=E9d0oiUts9*uOg z6H^o86S2T1Mh%Z2iqlrDep+Kr2eUUvfY~PH!OU+2(y_6h_gD3f*@JYU1K14_Ep@w= z&JzY|3qfqssPw3zF_QBj?fAQanNe&)R66FbWN}dXIakSgYSywDnvgb@_52Ywo97ys zo&GtP_5N*$mhDm4WngpEX5dsi5d)$c>>*%wdTTH{+Y4L~?4s+BhidvIFbBawa23wK zJS~tt9?ap}AIuD!>s(one;KU}`%7Rp)B!Mu^KzXhfRSPL5HKqcrt7tJ+ZoJ;T^^-1 z>~gogebcId3%+)OQg73@;5=j;9`O||pKY12O~K0PKe zDp~sD0@iuFmTy{=IzXhTRCPcrpA~n_GIk$TCu)rvkGYbTfW-|q55d@t7`u=EN7@Ye z>P?oFZM0e51g*2uVn$)WVI!Cxm7b9nleT%PHtJV{IgDds8pg(_rllvQ(Myt+LT96K zQZ^hGEzQyW7EagFO#yQd#E*)LNsUiW8y1tS<{z6HlNN{3BgLdgsm+nCCmaLjvN9}2 zoA?nZgav;yL(8bgOs%D`?@UWUOD&$I4cAOChi4Sn9ozw29^3@X(@YheEnqIzPmzxs zSMQq$a5#RcXKd^c+rzd%_opK`ah{g&ESUNjy#+iG&voGoRFD<@Y`zwM1~&bSRlhwB zge{SY{%6wk97&4hxQju6nh>86 zj|^Ka)(V^srvIm47jWezO0$x6vR*IL8m1;(*#LKi-s2gu4HJ@vOKILilM<3rrR3BY zu5YQvFgNT8@Mi;G%+>fbnAIEso$YsEsaD;lZ)@wwdte^##pP=IWo56+(-Kx$q1AsO zY*u6{nDrb3W`^iN0ZeR|vYsi-!n)I%=bf1IG16vC&$Dw8%X)XlzjONz(b zlcd0nnxC+$(uw#?69=KDREO&^t-8Xt?*D-?DXheQ(u*u=HKT*_)8 zBMbP0EtS##oYpk{vdgMPrEkq;CX@2GW?M@KlO(l3E)L4%I-$zs zT5ZiIp!h3?>xIkpW#wQ!tNC47l6q1wzd)#|8uGJ~q>gH6B0_B#ay@}i6J&+->c^~1 zt`{gf+AAZxtg^qovJ;eLue|iKS_01`5gQHr9@x_#i{#Nr= zME0Va#e^OhN;jpTL5O)6tiG!L9zw$l^_u8EwNOhELj4Q%GqBWxEfq0z1{dm?u(bR? zLCBWB55|e+pN){s|Jy>n5k{CzpASnb_iKa-%eDAp;14uPvn+xYWmx7Xu(W0g!wS$Dh0>tDm!YrDn7NGbd}8cZgbNOOl-u)iKCaR+$}am0OfmJ`c89 z(y#>dpo; z5d9RFR^gU)6}02AS5lUS+QVp0S@x}{EN^AC#9@Qk4T+tZ*t{Q>8f|`pP%s_MK9x94 znA0MJtnjs0Oglp?zvz~OI@UXRXmPUQ+A2g|SISSWP=LM>TE+tj>sYQCwp~w&)z!7++A%zb#SU{&j@Jt@H><&s7u+t~ zG7}-LC3e&;@536T7D8)SO>HR1YTVvJtGj2YQ!S%zIXI?tSZrvumn`qY>I%!m=CYKj zt&IkI^>{nM;_+%nxC|E4Y1VC6+7Y_e4Ryj!mt#UJI~f*}IVk5tLM&gwvcggax^lgh zy_u5(D-nibpSs2*U}LX@6`*>unipUNsh;N-hMMXrBf41413V?Ek20`FYm=w)5_V}X zcHpZn;qqWFWp-Dq>v{<6XJduLQMC?~rTR+6ZdP-v`jRw6N$e7CUW3p8HS`)GtW41Q zG*H(Tgf=0h`8hW<{9+Lrr24&&P(Ow&RlG51SQv+jJjz?S+rw(v2(gP=Z609pMq0x- zs;wIei-)9kt7R-K&Rr+9DIURUt)^pVHo-KaWmewb~#4xy&Q-|hPu+$}69e!o~3g>U15P7bj@^v4prF>&!=7#NUghk6q@fm6{ zHC9GMSk24)QG${f5pKSU5Y7S!)j&VBQ$r&V(qgwEgvA-LG8Q2%wgW;~XrV1dNK1Pa zp$@862aTbn9f6RRVkbgcir0mj6(a^qFLGIokQRF#AuYc;&5X385YqD7S*X24NK0XD zZlsutkXC~)5z@-7f(F+7Mj)iu1tB#*%QJ*fo$LzgD%mVp=`h%886Rwn&>PJ|rj#&ndXW^&-yt+u4TWJb9jb=jK`34gxpc&dRSk_osJ|LIjgaQo0BeVuR-V&I z&Gt5g2s*ijVBJKI6;5wg<$R++`AQdM_6V!1rK=WMOkd=+-BIhVO2v^@i$EkNl!*(Q z`3fz?B{tk#wHrD}aXA)liA0E9EGakogt+Fw;##EVCE2kg%STYPb6@x7A+Eo|@_~XACZ4l#xQ9s0~BP?cJbbDi&s*BhnAhXl3 zwB%fqT>2<4Q>~UZeYA8q!@*@DEcRqEbxqn2i#0Q=C*9k+CrZZ&vRZ@^oo+QpMo5x3 zcTTP|5Mn1d>aF0UcFiHEoM~$JS)RfQfMrqF-LSq|oTGZ+<6&VUp)j7s55uCTx@ z!)mF_>l$up5vf%P=W{HR36YB57^`J7L>8@{=Ul&q6$DGKhh$c&*AA3@`YFz1t>*Rp za7C@Sj14!t^yh?RXdFT<7?St)SG+Qds-Z5W*7@$nt8*XvKexU`TF^{oB8!$+@ z>)qA~8e2e}+m_E^@pP?qy*u{I>=^AF)eaUtaiIPomPxSK7+SiWuvks)5s38 zey}(m)a|4^dWh0&lGSnqBGyB6mP-F~O9BMle>aB(xV~yBogtXX02=!DRWVN;*jvc=0HnK2u3ZbE@RzJ=t zVs2sR0YX}e4)M5lQT+-Kicv%5MsTrW1u_al7ZB1?_>MFxFdHGw?@poCDuL@7(pqv6 z;*iv~WLIFZ|J5C~xl$tczZm%P+(c#5TUN_S2wER<)ATzmwymU2tMF0U?p52WZ-9mF zbYOF!s5`8)Is(xEUZefj%S8CrQjFC!l=`*P=@XpkXkfu9V|4S9#=dK z3w)uK0So{1xZ4OKBi)KjFB*3i78*s53(U|qP}=G^JwrJ-$Le|rY?Y~0T!4|6sXLo*AQYl(Z4mBSZJgd^dW~Gv z>)_~dO0$Jl%U(oct=1<)`SIGcQMYfd5wMyf&b=_bz z;^f6Z!J57Y7B{U$R}G8ZB)!emO=K`EZG>R6W0?<28$GycFkfVxa&1X#rz~T1@M2;K zEN$e_lCz7(rNGig4&yGt(nby}Q zI;C7)rfZkjMh^ujH-ZB#t)aG5Q*iU12P;&~jn-9ItyPQbceS^S`tan_7Z#USZPAFRIWEe>l)S591Y7)IovzkbumI*pe6lNRy`x#pP_iIuv!Ms zG!_}xeIauGOy%GTt9i#Pu26?pgqtn1xf(Gv1flL~r~sj^YDk)ci&iz%xiGXGA>5)u zidzV^S3}-&)tgn&(hzDYQg)llC_B)VTVQHU#52#`xk|-+tEJXF{jyv+K2N{wsO4mt z0<{B=Lg_v?#PuR9j#MZ8m`bY?%`#tUw#I7iGM`hw;fip1#(d?>8ms(hfl_*{)lz?< z)*_SE7}3$)QRP`4=j!%?b0t#w{RkD5b6Yj zgAZFw43JWbwL##d?o~R%;`GLDuw96GA*`0_U5YCRwSum0M9g)UAaivqKNKMjC^K*E zSrk|tQ1)y{%LlMnj=hpIA=IhRz)n_fw^W(E(dsG$&XVxZTS6+KHuZC%RNQ2BZLv%n zTe2Rj?sm+nPy@AIKDNonasVdFu;ZYWA1_myZMIq(=Ne*4mk@bmu5xCx)%6iXY#9ny zbag9l?z+5i&?Im1G0%kAS<5}Uz{h+CW;ZU{t`TqJ1pLR;mNQT})vzNB3308Hr>$7V zR4vb`Dj&;JoVQsm^;T&8fgKkbJr7n7B-B>uhp^bVMl+a|8}$R7d{%0GZCfKVETy0| zVu8A=D;vz|BGhhBZJrp^wO1MU1-NXd!Lq_dyC66KOKUZrf*$LZy5VyTSgkJ+dJ#^X z(&kvG{qg32wH3zNb z0y%M1xIA{PvT2vqav4H*R2196Z6TIs@8I+YOTCJ+WWs8%#_^PJLbuc#N#v}>c%;egeA8<8#4J8iu-87Du?56cQzZPnv`fxFqi!D^w^{9HXBOVDOxfoIQ7 zfW<A<3=w%ogJHAc(%%@{kd zTB~s^Z(pIs)1bpPtvno6Y=~=XSZpd|24Fn! z2V4zcd`+@)?ReEJs&xT=$brBhpb~(6o%;DJX1Z|z85=b9L#91JGgYQNQMbvoC+Rjg zOBJ$oflP;N-6qqXtlMPTQ*@h5d#Y~#6*GRC?l)ccv$0+>1$zJ|e)aPvHp62+z>F8D z-WqfA2*43o39$WF>%0a`|Fr-=h3ux;HUYLtnRv3iGWVoY7_-EnSF^$(R2id3&8h&* z51Co-)Oi=!4%kap=DlBD?czhQ_#yKEjsf&K2^0f91^6Me;^%1K^Cq?+{xZOPt^!Q= z4ZzxbtMhgBNe~zasE#+m{1jy-yhTyz{edyN_!ip_#!^n(AL(f<^R!y!d?_}}14$bYV$e^It@ z{4dcX-o$iSs{0jXNpD1cmTUqu?Ph-I z%-|N?CbOXJI`7o=zhXx1GU35B)_9NZ_$HQ>d8f72b-(WYCT2Ab!H>Bg*8Phzr6c$v zAJyaE#B@1kl9ZFDU6t=omr|SUgr4qy!p!2No)4L)#gD-5;BR&PFIX-AO%6!@#ti<4 z@ml%+Q3ZHJH*qBFiCg*+ks16!=O1;Q%;0UE?|^yU`o%D!WdYOAS=Y(5i|Mw`_*)GK zOi&WcU8EJvIoeToBr{b_x~&AfqbCHles6p3}!x8!HmBqqk|dvMt6J@b2NRY`xRwocndn??||uY zPxmA9ocjxyE%ckNlUeR7wfz~;h1XzaBs16lf|-Gx?nh?cW-z;;IGFJ+VEmIx;fwjY zfmx9XI%id3EIwprPz}t4)pearyN1p+!St)G#}{Q5;0c}FP>&~b-UaA3nOoNunw_Nv zn4$g`aw>E9TJ;2E2HWd4nMcqC%!+jdGkteG{=dWa9Dlv_gk%Q$>Nc4LMCvw~c0Zl_ z>pGbg7zpN(N9p>Tm;GJ@|J(IB2gyIz z>uiO8yN>57bL^k%b+uh^na=GLK7Vm1;ZHiZYX4lX^SYg5h}Y@-khzECbvi#}wv+`- zzkjaR|G8er9Y=jM6unO8Xrax~^UwAAUtPa*{`_;j{{Qdm^;+n|H(jSAT8iIw`uhnc z7h!TxqVVmg9Kg4e;(9;AWKrVr?W`QfcQK{Hg9sBIFvquxavI+y6xUxO zOeK|ge3w#=7vD9M$M~+P zw0|98s-@)PySDNI-*uGkzekwrDjV@#Z;8niVe(uOiEpnZ1^BKnERdAvPh3Qz$<)!* zP#mDW;a@JIvJAVCh?imaed;1Uq3tUw*ui#s<{~EA!EP*0)7}rep*?JWk!cS*?iUwv zjdq~$G{bg(?jq)yVK)_*VVjx>Uk6ZgF^kkf+#m&urjGEf_tHhIaD;D&xC7f1D#DyV zVIr5*N<1coi}n^!YmrZCBVLfKqPsJwt=LFvCrrga?L{Q1gD4<%6qe$kP9lobSsWmB z5w0$vt|Fe)O&lk67Zpl?dWaNKPjMO~_M1#N(2}TJZ;@FNLYxfY8ifeqSqfk7b`a*3 zf)FV#Q#eH-*cC#5G0PRgSbGQ$C=3)$OGBtXvs zg`uLo8-yT72wUAC#E2IZZc`Xo7Q%3`u`GmTP7qwmL5LTT-z;s$A=Xj&OGN#u~S#2t{hR0AJYX+6+NkwaP~?vPfCFi%jv$R({2k4bAq zdoR#CBA>KQydb?Ry4MG-7aK_%gsA~&qlhGJ5(T8q!qSijT9pUd5Qni<9DrcjCS1Ki z+eJJ`^sUaL^hR`nsL%+4Qw<0c8$sA5PD3#57S(+~dqgH_uQ*TICp>*Y`^99^0dbjh zQ26?R4vAT$_rwj-VbQcP=!nQ69Tj&-$3$2Y&~cGVIw2mDPKx&ap!Y>S=>zeC^r7e; z0QyL5B%KnbK+tIsNjf76NFNJJ5a_IkB7GtbkUkZzO+lZDc+xp>oOE7PXa@RRq>#Q4 zr$J(WLmY6!<~ZPsBC|P!IBy8oD10S6TR?Dc1YuqaQ%Cu-B))E8>LY(G3BO?Q6-mq{ zUzNmn))M@UBo>jsmBd~0bxGiLLq90ha^GA&iw?d8XNy0fC%5y4-;ZPn(;vkg`0Z=NphVqjn zMzluiN0K-}ek_TKZNNWEB9;6^5@*Q&l0*$F_^Bkuk)KK8bMh~e@M;TwE{Q4R7ib0Y zOSD2e@ULhEuo&AMB|d0}5?`Sm+C!<=0?LZ^P<}^CP`N~;b{O4TO0;AXE^SDV(AZ+!I12F{>wpu~rBVC{z|rdqJqz7Q%{N5UPqh5KPrX zSZ`2ukxQx}9+PT{_I*ILL_VpuctNTox<`QOijAat!qgY!DI!T;qJUIiSRz3UL=>r^ zI6(3iuKhrbL_Eny94GmT3jINTB8AjgoF+99)dzt5MJ7nZbwyRL4M0@`h37yB?%g2F z8wjDPxJ=;`h2TLDnu}S3AdKw};Q@tU(R46`dOaYl7z`mq+@WxZLgyh6!bI*66oyke zDO|LV0<{+Tq&DIO38(aEP+PH)gj4!ZP7?$WLM*6sjTDw7P=bgeC5i*2QNlGDlqBLw$>KOEMN~)urHT|%nmA2L z7u8ci86p!T;$l$bwNw;2MtG({aF2yBFAYMbxJ=;`h2V4u9SWBybRG?1vdA3`VP-sb$FF|o$@?+wpb>EBJq8ZbMgABFw<(y%LYN`C zkA<*oB!oQ_W(iX!gmwuKVlpAj5d{!TbA@FbXr727%@+qq3xw-<&_WSU$`Qv&i$sM9 zAVs8*7K_uQC8GL7&{B~}65>2*nedzh$`zAI%f)5V+rl>slqY79R)`xQQ7;YEX_JlW ztP(ld5H3-8N+Dl_O@=Tt9m2ZF5Y~#v6oN7!^qvA?oyea8;Wh>HR0!)u_o)z;jfSv? z!bV}52BF;;2r<(jY!(F+o>M409l}--H66l+u@F9>uwA&m1)*;ygp9W!6o}&xOglw| z8K7Mvg|u6oChZZ`XM*;MOwvAap0rSOR97Ixy#-;q0^uuho`U-f2>y#9To#iTLpVj@76kc< zN%$?{d2S|@B}<@OGl}o0)SCsR%~B}en#7`|JkL>iO67)0v=UHe&W5s1K)GoWKT`>s z1Eu#eD7Q>v%`zyrshD%2{AdzAa-l4n3uOA7hth5yl$hmE?wQ0+D$l8ueH+RH zlZbvB%7*z+KA`fGNtDim(su!rj65ihP2vOC;JKOZc`n(ZR35T zcAFN-=6)@al3t2d#=_IcF1xLsbgC>$Z8fz}&egM*zwfzk)FG1}?l7=0CUG<|!W`2d z&;Qb%p=RM?{6~a4hv2CY0lmIj(FsS9^M_G-?X{e*;?KwI_uW5XN;b(($@D`)%%b@AL9W?K0HugRcsGKvT~hujSoh$U_M}nKbKQKcOmc~_*_*lO_J{OmmfOv@%)~s zAuIVnPh0}w&$Y&q9_m_2XybJ4CtWKA%?TPS|47&Pi1#Xl`FX5sd?4O`lGW z&ml0n0`LOp0Z6~3h8v;%e`gI&O0U>~p>;6s^Lfm6U4;A7w{@G0;a za1J;RaE5#VTmUWtUjlrtlIuYhe^9*+aKYenrTmZQGXYMa@c{o~X(6BhOMwB<1_FbC z!N3q83Wx@V0>gk9AQo^0_%ovepgu+n=3wOPh(cb`c=CT&m>ias1|uEF07e62fU&?J zU@*X^spo-b19N~G02i8*!27_50H0y!AF=oXen4ZO3E&R|0D(Xd&=hC}GzVG$!5oAw z5eNZ7fiR#I5Dv5k+5lFdEzk~V4;%&VqY`|8{{XNL*aPeZ-i5v%*bHm}aF&ub0`CB8 z5Wf~!4deq{fN)`?{lx=U9xf^wKuw?uP#xg>=6vRpG=nh~_&DMafR8Ok0DNBZFq)RX z(H#H|0@r{b=)M5|7sSRuZGcaSmIb&daS^frH~FaQx4?CPYx)c@wMoEq;4NS(Fb&8C zCIeG|iNFq2>_^}>a0gfhiRkbft|x=OeKEy{;Q@h<$ww;4{uOA=Co+R4^YUz6b3w@BsK1FdLW! ze2t^M0;~nb0Y4+-Qh+N^258hYwBcgeHKcz+502sEBwzp#2c> zKLhuH-;m!f#6r*$LAWUT9jFRa1}XtOz4NqR3#b891F8UAtt$fVKv}>I zC!gRffx7~ofi60XG(BPWK>3AF$IX5wTZ8!!r{#=Is0 z6M+f9cwiim35*3&fzd!ZkOpu(WM$|98e@Q>k&|JwfKXsMFcsjDGy|pqjAOx}z)XOx z!}uA%TL9x|(~m7V3!o2MkQ!!6ZFO1Zz-KL9NG1@H`@_AlTma35gJ9s$%I13v)|fQP^%j*y>q z!w|_&fM0;;05g9LyaKq0a|2`ss5^okblVwR65yG)G*B6s44(jS0l2Devn%}(ZUQvU z!j~W53-|zy0B@in&;Y0pcmbY3J)ka72dE9y0=Pe|0aOR70WE;$0C(0+i-f6ThpkpF z6x<$Y2V}LymlbFOaF-npv;x8ac5+8>XMmd)ZeX~X;gLoF-1xkN4>viq`vBYsaTCO$ z$xRS9K3wmk!BM~vU=T1Ez<|z@#vm{n2!zuxFn3)4yJxtfCph%Ki~>!-@2J3UfDiC1 zz*)efYYl`0_kg>=9e`8vHt;=g0l>Wd%QM^v6}P@^AtyB^*FE&%obMF;zjusOhQ@fzg>0^b811K)wa0=@(; z0$%{11IK|QKyTnMun%|-I0Wnm4gd!M#vcQY0zGy612D^AfhPe&e;?ta+@5`mzvw`3(3J_yjl$P@^v^&n{I|e2z$L)&`x@cP0R31w>eqlPz||tSf$(+UCQvj_wp3Qp#B3rK)(c>5 ze*k#&MVrltzYX2UtRuo*flh#We7I+94XSZM2|IZ*i1^xv*0v-cT zfS-Zaz!cym@Inv&2L2Uz1^f<3D3Crfn0I9uwg=k*#Q@%=;X~lOw&8UU2Q(|dimoyh z$5hoXSX#^$^CM)heeXrc=S<>Qq})OT4UkI*@C=BvWA>`r*$wJXkBOE|g95yLys`H1 zbY?+JYF_m!t!_{2Ae(qLq=FxfvIofSvYjk;^q1W|jQ1gSX#DZ^FM9W1g*<%ykclsD z9%`OrW$_xGmFA&Z6_CS>xChxQeLhOCd0r8|1LO+)Jxa~`kE~}P&y&UO0kWU`%ud_} zdl+v{^eb6I$~fG;5;A1TSeD0|MTsG@dw}ttBe&f?-!*iu@h&`?Xdd}y_1Y_YTel50 zJFMTkhL!R5W!mj#F#u^jjJHIc{Hpc+$v=*-1rLAk#`O5yEEdwkc)wJ;b)&Xjs_NjV zdtllA(JbB{B6pIXnnmd-Ia{_kh{aKIY=H5CtFk59_Zu4bRWsS-?x5~kit+LG^Y-$ran9oG#{RN16^?t*HM6Qh>#ZiF|t?7 z`i|NwX@bG-$fo$Ip!3!GIqY&X1+qL0CcOSpEufOrt;T1o_qwPBsJ*8+iX$=TU*pA* z8xIY+yLL+KXnPaJW)p9m@%A{1;;|_0jHAeo2j6iNp|l;GL?_rDHJsGTy6kqN-f8`M zq&=Qo^+l;oP->`?m`~3^PU3YOJdKxM?mfO{XF%y&>ujERPU0#(_dAIl@RYxH5~YS? zOc}4ed^mIFRj0NoQ*5;|UXVHde)OZ_Q#Y=$#niHhK}>7BVsm-%>K6OR;82?f-^N)E z?eua^Sj|aa?v>aSZWd1nY;03(Cf#`zMgg|?t7bu z@$S$=HSNpE?`O$%H{me?wKT@#6E`t}_6KE!PZDxCS5^#4 z$Dlno0)tkc8uHb$!eJzwwAtY#s)5-r+fI;Co#n zyV#D)DE$w471j!QKN^cN)gJKkFy2%;X3T`hjCs$vMC;3H@hYMV`Z=Il6>Xs^c5iUc z@xjG!zypgsmwDs$s>7p8MW6j~+#$sHYEiALh&-hAFkZ+yu-6X(?|k&}ZCmc6s)%#R zXdK=s;_}yTq^Xvl-;^OX5BI8~N(w5fSJ=aN*Xr)2J(s+{_iQ70>VxS*HIa~lQF^bM zSdA0`4%M}@c7>8l1{EJMo1S`&OChEVN_!aMe|oHWPZVPOwPUGST|8tS`am9EU9?EW z0n9`KR&Q6G@@40~J#-5aun)@m zox7}iqPF5$8d~3Y5p0>ftk%Q!`K(6@oZ>ma^dV?03)=COt%{wG@e{$no=PWw0#OATJp12N=0OLiq+g48=c8iQ zg_;>J*8S%9oNL2tf956QwC9U|xKmI}?8rd>ck~k9XW;B^ylOYRRAP(kx2IM`%V5jK zt=~tUqSI*f6H>&EMjJ+Xt9M+oTMm-nyYQ3Kx3+~s?;mk?G*-D^yv6O&SSj@-Qgj&u z(da>K*oZu69>%+Mw|vpf>B4}!-BB_pJ%*e@7k23)=Hjm4V#SnM|Jj5vHD&f;=`QUvHC1sn32|FZiLR`wXzO!=M=#OkM z<(r6YOl!PhIQ~G34NV*pf3SHN?;yT1HP-68^y}lcnC?x)Z>*W|+TzXD_<^Y*b@OZ< zEx<$q`mt%(EuqEYf-yFJ$&}o!IBb!5+~zkE;Glv4;*Y-r#&N=7_g{ z)jp?f4#qp4kKfAKP^Ns_Uu-dh{Y4k1HQoyCxC zJ6PmR#B|liQ@|Q*#(DA8d&vFYUoMP#=7uG|+Jp(!1~oBWc>F4D$l42)K6)mbf-#XB z2YLGxv=mh)p?vin$w<5D>G%7GZ!P@|ceuXlf@Zu$xnRqw)@wI@`PvpEhlm8E^{5`A zEk-NuA90&M^Du^+kv1$utY+@Uo0{)if1mg1!S~TIhR0N-#pxycmxQ-}DQ4Z?))wv33E_3P8c)rcUct3QT!#k4Rb}tiai*XDUEx6QqhN?fpWJ^y!&i^d@ zQY)KByHJq~kAN6>u>5V^N`$R`H0v#!$6I>L)2ib(c9(yCYK!?#7gBfOoQ-AYo>d%s z3v1<5s|d|Tk;Z#V<6cb2wh!6U77GKwG;E1mT!{g z>b_!<)N0_G)gx>k#=E8`%{ub^_xqpY@9~VZiS5Kyr1db~PQ9(s(32NuCgLYK!(&Z5 zQED=q8o5_S?hWJWnj`0#zp{B4Z@iv({`W7&qB_mpxuq1xmOecbkH<%GA!1--~AcG)YltB5$jfGF<}a7x&a=g zQ4f#fOLlZv>~@kKjn(b%!Omg_JldUxM|pVEOL)}2tyj5Ncr?KY7tpyd7E2-o9RhsQEIAO&BKbA5=bKk%}J`+?d)PVQ+X(j zp!Dh@I!#55;&qRYD+E<~e#yI*T4!8kHkQV85!vtvSPTzWq?n%3I%>x!>kqk^rs~Qr zFz=rXACukM(zN(+H`6*jx5Hh;r&BS3KkOppY2b5RM8Gt8x18El+@FS0jn{s!%Um$y zRYET}^%SCB7tiV{s!hj<6jOiHbhQ2#-Nnr5xP^c*He@EQOKQ9&$H~=uiMbef9>$Bo zOMX?yBkP_xf^7V?yBONgj14ee5$=d-H;yLlUPF<7MK%f7z&~d^@+LnpWMn zAMGW+nSr9xXJCvU>Ma(+kaf}HZ<1k*>mG7uFY$Z^Zb;}g4lrITzJAt&m^xAQI4gK6 z$Es|+cznaRPkwy7|B)|Z0=2!slfGgbjyAw}g?Pzu>y?th4}w$=zozIN#0PYk%-=&$^XH*c%>!()519XBL|8@A3$gp7$4B-oa5%7$_#e zGr)KQ`O6s@A6^r2%!@looX3rKlHad=IriZ|wSgIf_V{gxILlK0yn|BHVni>GS~?qr zXsdi#(P*|@)_HxDwy;Hci!QUVdvK1{)`dfEpWD;R*kTSu ziz;)F_7Xg5Anlw5uk!ZQbiHfyc!?Nn_p_f5%AM3C@67MEn94&%0@8XkAF7=`N-e1M zFm7&Gk#hSF6|0&1czAG;t{ppV(9|nGVaH+AJr6OgbMEPH_Dx!Kw1X|?;85{Utr`Fp&(8 z0OMuwGgIHLdu?WW7n_Iiy7*Q{YOSrcMp|Ku>6<8yFn8m%>$9tWccFb`^~V@37%N=E z#w3c+dFb-ViK008Kkzj2G+y2Q-S+gePv*o;M4r6r!w7U9B<3Tp0OKX@zg4UMW9wH3 zOUS%krrv5Z-Y{Om|7^rOv(lFv4K1c}vbf6nv`rSPm&tApSQn(^WKn89@-trcUUL15 z&#$_R+O|~2>)_v7kyQMbCBLHOjr?vVi$P3lyg+{G{Y9&4227f6^Dtg9f7fESs>7C7 ziY>-C`Tp^)bg9Cv|F0S1I&<%cAF_CF^7YeWO{>-~_n~aU(Qryg#v8MpyMCN~X7#lI zv=UZT9_y+x!g~RZRG--%#;fHgeY!f+xmU4HvS||T5MrAxog5=3EWnz5VT{;;6ajzs z)Ey<`HT6fUJ=^kVMu?BTDxeYfjuo$2+T2Xxz7VT<{x}i45Nl%R@#52k$n%fa1gs}d zbAFO{uy$j{lGNhEyWN-LuEV52Z(p87ulWg&9Mmgsg6Nimy!BgMVoDB{Y25K%3Gtsh z-Z*dlzjnv_>{4_C?s)&ORQ7VV-SHL`1-e|ncPX+3x~J+yktZK~0_jB-&eH_GHfnL^oH{gkzwu5=GwMEOJ!=7flr<()5`VgsLeHcX!9L6^VGTSR*l+U&v(WgO%@N4U%*Y|R}b^*!MtEY<4<;ZyRRMBs_yj-3#T|A?{aJuk%8$WfNnW2>t z@YUX4^;hl%Z_n)agMwVSJAS0u}^oPls2M@CSOc9-jQ+p41 zxFC;vcjkDx4V*d;9@tH@hWG_!2l5E`Quhe*j2Lk8{mS8L9_rTVCStge?q0fKX329S zTe2%q^QQR4ai)09w3b=gWpAR%^5u-hLs#28svw4!`He#= zJh%xTQ*%wJZ|6=(uz5ty64~$w$b<(kSN6^9*SuHB{+(?eixK02n0muP@~vO=_p`+$ z&J&*^t&HQpy#hUTcAoH930qs(+~t7zq8n_F(D_>1p7VF>6?yevj&7!w!Q3(WvE`<% zlyA!)&KILrp%$OP(}J?b-yiTzJwU|t#`=>Grs)(d%-+>)jK{n=Zp7{ z*5i*oNDuWMn9rm6;t_NI^Q8gu;4e^tuk{)}e%fHuTlV?*A3ItG^I;4u-3vs^)ks@u zf%<%B_UK*>%(XwAK3~h7%Xodnu;=>kaOxWqKP^GM;Aq10bnpT(7im4Z!2`2CyG@&R zb0@gj$7yM~f51h{G3GuD9=xSds`Fs+xW~%ow&Tv!W7a$$zbHtD}DD8^=`)2U)bHgUg7{Omr{dkAKSBlBeZs%@~-+b|GnIq-Y(XZZ| zP0tbAn0DzR?eeN}jT<$OS!x`B2k%a!rB)+`y?cLHzuwnc9BYjjG%aV&=0)N+r1e;& zXleVI@~&8aS;g7MA%wK877L%X@c8@OwYrxzBfF)G#3VQeOkb=O*rm~lH5JaBmyC75<3*J%B<8l(P@{OMwG7HQ?mOT-UrG2Q>;y-7gsQmyi}7F=1-`SCa1 zYnTJ6 z(t3o!!xJ9$Lz|7Aw|rZQ&0~ZRtLZTT9$Zvftq5s3w_pedi@$dOPj=W(E!>7v`~7ve zu{wW@z)$&p@?9bNyo)of9x0#B6-VAhgI~-QkG5bl8;C8TzKNp`H)I~hpE5>nh;mvk zzFLpw!95J@q@@E(Md*4Q;-8z?*zPrbTe~EA6zD#?dWC^KQ9bM|^-_jxN6AJh@4>SK z4&+LBrB@|dHkvG(g0%g=gt}c_kIQ*@4%&bedYwJS;PFciBd0M>&rhEb-3EE{G7>{@ zYM$5vPmhK0;GDlLl|ESPVcj3$q2D-PmnW_>Fa5a1!i}=MzI!cak1DSa-W#>bu}F_W z`jp(=W6`$TNvZh%pp4&-{w9N_Z;Krp(ax(@ir2`-nR1kv10iq^jx!i@#!X13BAL1^Qf{y8}RQf32gI8sTwno zBTi;K`ak3#x@^X7>W^aHX1R2&Vr#S({L4Ya4(3=W+DnDC+OYK$aB$q2fbezF}aKhr43~D zh&VIfFm3+4A{i43S6bgN?cZNVO~s(+7U@^VG9QE%+wFjQVnJe;EX)%PICTE=N%>#a z!{fnvt+cL}SDan2Z9jj{#jxhU{zI#nZi%uxR#XS03jg4QDzQ*;nS&}p5T8%7`= z^4%!T?!deW+_(A}s9{^q>jZZx^x z)-C_o)ul`+1zKDFaTCg4U2)iG(}n!oD||}L5#R5YTjP;4d_>(nvahM7h}Wl(NZ+jU%}wqA|t3}HD9qhLarb#?UAqSJG@txP5b5?mM_R+L@&Ea z`yxM*lO~G=@5@2^PPMZOuj7;&6E!R_*%DAKsGJ{Qv*} diff --git a/package.json b/package.json index ab851d5..314b9ba 100644 --- a/package.json +++ b/package.json @@ -27,8 +27,11 @@ "android-sms-gateway": "^3.0.0", "astro": "7.0.8", "iconify-icon": "^3.0.2", + "jszip": "^3.10.1", + "mammoth": "^1.12.0", "ofetch": "^1.5.1", "otplib": "12.0.1", + "pdfjs-dist": "^6.2.108", "thirty-two": "^1.0.2", "typescript": "^6.0.3", "validator": "^13.15.35" diff --git a/public/theme.css b/public/theme.css index ea6dcb7..0871fcc 100644 --- a/public/theme.css +++ b/public/theme.css @@ -719,3 +719,94 @@ progress::-webkit-progress-value { border-radius: 2px; } */ + +/* File input. Adapted from the commented block above, which referenced a + --bg-color token that does not exist; the real one is --color-bg. */ +input[type="file"] { + background-color: var(--color-bg); + border: 2px dashed var(--color-mg); + color: var(--color-fg); + cursor: pointer; + padding: var(--gutter-small); + width: 100%; +} + +input[type="file"]:hover, +input[type="file"]:focus-visible { + border-color: var(--color-accent); +} + +input[type="file"]::file-selector-button { + background-color: var(--color-accent); + border: none; + color: var(--color-bg); + cursor: pointer; + font-family: inherit; + font-size: inherit; + margin-inline-end: 1ch; + padding: 0.25em 1em; +} + +/* Cards on /tools. .tool-card was referenced by the placeholder markup but + never had any styles defined. */ +.tool-card { + border: 1px solid var(--color-mg); + display: block; + padding: var(--gutter-small); + text-decoration: none; + color: inherit; + transition: border-color 120ms ease; +} + +.tool-card:hover, +.tool-card:focus-visible { + border-color: var(--color-accent); +} + +.tool-card h3 { + color: var(--color-accent); +} + +.tool-card h3::before { + content: "./"; + color: var(--color-mg); +} + +.tool-card .tool-summary { + color: var(--color-fg); +} + +.tool-card .tool-description { + color: var(--color-mg); + font-size: var(--pt-small-pica); +} + +:root[data-reduce-motion="on"] .tool-card { + transition: none; +} + +@media (prefers-reduced-motion: reduce) { + :root:not([data-reduce-motion="off"]) .tool-card { + transition: none; + } +} + +:root[data-more-contrast="on"] .tool-card, +:root[data-more-contrast="on"] input[type="file"] { + border-color: var(--color-fg); +} + +:root[data-more-contrast="on"] .tool-card .tool-description { + color: var(--color-fg); +} + +@media (prefers-contrast: more) { + :root:not([data-more-contrast="off"]) .tool-card, + :root:not([data-more-contrast="off"]) input[type="file"] { + border-color: var(--color-fg); + } + + :root:not([data-more-contrast="off"]) .tool-card .tool-description { + color: var(--color-fg); + } +} diff --git a/src/layouts/BaseLayout.astro b/src/layouts/BaseLayout.astro index fbbfc67..cbe78c0 100644 --- a/src/layouts/BaseLayout.astro +++ b/src/layouts/BaseLayout.astro @@ -16,8 +16,8 @@ as="font" crossorigin="anonymous" /> - - + + Home
  • Contact
  • Projects
  • +
  • Tools
  • Git
  • diff --git a/src/lib/RateLimiter.ts b/src/lib/RateLimiter.ts new file mode 100644 index 0000000..d4efcc3 --- /dev/null +++ b/src/lib/RateLimiter.ts @@ -0,0 +1,70 @@ +/** + * Keyed sliding-window rate limiter. + * + * Generalises the per-phone and global windows in Otp.ts so endpoints other + * than the contact form can share one implementation. State is per-process and + * in memory, which is fine here: the container is read-only and restarts reset + * the window, which fails open rather than locking anyone out. + */ + +export class RateLimiter { + readonly #hits = new Map(); + readonly #limit: number; + readonly #windowMs: number; + #lastPrune = 0; + + constructor(limit: number, windowMs: number) { + this.#limit = limit; + this.#windowMs = windowMs; + } + + /** Drop expired entries so the map cannot grow without bound. */ + #prune(now: number): void { + // Pruning every call would be wasteful; once per window is plenty. + if (now - this.#lastPrune < this.#windowMs) return; + this.#lastPrune = now; + for (const [key, times] of this.#hits) { + const recent = times.filter((t) => now - t < this.#windowMs); + if (recent.length === 0) this.#hits.delete(key); + else this.#hits.set(key, recent); + } + } + + isLimited(key: string): boolean { + const now = Date.now(); + this.#prune(now); + const times = this.#hits.get(key); + if (!times) return false; + return times.filter((t) => now - t < this.#windowMs).length >= this.#limit; + } + + record(key: string): void { + const now = Date.now(); + this.#prune(now); + const recent = (this.#hits.get(key) ?? []).filter( + (t) => now - t < this.#windowMs, + ); + recent.push(now); + this.#hits.set(key, recent); + } + + /** Seconds until the oldest hit in the window expires. */ + retryAfterSeconds(key: string): number { + const times = this.#hits.get(key); + if (!times || times.length === 0) return 0; + const oldest = Math.min(...times); + return Math.max(1, Math.ceil((this.#windowMs - (Date.now() - oldest)) / 1000)); + } +} + +const GLOBAL_KEY = "__global__"; + +/** + * Resolve a client identifier. The app runs behind a reverse proxy, so + * X-Forwarded-For is the real source; clientAddress would be the proxy. + */ +export function clientKey(clientAddress: string): string { + return clientAddress || GLOBAL_KEY; +} + +export { GLOBAL_KEY }; diff --git a/src/lib/ats/dates.ts b/src/lib/ats/dates.ts new file mode 100644 index 0000000..c7e85b5 --- /dev/null +++ b/src/lib/ats/dates.ts @@ -0,0 +1,158 @@ +/** + * Date range parsing. + * + * Resume dates are where a surprising amount of ATS damage happens, because + * the formats that look tidiest to a human ("2020–22", "Jan '20") are the ones + * no parser handles. Where a format fails we say so explicitly rather than + * silently dropping the range. + */ + +import type { DateRange } from "./types.ts"; + +const MONTHS: Record = { + jan: 1, feb: 2, mar: 3, apr: 4, may: 5, jun: 6, + jul: 7, aug: 8, sep: 9, oct: 10, nov: 11, dec: 12, +}; + +const SEASONS: Record = { + spring: 3, summer: 6, fall: 9, autumn: 9, winter: 12, +}; + +const MONTH = String.raw`(?:jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|jun(?:e)?|jul(?:y)?|aug(?:ust)?|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)`; +const SEASON = String.raw`(?:spring|summer|fall|autumn|winter)`; +/** Hyphen, en dash, em dash, tilde or the word "to". */ +const SEP = String.raw`\s*(?:[-–—~]|\bto\b)+\s*`; +const CURRENT = String.raw`(?:present|current|now|ongoing|today)`; + +export const CURRENT_INDICATORS = /\b(present|current|now|ongoing|today)\b/i; + +const PATTERNS: { re: RegExp; kind: string }[] = [ + { re: new RegExp(`${MONTH}\\s*\\.?\\s*\\d{4}${SEP}(?:${MONTH}\\s*\\.?\\s*\\d{4}|${CURRENT})`, "i"), kind: "monthYear" }, + { re: new RegExp(`\\d{1,2}\\/\\d{4}${SEP}(?:\\d{1,2}\\/\\d{4}|${CURRENT})`, "i"), kind: "numericMonthYear" }, + { re: new RegExp(`${SEASON}\\s*\\d{4}${SEP}(?:${SEASON}\\s*\\d{4}|${CURRENT})`, "i"), kind: "season" }, + { re: new RegExp(`\\b(?:19|20)\\d{2}${SEP}(?:(?:19|20)\\d{2}|${CURRENT})\\b`, "i"), kind: "yearOnly" }, +]; + +/** Formats that look reasonable but that no parser examined handles. */ +const KNOWN_BAD: { re: RegExp; problem: string }[] = [ + { + re: /\b(?:19|20)\d{2}\s*[-–—]\s*\d{2}\b(?!\d)/, + problem: + "A two-digit end year (“2020–22”) matches no date pattern, so the range is dropped entirely.", + }, + { + re: /['’]\d{2}\b/, + problem: + "Apostrophe years (“Jan ’20”) are not supported by any parser examined; write the year in full.", + }, +]; + +function pad(month: number): string { + return String(month).padStart(2, "0"); +} + +/** Normalise one endpoint to YYYY-MM or YYYY. */ +function normaliseEndpoint(raw: string): string | null { + const text = raw.trim().toLowerCase(); + if (CURRENT_INDICATORS.test(text)) return null; + + const monthYear = text.match(new RegExp(`(${MONTH})\\s*\\.?\\s*((?:19|20)\\d{2})`, "i")); + if (monthYear) { + const key = monthYear[1].slice(0, 3).toLowerCase(); + const month = MONTHS[key]; + return month ? `${monthYear[2]}-${pad(month)}` : monthYear[2]; + } + + const seasonYear = text.match(new RegExp(`(${SEASON})\\s*((?:19|20)\\d{2})`, "i")); + if (seasonYear) { + const month = SEASONS[seasonYear[1].toLowerCase()]; + return `${seasonYear[2]}-${pad(month)}`; + } + + const numeric = text.match(/(\d{1,2})\/((?:19|20)\d{2})/); + if (numeric) { + const month = Number(numeric[1]); + return month >= 1 && month <= 12 + ? `${numeric[2]}-${pad(month)}` + : numeric[2]; + } + + const yearOnly = text.match(/\b((?:19|20)\d{2})\b/); + if (yearOnly) return yearOnly[1]; + + return null; +} + +/** Find and parse the first date range in a line of text. */ +export function parseDateRange(text: string): DateRange | null { + for (const bad of KNOWN_BAD) { + if (bad.re.test(text)) { + const raw = text.match(bad.re)?.[0] ?? text.trim(); + return { + raw, + start: null, + end: null, + isCurrent: false, + parsed: false, + problem: bad.problem, + }; + } + } + + for (const { re } of PATTERNS) { + const match = text.match(re); + if (!match) continue; + + const raw = match[0]; + const halves = raw.split(new RegExp(SEP, "i")); + const startRaw = halves[0] ?? ""; + const endRaw = halves.slice(1).join(" "); + const isCurrent = CURRENT_INDICATORS.test(endRaw); + + const range: DateRange = { + raw: raw.trim(), + start: normaliseEndpoint(startRaw), + end: isCurrent ? null : normaliseEndpoint(endRaw), + isCurrent, + parsed: true, + }; + + if (!range.start) { + range.parsed = false; + range.problem = "The start of this range could not be interpreted."; + } else if (/^\d{4}$/.test(range.start)) { + range.problem = + "Only a year was given, so month precision is lost and tenure is rounded."; + } + return range; + } + + // A lone date with no range, e.g. a graduation year. + const single = normaliseEndpoint(text); + if (single) { + return { + raw: text.match(/\b(?:19|20)\d{2}\b/)?.[0] ?? text.trim(), + start: single, + end: single, + isCurrent: false, + parsed: true, + }; + } + + return null; +} + +export function hasAnyDate(text: string): boolean { + return /\b(?:19|20)\d{2}\b/.test(text) || CURRENT_INDICATORS.test(text); +} + +/** + * Detect a date range broken across two lines, which Textkernel flags as 418 + * and RChilli as 4109. Both treat it as fatal. + */ +export function isVerticalDateRange(line: string, next: string | undefined): boolean { + if (!next) return false; + const endsOpen = /\b(?:19|20)\d{2}\s*[-–—~]\s*$/.test(line.trim()); + const nextIsDate = /^\s*(?:(?:19|20)\d{2}|present|current)\b/i.test(next.trim()); + return endsOpen && nextIsDate; +} diff --git a/src/lib/ats/diagnostics.ts b/src/lib/ats/diagnostics.ts new file mode 100644 index 0000000..7360597 --- /dev/null +++ b/src/lib/ats/diagnostics.ts @@ -0,0 +1,605 @@ +/** + * The diagnostic catalogue. + * + * These checks are ported from what real commercial parsers publish, not + * invented: Textkernel (formerly Sovren) documents both its document-conversion + * result codes and its full ResumeQuality assessment list, RChilli publishes a + * near-identical taxonomy under different numbering, and Greenhouse documents + * its own failure list. Each finding cites its source, because an opinionated + * tool is only credible if its opinions are attributable. + * + * Deliberately NOT a score. Textkernel's own documentation says its quality + * output "should NEVER IN ANY SENSE WHATSOEVER be used as an indication that + * the Parser has failed", and notes most resumes trip at least one flag. + * Rolling these into a single number would misrepresent them. + */ + +import { isVerticalDateRange } from "./dates.ts"; +import { SECTION_LABELS } from "./sections.ts"; +import type { + DocFacts, + Finding, + Line, + ParsedResume, + Section, +} from "./types.ts"; +import type { UnrecognisedHeading } from "./sections.ts"; + +/** Characters that are ordinary in a resume; anything else counts as garbage. */ +const ORDINARY = /[\p{Letter}\p{Number}\s.,;:!?'"()[\]{}@#&%$/\\+=*_•·—–-]/u; + +function symbolRatio(text: string): number { + if (text.length === 0) return 0; + let symbols = 0; + for (const char of text) if (!ORDINARY.test(char)) symbols++; + return symbols / text.length; +} + +function averageWordLength(text: string): number { + const words = text.split(/\s+/).filter((w) => w.length > 0); + if (words.length === 0) return 0; + return words.reduce((sum, w) => sum + w.length, 0) / words.length; +} + +const LETTER_SPACED = /\b(?:[A-Za-z]\s){3,}[A-Za-z]\b/; + +const PII_PATTERNS: [RegExp, string, string][] = [ + [/\bpassport\b/i, "passport number", "121"], + [/\b(?:driver'?s?|driving)\s+licen[cs]e\b/i, "driving licence", "122"], + [/\bmarital\s+status\b/i, "marital status", "123"], + [/\b(?:date\s+of\s+birth|d\.?o\.?b\.?)\b/i, "date of birth", "124"], +]; + +export interface DiagnosticInput { + facts: DocFacts; + lines: Line[]; + text: string; + sections: Section[]; + unrecognised: UnrecognisedHeading[]; + resume: ParsedResume; + columnsChangeReading: boolean; +} + +export function runDiagnostics(input: DiagnosticInput): Finding[] { + const { facts, lines, text, sections, unrecognised, resume } = input; + const findings: Finding[] = []; + const add = (f: Finding) => findings.push(f); + + // ---- Tier 1: can the text be read at all? ---- + + if (facts.encrypted) { + add({ + id: "encrypted", + severity: "fatal", + title: "The file is password protected", + cause: + "Encrypted documents cannot be opened for text extraction at all.", + fix: "Save an unprotected copy and submit that instead.", + citation: "Textkernel document conversion code ovIsEncrypted", + }); + } + + if (facts.textItemCount === 0) { + const outlined = facts.vectorOpCount > 50; + add({ + id: "no-text-layer", + severity: "fatal", + title: "This document contains no readable text", + cause: outlined + ? "The page is drawn as vector shapes with no text behind them. This is what happens when a design tool converts text to outlines or curves on export — it looks identical on screen but there are no characters in the file." + : "No text layer was found. The document is most likely a scan or an exported image.", + fix: "Export again from the original document with live text. Do not flatten, rasterise, or convert text to outlines.", + citation: + "Textkernel codes ovIsImage / ovNoText — OCR is opt-in and off by default, so most systems see nothing here", + detail: outlined + ? `${facts.vectorOpCount} vector drawing operations and zero text runs.` + : undefined, + }); + // Nothing downstream is meaningful without text. + return findings; + } + + if (facts.truncated) { + add({ + id: "truncated", + severity: "major", + title: `Only the first ${facts.pagesParsed} pages were read`, + cause: `The document has ${facts.pageCount} pages. Parsers cap how much they will process, and anything past the cap is discarded.`, + fix: "Move the most important content to the first two pages.", + citation: "Textkernel code ovTruncated / PdfMaxPages", + }); + } + + const ratio = symbolRatio(text); + if (ratio >= 0.05) { + add({ + id: "garbage-text", + severity: "fatal", + title: "The extracted text is largely unreadable characters", + cause: + "Around " + Math.round(ratio * 100) + "% of the extracted characters are neither letters nor digits. This almost always means an embedded font has a missing or incorrect ToUnicode map, so the file draws the right shapes but reports the wrong characters.", + fix: "Re-export the PDF with standard fonts embedded, or submit a DOCX instead.", + citation: "Textkernel code ovProbableGarbageInText (threshold: 5%)", + }); + } + + const avgWord = averageWordLength(text); + if (avgWord > 20) { + add({ + id: "long-words", + severity: "major", + title: "Words are running together", + cause: `Average word length is ${avgWord.toFixed(1)} characters. Spaces between runs of text are being lost, usually because the document uses unusual character spacing.`, + fix: "Re-export from the original document, avoiding manual letter spacing.", + citation: "Textkernel code ovAvgWordLengthGreaterThan20", + }); + } else if (avgWord > 0 && avgWord < 4) { + add({ + id: "short-words", + severity: "major", + title: "Text is being broken into fragments", + cause: `Average word length is only ${avgWord.toFixed(1)} characters, which suggests words are being split apart during extraction.`, + fix: "Avoid letter spacing and justified text; re-export and check the result here again.", + citation: "Textkernel code ovAvgWordLengthLessThan4", + }); + } + + if (text.length > 500 && lines.length < text.length / 300) { + add({ + id: "few-line-breaks", + severity: "major", + title: "Very few line breaks were found", + cause: + "The document produced a large amount of text but almost no line structure, so entries will run into one another.", + fix: "Check that the layout uses real paragraphs rather than a single text frame.", + citation: "Textkernel code ovTooFewLineBreaks", + }); + } + + const shortLines = lines.filter((l) => l.text.trim().length < 15).length; + if (lines.length >= 10 && shortLines / lines.length > 0.6) { + add({ + id: "short-lines", + severity: "minor", + title: "Most lines are very short", + cause: + "A high proportion of very short lines usually means a narrow column or a table is fragmenting the text.", + fix: "Widen the text area, or move content out of narrow table cells.", + citation: "Textkernel code ovLinesSeemTooShort", + detail: `${shortLines} of ${lines.length} lines are under 15 characters.`, + }); + } + + if (facts.nonEmbeddedFonts.length > 0) { + add({ + id: "fonts-not-embedded", + severity: "minor", + title: "Some fonts are not embedded", + cause: + "When a font is not embedded, character mapping depends on what the reading system substitutes, which can silently corrupt the extracted text.", + fix: "Embed all fonts on export, or stick to standard fonts.", + citation: "PDF text extraction depends on the font's ToUnicode CMap", + detail: facts.nonEmbeddedFonts.slice(0, 6).join(", "), + }); + } + + if (facts.hadLigatures) { + add({ + id: "ligatures", + severity: "info", + title: "Ligature characters were present", + cause: + "Combined glyphs such as “fi” are stored as a single character. Parsers that do not normalise them turn “proficiency” into “proficiency”, which breaks keyword matching.", + fix: "Usually harmless, but if keyword matching is important, disable ligatures on export.", + citation: "pdf.js normalises U+FB00–U+FB04; many parsers do not", + }); + } + + const spacedLines = lines.filter((l) => LETTER_SPACED.test(l.text)); + if (spacedLines.length > 0) { + add({ + id: "letter-spaced", + severity: "major", + title: "Letter-spaced text was found", + cause: + "Wide letter spacing makes the extractor insert a space between every character, so a heading becomes “E X P E R I E N C E” and matches no known section name.", + fix: "Remove the letter spacing and use a larger or bolder font instead.", + citation: "Greenhouse lists “resumes with spaced-out letters” as a parse failure", + detail: spacedLines.slice(0, 3).map((l) => `“${l.text}”`).join(", "), + }); + } + + // ---- Tier 2: layout ---- + + if (input.columnsChangeReading) { + add({ + id: "columns", + severity: "fatal", + title: "A column layout changes what the text says", + cause: + "A PDF stores no reading order, so a parser has to guess where columns are. Reading straight across merges the two columns line by line — compare the “Naive top-to-bottom” and “Column-aware” views above to see the difference on your own document.", + fix: "Use a single-column layout. It is the single highest-impact change you can make.", + citation: + "Textkernel code 433 / RChilli 4108. Textkernel reports ≥15% of CVs use columns, and their rule-based detector only got 60% of separators right before they switched to machine learning", + }); + } + + if (facts.imageCount > 0) { + add({ + id: "images", + severity: "minor", + title: `${facts.imageCount} image${facts.imageCount === 1 ? "" : "s"} found`, + cause: + "Images carry no text. If any of your content — a skills bar chart, a logo containing your name, a graphic header — lives inside an image, none of it is read.", + fix: "Make sure nothing that matters exists only as a picture. Rating bars in particular convey nothing to a parser.", + citation: "Workday: “use resumes that don't have images or image-based styles”", + }); + } + + if (facts.docx) { + const { headerText, footerText, textBoxText, smartArtText, tableCount, usesColumnSections } = facts.docx; + + if (headerText.length > 0 || footerText.length > 0) { + const sample = [...headerText, ...footerText].slice(0, 4).join(" · "); + add({ + id: "docx-header-footer", + severity: "fatal", + title: "Text in the page header or footer will not be read", + cause: + "Word stores headers and footers as separate parts of the file. Common extractors never open them, so this text is invisible to the parser even though you can see it on the page.", + fix: "Move this text into the body of the document, at the top of the first page.", + citation: + "mammoth.js issues #9 and #266; Greenhouse lists “name/contact info in headers” as a parse failure", + detail: `Found in header/footer: ${sample}`, + }); + } + + if (textBoxText.length > 0) { + add({ + id: "docx-textbox", + severity: "major", + title: "Text boxes move your content out of order", + cause: + "Text box content is extracted after the paragraph that contains the box, not where it appears on the page. Contact details in a text box typically land in the middle of the document.", + fix: "Replace text boxes with ordinary paragraphs.", + citation: "mammoth.js documents this relocation behaviour", + detail: textBoxText.slice(0, 4).join(" · "), + }); + } + + if (smartArtText.length > 0) { + add({ + id: "docx-smartart", + severity: "major", + title: "SmartArt content is not extracted", + cause: + "SmartArt lives in a separate diagram part that text extractors essentially never read.", + fix: "Rewrite this content as normal text.", + citation: "SmartArt is stored in word/diagrams/, outside the document body", + detail: smartArtText.slice(0, 4).join(" · "), + }); + } + + if (tableCount > 0) { + add({ + id: "docx-tables", + severity: "major", + title: `${tableCount} table${tableCount === 1 ? "" : "s"} found`, + cause: + "Table cells are flattened in the order they appear in the file, which is not necessarily the order you read them. Tables used to fake a column layout are a common cause of job titles merging with the wrong dates.", + fix: "Use ordinary paragraphs with tab stops or spacing instead of tables.", + citation: "Greenhouse lists “complex layouts with tables” as a parse failure", + }); + } + + if (usesColumnSections) { + add({ + id: "docx-real-columns", + severity: "info", + title: "Real Word columns are in use, and that is fine", + cause: + "Word's own column feature only affects how the text is displayed. The underlying document order is still linear, so parsers read it correctly.", + fix: "No change needed. This is the safe way to do columns — unlike tables or text boxes.", + citation: "w:cols is a rendering hint; paragraph order in the XML is unaffected", + }); + } + } + + // ---- Tier 3: structure ---- + + const present = new Set(sections.filter((s) => !s.inferred).map((s) => s.id)); + + if (present.size === 0) { + add({ + id: "no-sections", + severity: "fatal", + title: "No section headings were recognised", + cause: + "Nothing in the document matched a known section name, so a parser has no way to tell work history from education from skills.", + fix: "Add plain headings: Experience, Education, Skills.", + citation: "Textkernel code 412 — no sections were found in the resume", + }); + } else { + if (!present.has("experience")) { + add({ + id: "no-experience", + severity: "fatal", + title: "No work history section was found", + cause: + "No heading matched the work history vocabulary, so employment cannot be attributed to a section.", + fix: "Title the section exactly “Experience”, “Work Experience” or “Employment History”.", + citation: "Textkernel code 413", + }); + } + if (!present.has("education")) { + add({ + id: "no-education", + severity: "major", + title: "No education section was found", + cause: "No heading matched the education vocabulary.", + fix: "Title the section “Education”.", + citation: "Textkernel code 414", + }); + } + } + + for (const miss of unrecognised) { + add({ + id: `unrecognised-heading-${miss.lineIndex}`, + severity: "major", + title: `“${miss.text}” was not recognised as a section heading`, + cause: `It looks like a heading, but it matches no section name a parser knows. As a result the ${miss.absorbedLines} line${miss.absorbedLines === 1 ? "" : "s"} that follow were absorbed into your ${SECTION_LABELS[miss.absorbedInto]} section instead of forming their own.`, + fix: "Rename it to a conventional heading. Creative section names are the most common avoidable parsing failure.", + citation: "Textkernel codes 325 / 415; RChilli 4002", + }); + } + + const duplicates = new Map(); + for (const section of sections) { + if (section.inferred) continue; + duplicates.set(section.id, (duplicates.get(section.id) ?? 0) + 1); + } + for (const [id, count] of duplicates) { + if (count > 1) { + add({ + id: `duplicate-section-${id}`, + severity: "minor", + title: `The ${SECTION_LABELS[id as keyof typeof SECTION_LABELS]} section appears ${count} times`, + cause: + "Repeated headings of the same type make it ambiguous which block holds the real content; parsers commonly keep only one.", + fix: "Merge them into a single section.", + citation: "Textkernel code 323", + }); + } + } + + const empty = sections.filter((s) => !s.inferred && s.lines.length === 0); + for (const section of empty) { + add({ + id: `empty-section-${section.id}`, + severity: "minor", + title: `The “${section.heading}” section came out empty`, + cause: + "A heading was found but no content followed it in the extracted text. The content is usually there visually but sitting somewhere the parser cannot reach, such as a table cell, text box or image.", + fix: "Check where that content actually lives in your document.", + citation: "Textkernel code 324", + }); + } + + // Contact details should sit at the very top of the extracted text. + const contactLine = Math.min( + ...[resume.basics.email?.sourceLine, resume.basics.phone?.sourceLine] + .filter((n): n is number => typeof n === "number"), + ); + if (Number.isFinite(contactLine) && contactLine > 5) { + add({ + id: "contact-not-at-top", + severity: "major", + title: "Contact details are not at the top of the extracted text", + cause: `Your email or phone number first appears on line ${contactLine + 1} of the extracted text, not in the opening lines. Parsers expect a contact block at the top and may not associate these details with you.`, + fix: "Put your name, email and phone in the first few lines of the document body.", + citation: "Textkernel code 311", + }); + } + + if (!resume.basics.email && !resume.basics.phone) { + add({ + id: "no-contact", + severity: "fatal", + title: "Neither an email address nor a phone number was found", + cause: + "No recognisable contact details appear anywhere in the extracted text. If they are visible on the page, they are somewhere the parser cannot read — a header, a text box, or an image.", + fix: "Put both in plain text at the top of the document body.", + citation: "Textkernel code 441", + }); + } else { + if (!resume.basics.email) { + add({ + id: "no-email", + severity: "major", + title: "No email address was found", + cause: "No text matching an email address was extracted.", + fix: "Add your email as plain text, not as a link graphic.", + citation: "Textkernel code 211", + }); + } + if (!resume.basics.phone) { + add({ + id: "no-phone", + severity: "minor", + title: "No phone number was found", + cause: + "No text matching a phone number was extracted. Unusual formatting or spacing can also prevent a match.", + fix: "Write it plainly, for example (555) 123-4567.", + citation: "Textkernel code 212", + }); + } + } + + if (!resume.basics.name) { + add({ + id: "no-name", + severity: "major", + title: "No candidate name was identified", + cause: + "The strongest name candidate scored too low, usually because the top of the document contains no plain, prominent line of just a name — or because the name is in a header or an image.", + fix: "Put your name on its own line at the very top, in plain text.", + citation: "Textkernel code 302", + }); + } + + for (let i = 0; i < lines.length; i++) { + if (isVerticalDateRange(lines[i].text, lines[i + 1]?.text)) { + add({ + id: `vertical-dates-${i}`, + severity: "fatal", + title: "A date range is split across two lines", + cause: + "The start and end of the range end up on separate lines, so the range cannot be reconstructed and the role may be recorded with no dates at all.", + fix: "Keep the whole range on one line.", + citation: "Textkernel code 418; RChilli 4109", + detail: `“${lines[i].text}” / “${lines[i + 1]?.text}”`, + }); + break; + } + } + + if (resume.work.length > 0) { + const undated = resume.work.filter((w) => !w.dates).length; + if (undated === resume.work.length) { + add({ + id: "no-job-dates", + severity: "fatal", + title: "No dates were found for any role", + cause: + "Employment with no dates cannot be placed on a timeline, so tenure and recency cannot be calculated.", + fix: "Add a date range to every role, for example “Jan 2020 – Present”.", + citation: "Textkernel code 419", + }); + } else if (undated > 0) { + add({ + id: "some-jobs-undated", + severity: "major", + title: `${undated} of ${resume.work.length} roles have no readable dates`, + cause: + "The date range for these roles could not be parsed, often because of an unusual format or because the dates sit in a separate column.", + fix: "Use a consistent “Mon YYYY – Mon YYYY” format on the same line as the role.", + citation: "Textkernel codes 224 / 225", + }); + } + + for (const entry of resume.work) { + if (entry.dates?.problem && entry.dates.parsed === false) { + add({ + id: `bad-date-${entry.dates.raw}`, + severity: "major", + title: `The date “${entry.dates.raw}” could not be read`, + cause: entry.dates.problem, + fix: "Write both endpoints in full, for example “Jan 2020 – Mar 2022”.", + citation: "Date handling verified against published parser patterns", + }); + } + } + + const untitled = resume.work.filter((w) => !w.title).length; + if (untitled > 0) { + add({ + id: "jobs-without-titles", + severity: "major", + title: `${untitled} role${untitled === 1 ? " has" : "s have"} no recognisable job title`, + cause: + "No line in the entry matched a known job title pattern. Abbreviated titles such as “Sr. Acct Exec” are a frequent cause.", + fix: "Write titles out in full, on their own line.", + citation: "Textkernel code 221; Greenhouse lists incomplete job titles", + }); + } + + const unnamed = resume.work.filter((w) => !w.organisation).length; + if (unnamed > 0) { + add({ + id: "jobs-without-companies", + severity: "major", + title: `${unnamed} role${unnamed === 1 ? " has" : "s have"} no recognisable employer`, + cause: + "The entry did not contain a separate line that reads as an organisation name.", + fix: "Put the employer on the same line as the title or directly beneath it.", + citation: "Textkernel code 222", + }); + } + + if (resume.work.length > 30) { + add({ + id: "too-many-jobs", + severity: "minor", + title: `${resume.work.length} separate roles were detected`, + cause: + "Either the resume genuinely lists an unusual number of roles, or entry boundaries are being detected incorrectly and single roles are splitting apart.", + fix: "Check the extracted entries below and consolidate older roles.", + citation: "Textkernel code 331 (threshold: 30)", + }); + } + } + + // ---- Tier 4: content and privacy ---- + + if (present.has("references")) { + add({ + id: "references-section", + severity: "info", + title: "A references section is present", + cause: + "References take up space that parsers do not use and recruiters do not read at this stage.", + fix: "Remove it and use the space for work history.", + citation: "Textkernel code 111", + }); + } + + const educationSection = sections.find((s) => s.id === "education"); + if (educationSection && educationSection.lines.some((l) => /\b(?:19|20)\d{2}\b/.test(l.text))) { + add({ + id: "education-dates", + severity: "info", + title: "Your education section contains dates", + cause: + "This one is counterintuitive: at least one major parser explicitly advises against dates in education, because they invite age inference and add nothing a recruiter needs.", + fix: "Optional. Consider removing graduation years.", + citation: "Textkernel code 233 — “Do not put dates in your education section”", + }); + } + + for (const [pattern, label, code] of PII_PATTERNS) { + if (pattern.test(text)) { + add({ + id: `pii-${code}`, + severity: "minor", + title: `Your resume mentions your ${label}`, + cause: + "This is sensitive personal information that will be stored in the employer's database indefinitely. It is not needed to assess your application, and in many countries employers would rather not hold it.", + fix: `Remove the ${label} unless a specific application explicitly requires it.`, + citation: `Textkernel code ${code}`, + }); + } + } + + if (facts.kind === "pdf") { + add({ + id: "is-pdf", + severity: "info", + title: "This is a PDF", + cause: + "At least one major parser flags PDFs purely for being PDFs, because DOCX preserves logical reading order while PDF only preserves visual position. That said, a well-built single-column PDF parses fine.", + fix: "No change needed if the extracted text above reads correctly. If it does not, try submitting a DOCX.", + citation: "Textkernel code 300 — “document was PDF”", + }); + } + + return findings; +} + +const SEVERITY_ORDER = { fatal: 0, major: 1, minor: 2, info: 3 } as const; + +export function sortFindings(findings: Finding[]): Finding[] { + return [...findings].sort( + (a, b) => SEVERITY_ORDER[a.severity] - SEVERITY_ORDER[b.severity], + ); +} diff --git a/src/lib/ats/extractDocx.ts b/src/lib/ats/extractDocx.ts new file mode 100644 index 0000000..29a8a3d --- /dev/null +++ b/src/lib/ats/extractDocx.ts @@ -0,0 +1,170 @@ +/** + * DOCX extraction, done twice on purpose. + * + * Pass 1 uses mammoth, whose documented limitations are exactly the behaviour + * we want to simulate: it does not read word/header*.xml or word/footer*.xml at + * all, and it moves text-box content to *after* the paragraph that contains it. + * That is why "my phone number vanished" is such a common complaint. + * + * Pass 2 opens the raw zip and looks for precisely the parts pass 1 dropped. + * Diffing the two is what lets us say "this specific text will not be read" + * rather than offering generic advice. + * + * Worth knowing, because it is counterintuitive: real Word column layout + * (w:cols) parses fine — the underlying XML is already in linear order. It is + * columns faked with tables or text boxes that break. + */ + +import JSZip from "jszip"; +import mammoth from "mammoth"; +import type { DocxFacts, TextItem } from "./types.ts"; + +/** Synthetic layout metrics, since DOCX carries no coordinates. */ +const LINE_HEIGHT = 14; +const CHAR_WIDTH = 5; +const FONT_SIZE = 11; +const TOP_Y = 1000; + +const ENTITIES: Record = { + "&": "&", + "<": "<", + ">": ">", + """: '"', + "'": "'", + "'": "'", + " ": " ", +}; + +function decodeEntities(text: string): string { + return text + .replace(/&(?:amp|lt|gt|quot|#39|apos|nbsp);/g, (m) => ENTITIES[m] ?? m) + .replace(/&#(\d+);/g, (_, code) => String.fromCharCode(Number(code))); +} + +function stripTags(html: string): string { + return decodeEntities(html.replace(/<[^>]+>/g, "")).replace(/\s+/g, " ").trim(); +} + +/** Pull the text out of every element in a Word XML part. */ +function wordTextOf(xml: string): string[] { + const out: string[] = []; + for (const match of xml.matchAll(/]*)?>([\s\S]*?)<\/w:t>/g)) { + const text = decodeEntities(match[1]).trim(); + if (text) out.push(text); + } + return out; +} + +/** SmartArt diagram parts store their text in DrawingML elements. */ +function drawingTextOf(xml: string): string[] { + const out: string[] = []; + for (const match of xml.matchAll(/]*)?>([\s\S]*?)<\/a:t>/g)) { + const text = decodeEntities(match[1]).trim(); + if (text) out.push(text); + } + return out; +} + +export class CorruptDocxError extends Error {} + +async function inspectRawParts(data: Uint8Array): Promise { + let zip: JSZip; + try { + zip = await JSZip.loadAsync(data); + } catch { + throw new CorruptDocxError("DOCX archive could not be opened"); + } + + const headerText: string[] = []; + const footerText: string[] = []; + const smartArtText: string[] = []; + const textBoxText: string[] = []; + let tableCount = 0; + let usesColumnSections = false; + + for (const name of Object.keys(zip.files)) { + if (zip.files[name].dir) continue; + + if (/^word\/header\d*\.xml$/.test(name)) { + headerText.push(...wordTextOf(await zip.files[name].async("string"))); + } else if (/^word\/footer\d*\.xml$/.test(name)) { + footerText.push(...wordTextOf(await zip.files[name].async("string"))); + } else if (/^word\/diagrams\/data\d*\.xml$/.test(name)) { + smartArtText.push(...drawingTextOf(await zip.files[name].async("string"))); + } else if (name === "word/document.xml") { + const xml = await zip.files[name].async("string"); + tableCount = [...xml.matchAll(/]/g)].length; + usesColumnSections = /]*w:num="?[2-9]/.test(xml); + for (const box of xml.matchAll(/([\s\S]*?)<\/w:txbxContent>/g)) { + textBoxText.push(...wordTextOf(box[1])); + } + } + } + + return { + headerText, + footerText, + textBoxText, + smartArtText, + tableCount, + usesColumnSections, + }; +} + +export interface DocxExtraction { + items: TextItem[]; + docx: DocxFacts; +} + +export async function extractDocx(data: Uint8Array): Promise { + const buffer = Buffer.from(data); + + let html: string; + try { + html = (await mammoth.convertToHtml({ buffer })).value; + } catch (err) { + throw new CorruptDocxError( + (err as Error)?.message ?? "DOCX could not be read", + ); + } + + const items: TextItem[] = []; + let row = 0; + + // Walk block elements and table-row ends together. Rows get extra vertical + // spacing so that entry splitting downstream sees one entry per row, which + // is how a table-based experience section is actually laid out. + for (const match of html.matchAll( + /<(p|h[1-6]|li)\b[^>]*>([\s\S]*?)<\/\1>|<\/tr>/g, + )) { + if (match[1] === undefined) { + row += 1; // end of a table row + continue; + } + const tag = match[1]; + const inner = match[2]; + const text = stripTags(inner); + if (text.length === 0) continue; + + // Headings and fully-emphasised paragraphs read as bold downstream, which + // is what section detection and entry splitting rely on. + const bold = + /^h[1-6]$/.test(tag) || + /^\s*<(strong|b)>[\s\S]*<\/\1>\s*$/.test(inner.trim()); + + items.push({ + page: 1, + x: 0, + y: TOP_Y - row * LINE_HEIGHT, + width: text.length * CHAR_WIDTH, + height: FONT_SIZE, + fontName: bold ? "docx-bold" : "docx-regular", + bold, + hasEOL: true, + text, + }); + row++; + } + + return { items, docx: await inspectRawParts(data) }; +} diff --git a/src/lib/ats/extractPdf.ts b/src/lib/ats/extractPdf.ts new file mode 100644 index 0000000..402e333 --- /dev/null +++ b/src/lib/ats/extractPdf.ts @@ -0,0 +1,205 @@ +/** + * PDF text extraction via pdf.js. + * + * Two behaviours here are load-bearing and easy to get wrong: + * + * 1. `getOperatorList()` must run BEFORE `getTextContent()`. Fonts are only + * resolved into `commonObjs` while building the operator list; calling + * `commonObjs.get()` first throws "Requesting object that isn't resolved + * yet". Without it there is no bold information at all, because + * `textContent.styles` only reports a generic fontFamily. + * + * 2. `hasEOL` often lands on a zero-width empty item that precedes the text of + * the next line, rather than on the text run itself. Filtering empty items + * out before grouping therefore destroys every line break. + */ + +import { createRequire } from "node:module"; +import path from "node:path"; +import type { DocFacts, TextItem } from "./types.ts"; + +const require = createRequire(import.meta.url); + +/** Mirrors Textkernel's `ovTruncated` / `PdfMaxPages` behaviour. */ +export const MAX_PAGES = 10; + +let pdfjsPromise: Promise | null = + null; + +function loadPdfjs() { + pdfjsPromise ??= import("pdfjs-dist/legacy/build/pdf.mjs"); + return pdfjsPromise; +} + +/** Resolve pdf.js's bundled font/cmap assets out of node_modules. */ +function assetDirs() { + const root = path.dirname(require.resolve("pdfjs-dist/package.json")); + return { + standardFontDataUrl: path.join(root, "standard_fonts") + path.sep, + cMapUrl: path.join(root, "cmaps") + path.sep, + }; +} + +export interface PdfExtraction { + items: TextItem[]; + facts: Omit; +} + +export class EncryptedPdfError extends Error {} +export class CorruptPdfError extends Error {} + +export async function extractPdf(data: Uint8Array): Promise { + const pdfjs = await loadPdfjs(); + const { standardFontDataUrl, cMapUrl } = assetDirs(); + + let loadingTask; + let doc; + try { + loadingTask = pdfjs.getDocument({ + data, + standardFontDataUrl, + cMapUrl, + cMapPacked: true, + // No canvas in this runtime, and no system font lookups. + isOffscreenCanvasSupported: false, + useSystemFonts: false, + disableFontFace: true, + // We normalise ourselves so ligatures can be detected before folding. + stopAtErrors: false, + }); + doc = await loadingTask.promise; + } catch (err) { + const name = (err as { name?: string })?.name; + if (name === "PasswordException") { + throw new EncryptedPdfError("PDF is password protected"); + } + throw new CorruptPdfError( + (err as Error)?.message ?? "PDF could not be opened", + ); + } + + const pageCount = doc.numPages; + const pagesParsed = Math.min(pageCount, MAX_PAGES); + + const items: TextItem[] = []; + const fonts = new Set(); + const nonEmbeddedFonts = new Set(); + let imageCount = 0; + let vectorOpCount = 0; + let hadLigatures = false; + + const LIGATURES = /[ff-st]/; + + const pageSizes: { width: number; height: number }[] = []; + + for (let pageNum = 1; pageNum <= pagesParsed; pageNum++) { + const page = await doc.getPage(pageNum); + const viewport = page.getViewport({ scale: 1 }); + pageSizes.push({ width: viewport.width, height: viewport.height }); + + // Resolves fonts into commonObjs, and gives us image/vector counts. + const ops = await page.getOperatorList(); + for (let i = 0; i < ops.fnArray.length; i++) { + const fn = ops.fnArray[i]; + if ( + fn === pdfjs.OPS.paintImageXObject || + fn === pdfjs.OPS.paintInlineImageXObject || + fn === pdfjs.OPS.paintImageMaskXObject + ) { + // Args carry intrinsic dimensions on most builds; when they don't, + // count the image rather than silently ignoring it. + const args = ops.argsArray[i] as unknown[]; + const w = typeof args?.[1] === "number" ? (args[1] as number) : null; + const h = typeof args?.[2] === "number" ? (args[2] as number) : null; + if (w === null || h === null || (w >= 50 && h >= 50)) imageCount++; + } else if ( + fn === pdfjs.OPS.constructPath || + fn === pdfjs.OPS.fill || + fn === pdfjs.OPS.stroke || + fn === pdfjs.OPS.eoFill + ) { + vectorOpCount++; + } + } + + const textContent = await page.getTextContent({ + disableNormalization: true, + }); + + // Font metadata is only available now that the operator list has run. + const boldByFont = new Map(); + for (const fontId of Object.keys(textContent.styles ?? {})) { + let bold = false; + try { + const font = page.commonObjs.get(fontId) as { + name?: string; + bold?: boolean; + black?: boolean; + file?: unknown; + }; + const name = font?.name ?? fontId; + fonts.add(name); + if (!font?.file) nonEmbeddedFonts.add(name); + bold = + font?.bold === true || + font?.black === true || + /bold|black|heavy|semibold|demi/i.test(name); + } catch { + // Font never resolved; fall back to no bold information. + } + boldByFont.set(fontId, bold); + } + + for (const raw of textContent.items) { + if (!("str" in raw)) continue; // marked-content boundary, not text + const item = raw as { + str: string; + transform: number[]; + width: number; + height: number; + fontName: string; + hasEOL: boolean; + }; + + if (LIGATURES.test(item.str)) hadLigatures = true; + + // pdf.js exports the same normaliser it applies internally. + let text = pdfjs.normalizeUnicode(item.str); + // Soft-hyphen artefact seen in some exports (per OpenResume). + text = text.replace(/-­‐/g, "-"); + + items.push({ + page: pageNum, + x: item.transform[4], + y: item.transform[5], + width: item.width, + height: item.height, + fontName: item.fontName, + bold: boldByFont.get(item.fontName) ?? false, + hasEOL: item.hasEOL, + text, + }); + } + + page.cleanup(); + } + + await loadingTask.destroy(); + + return { + items, + facts: { + pageCount, + pagesParsed, + truncated: pageCount > pagesParsed, + encrypted: false, + textItemCount: items.filter((i) => i.text.trim().length > 0).length, + fonts: [...fonts], + nonEmbeddedFonts: [...nonEmbeddedFonts], + imageCount, + vectorOpCount, + hadLigatures, + pageSizes, + }, + }; +} diff --git a/src/lib/ats/fields.ts b/src/lib/ats/fields.ts new file mode 100644 index 0000000..d2603d9 --- /dev/null +++ b/src/lib/ats/fields.ts @@ -0,0 +1,304 @@ +/** + * Field extraction by feature scoring. + * + * Each candidate string is scored against a list of weighted predicates and + * the highest scorer wins. The design comes from OpenResume, and the reason to + * prefer it over plain regex is the negative evidence: a name scores *down* for + * containing an @, a digit, a comma or a slash. That is what keeps an email + * address or a job title from being picked as the candidate's name. + * + * It is also explainable — every score can be shown to the user, which matters + * for a tool whose whole purpose is to make parsing legible. + */ + +import { parseDateRange } from "./dates.ts"; +import { firstDescriptionLine, splitIntoSubsections } from "./subsections.ts"; +import { startsWithBullet } from "./groupLines.ts"; +import { findSection, matchVocabulary } from "./sections.ts"; +import type { + EducationEntry, + Field, + Line, + ParsedResume, + ScoredCandidate, + Section, + WorkEntry, +} from "./types.ts"; + +export const EMAIL_RE = /\S+@\S+\.\S+/; +export const PHONE_RE = /\(?\d{3}\)?[\s.-]?\d{3}[\s.-]?\d{4}/; +export const LOCATION_RE = /[A-Z][a-zA-Z\s]+,\s*[A-Z]{2}\b/; +export const URL_RE = /(?:https?:\/\/|www\.)\S+\.\S+|\S+\.[a-z]{2,}\/\S+/i; +export const GPA_RE = /[0-4]\.\d{1,2}/; +export const DEGREE_RE = /\b(?:Associate|Bachelor|Master|Doctor|Ph\.?D|M\.?B\.?A|B\.?[AS]\b|M\.?[AS]\b)/i; + +const SCHOOL_WORDS = [ + "College", + "University", + "Institute", + "School", + "Academy", + "Polytechnic", +]; + +const JOB_TITLE_WORDS = [ + "Accountant", "Administrator", "Advisor", "Agent", "Analyst", "Apprentice", + "Architect", "Assistant", "Associate", "Attorney", "Auditor", "Chef", + "Clerk", "Consultant", "Coordinator", "Designer", "Developer", "Director", + "Editor", "Engineer", "Executive", "Founder", "Head", "Intern", "Lead", + "Manager", "Officer", "Operator", "Owner", "Partner", "President", + "Principal", "Producer", "Programmer", "Recruiter", "Representative", + "Researcher", "Scientist", "Specialist", "Strategist", "Supervisor", + "Teacher", "Technician", "Trainee", "VP", "Volunteer", "Webmaster", "Worker", +]; + +/** Precompiled once; this is tested against every header line of every entry. */ +const JOB_TITLE_RE = new RegExp(`\\b(?:${JOB_TITLE_WORDS.join("|")})\\b`, "i"); + +interface Feature { + label: string; + score: number; + test: (text: string, line: Line) => boolean; +} + +const hasAt = (t: string) => t.includes("@"); +const hasNumber = (t: string) => /\d/.test(t); +const hasComma = (t: string) => t.includes(","); +const hasSlash = (t: string) => t.includes("/"); +const hasParenthesis = (t: string) => /[()]/.test(t); +const isAllCaps = (t: string) => /[A-Za-z]/.test(t) && t === t.toUpperCase(); +const onlyLettersSpacesPeriods = (t: string) => /^[a-zA-Z\s.'-]+$/.test(t); + +const NAME_FEATURES: Feature[] = [ + { label: "letters, spaces and periods only", score: 3, test: onlyLettersSpacesPeriods }, + { label: "bold", score: 2, test: (_t, l) => l.items.every((i) => i.bold) }, + { label: "all caps", score: 2, test: isAllCaps }, + { label: "two to four words", score: 2, test: (t) => { + const n = t.trim().split(/\s+/).length; + return n >= 2 && n <= 4; + } }, + { label: "single word", score: -3, test: (t) => t.trim().split(/\s+/).length < 2 }, + { label: "reads as a sentence", score: -3, test: (t) => /[.!?]$/.test(t.trim()) }, + { label: "contains @", score: -4, test: hasAt }, + { label: "contains a digit", score: -4, test: hasNumber }, + { label: "contains parentheses", score: -4, test: hasParenthesis }, + { label: "contains a comma", score: -4, test: hasComma }, + { label: "contains a slash", score: -4, test: hasSlash }, + { label: "looks like a job title", score: -3, test: (t) => JOB_TITLE_RE.test(t) }, +]; + +const CLAMP_MIN = -4; +const CLAMP_MAX = 4; + +function scoreCandidates( + lines: Line[], + features: Feature[], +): ScoredCandidate[] { + return lines + .map((line) => { + const text = line.text.trim(); + const matched: string[] = []; + let score = 0; + for (const feature of features) { + if (feature.test(text, line)) { + matched.push(`${feature.label} (${feature.score > 0 ? "+" : ""}${feature.score})`); + score += feature.score; + } + } + return { + text, + score: Math.max(CLAMP_MIN, Math.min(CLAMP_MAX, score)), + matched, + }; + }) + .filter((c) => c.text.length > 0) + .sort((a, b) => b.score - a.score); +} + +function toField( + value: T | null, + score: number, + sourceLine: number | null, +): Field | null { + if (value === null || value === undefined) return null; + // Map the clamped score onto a 0..1 confidence. + const confidence = Math.max(0, Math.min(1, (score - CLAMP_MIN) / (CLAMP_MAX - CLAMP_MIN))); + return { value, confidence, sourceLine }; +} + +/** First line matching a pattern, returning the match and its line index. */ +function findByPattern( + lines: Line[], + re: RegExp, +): { value: string; index: number } | null { + for (let i = 0; i < lines.length; i++) { + const match = lines[i].text.match(re); + if (match) return { value: match[0].trim(), index: i }; + } + return null; +} + +export interface FieldExtraction { + resume: ParsedResume; + scores: Record; +} + +export function extractFields( + lines: Line[], + sections: Section[], +): FieldExtraction { + // The name is near the top; searching the whole document invites false hits. + // Section headings are excluded outright: "SKILLS" otherwise scores well as a + // name, being short, capitalised and free of digits. + const headLines = lines + .slice(0, 10) + .filter((l) => matchVocabulary(l.text) === null && !startsWithBullet(l.text)); + const nameScores = scoreCandidates(headLines, NAME_FEATURES); + const bestName = nameScores[0]; + const nameIndex = bestName + ? lines.findIndex((l) => l.text.trim() === bestName.text) + : -1; + + const email = findByPattern(lines, EMAIL_RE); + const phone = findByPattern(lines, PHONE_RE); + const location = findByPattern(lines, LOCATION_RE); + + const urls: Field[] = []; + lines.forEach((line, index) => { + const match = line.text.match(URL_RE); + if (match && !EMAIL_RE.test(match[0])) { + urls.push({ value: match[0], confidence: 0.8, sourceLine: index }); + } + }); + + const resume: ParsedResume = { + basics: { + name: + bestName && bestName.score > 0 + ? toField(bestName.text, bestName.score, nameIndex === -1 ? null : nameIndex) + : null, + email: email ? toField(email.value, CLAMP_MAX, email.index) : null, + phone: phone ? toField(phone.value, CLAMP_MAX, phone.index) : null, + location: location ? toField(location.value, 2, location.index) : null, + urls, + }, + work: extractWork(lines, sections), + education: extractEducation(lines, sections), + skills: extractSkills(sections), + }; + + return { + resume, + scores: { + name: nameScores.slice(0, 8), + }, + }; +} + +function lineIndexOf(all: Line[], line: Line): number | null { + const index = all.indexOf(line); + return index === -1 ? null : index; +} + +function extractWork(all: Line[], sections: Section[]): WorkEntry[] { + const section = findSection(sections, "experience"); + if (!section) return []; + + return splitIntoSubsections(section.lines) + .map((entry) => { + const descStart = firstDescriptionLine(entry); + const headerLines = entry.slice(0, Math.max(descStart, 1)); + const headerText = headerLines.map((l) => l.text).join(" "); + + const dates = parseDateRange(headerText); + + // Strip the date range so it cannot be mistaken for a title or employer. + const withoutDates = headerLines.map((l) => ({ + line: l, + text: dates ? l.text.replace(dates.raw, "").trim() : l.text.trim(), + })).filter((l) => l.text.length > 0); + + const titleCandidate = withoutDates.find((l) => JOB_TITLE_RE.test(l.text)); + // Whatever is left on the header lines that is not the title. + let orgCandidate = withoutDates.find((l) => l !== titleCandidate); + + // Templates very often put both on one line: "Senior Engineer, Acme Corp" + // or "Engineer at Acme". Split on the separator rather than reporting a + // missing employer, which is what a real parser would do here. + let titleText = titleCandidate?.text; + let orgText = orgCandidate?.text; + if (titleCandidate && !orgCandidate) { + const split = titleCandidate.text.match( + /^(.+?)\s*(?:,|\||–|—|\bat\b|\bfor\b)\s*(.+)$/i, + ); + if (split && split[1].trim() && split[2].trim()) { + titleText = split[1].trim(); + orgText = split[2].trim(); + orgCandidate = titleCandidate; + } + } + + const highlights = entry + .slice(descStart) + .map((l) => l.text.replace(/^[\s•⋅∙⦁●○⚬⬤-]+/, "").trim()) + .filter((t) => t.length > 0); + + return { + title: titleCandidate && titleText + ? toField(titleText, 3, lineIndexOf(all, titleCandidate.line)) + : null, + organisation: orgCandidate && orgText + ? toField(orgText, 2, lineIndexOf(all, orgCandidate.line)) + : null, + dates, + location: null, + highlights, + } satisfies WorkEntry; + }) + .filter((e) => e.title || e.organisation || e.dates); +} + +function extractEducation(all: Line[], sections: Section[]): EducationEntry[] { + const section = findSection(sections, "education"); + if (!section) return []; + + return splitIntoSubsections(section.lines) + .map((entry) => { + const text = entry.map((l) => l.text).join(" "); + const institutionLine = entry.find((l) => + SCHOOL_WORDS.some((w) => l.text.includes(w)), + ); + const degreeLine = entry.find((l) => DEGREE_RE.test(l.text)); + const gpa = text.match(GPA_RE)?.[0] ?? null; + + return { + institution: institutionLine + ? toField(institutionLine.text, 3, lineIndexOf(all, institutionLine)) + : null, + degree: degreeLine + ? toField(degreeLine.text.match(DEGREE_RE)?.[0] ?? degreeLine.text, 3, lineIndexOf(all, degreeLine)) + : null, + dates: parseDateRange(text), + gpa: gpa ? toField(gpa, 3, null) : null, + } satisfies EducationEntry; + }) + .filter((e) => e.institution || e.degree); +} + +function extractSkills(sections: Section[]): string[] { + const section = findSection(sections, "skills"); + if (!section) return []; + + const skills = new Set(); + for (const line of section.lines) { + const cleaned = line.text.replace(/^[\s•⋅∙⦁●○⚬⬤-]+/, ""); + // Skills are usually comma, pipe or bullet separated on one line. + for (const part of cleaned.split(/[,|;·]|\s{3,}/)) { + const skill = part.trim().replace(/[.:]+$/, ""); + if (skill.length > 1 && skill.length <= 40 && !startsWithBullet(skill)) { + skills.add(skill); + } + } + } + return [...skills]; +} diff --git a/src/lib/ats/groupLines.ts b/src/lib/ats/groupLines.ts new file mode 100644 index 0000000..fb66d3c --- /dev/null +++ b/src/lib/ats/groupLines.ts @@ -0,0 +1,172 @@ +/** + * Turning positioned text runs into lines of text. + * + * The merge rule follows OpenResume: runs closer together than one typical + * character width belong to the same word. This is what reassembles a phone + * number that PDF kerning (`TJ` arrays) split into three separate runs. + */ + +import type { Line, TextItem } from "./types.ts"; + +/** Bullet glyphs that begin a description line rather than a new entry. */ +export const BULLETS = [ + "⋅", + "∙", + "\u{1F784}", + "•", + "⦁", + "⚫", + "●", + "⬤", + "⚬", + "○", +]; + +const BULLET_SET = new Set(BULLETS); + +export function startsWithBullet(text: string): boolean { + const first = [...text.trimStart()][0]; + return first !== undefined && (BULLET_SET.has(first) || first === "-"); +} + +function modeOf(values: T[], weights: number[]): T | undefined { + const tally = new Map(); + values.forEach((v, i) => tally.set(v, (tally.get(v) ?? 0) + weights[i])); + let best: T | undefined; + let bestWeight = -1; + for (const [value, weight] of tally) { + if (weight > bestWeight) { + best = value; + bestWeight = weight; + } + } + return best; +} + +/** + * Width of one character in the document's dominant font, used as the + * threshold for whether adjacent runs are one word or two. + */ +export function typicalCharWidth(items: TextItem[]): number { + const real = items.filter((i) => i.text.trim().length > 0); + if (real.length === 0) return 5; + + const weights = real.map((i) => i.text.length); + const modalHeight = modeOf( + real.map((i) => i.height), + weights, + ); + const modalFont = modeOf( + real.map((i) => i.fontName), + weights, + ); + + let sample = real.filter( + (i) => i.height === modalHeight && i.fontName === modalFont, + ); + if (sample.length === 0) sample = real; + + const totalWidth = sample.reduce((sum, i) => sum + i.width, 0); + const totalChars = sample.reduce((sum, i) => sum + i.text.length, 0); + return totalChars > 0 ? totalWidth / totalChars : 5; +} + +/** Join runs on one line, inserting a space only where the gap warrants it. */ +export function assembleLineText(items: TextItem[], charWidth: number): string { + const sorted = items + .filter((i) => i.text.trim().length > 0) + .sort((a, b) => a.x - b.x); + if (sorted.length === 0) return ""; + + let out = sorted[0].text; + for (let i = 1; i < sorted.length; i++) { + const prev = sorted[i - 1]; + const cur = sorted[i]; + const gap = cur.x - (prev.x + prev.width); + if (gap > charWidth * 0.5) out += " "; + out += cur.text; + } + return out.replace(/\s+/g, " ").trim(); +} + +function makeLine(items: TextItem[], charWidth: number): Line | null { + const real = items.filter((i) => i.text.trim().length > 0); + if (real.length === 0) return null; + return { + page: real[0].page, + items: real, + text: assembleLineText(real, charWidth), + y: real[0].y, + x: Math.min(...real.map((i) => i.x)), + }; +} + +/** + * Group in content-stream order, breaking on pdf.js's `hasEOL`. + * + * Empty items are kept until this point precisely because the EOL flag is + * usually carried by a zero-width item rather than by the preceding text. + */ +export function groupByEOL(items: TextItem[], charWidth: number): Line[] { + const lines: Line[] = []; + let current: TextItem[] = []; + + for (const item of items) { + if (item.text.trim().length > 0) current.push(item); + if (item.hasEOL) { + const line = makeLine(current, charWidth); + if (line) lines.push(line); + current = []; + } + } + const last = makeLine(current, charWidth); + if (last) lines.push(last); + return lines; +} + +/** + * Group by visual position: sort top-to-bottom then left-to-right, treating + * items within `tolerance` points of each other as sharing a baseline. + * + * This is what most mid-tier parsers do, and it is exactly what interleaves + * the two halves of a column layout into nonsense. + */ +export function groupByPosition( + items: TextItem[], + charWidth: number, + tolerance = 3, +): Line[] { + const real = items.filter((i) => i.text.trim().length > 0); + const sorted = [...real].sort((a, b) => { + if (a.page !== b.page) return a.page - b.page; + if (Math.abs(a.y - b.y) > tolerance) return b.y - a.y; + return a.x - b.x; + }); + + const lines: Line[] = []; + let current: TextItem[] = []; + + for (const item of sorted) { + if (current.length === 0) { + current.push(item); + continue; + } + const ref = current[0]; + const sameLine = + item.page === ref.page && Math.abs(item.y - ref.y) <= tolerance; + if (sameLine) { + current.push(item); + } else { + const line = makeLine(current, charWidth); + if (line) lines.push(line); + current = [item]; + } + } + const last = makeLine(current, charWidth); + if (last) lines.push(last); + return lines; +} + +export function linesToText(lines: Line[]): string { + return lines.map((l) => l.text).join("\n"); +} diff --git a/src/lib/ats/index.ts b/src/lib/ats/index.ts new file mode 100644 index 0000000..1aa06ac --- /dev/null +++ b/src/lib/ats/index.ts @@ -0,0 +1,281 @@ +/** + * Entry point for the ATS resume inspector. + * + * Everything runs from an in-memory buffer and nothing is written to disk or + * cached. The container this runs in is read-only with only /tmp as tmpfs, so + * "your resume is never stored" is a property of the deployment rather than a + * promise we are asking anyone to take on trust. + */ + +import { extractDocx, CorruptDocxError } from "./extractDocx.ts"; +import { extractPdf, EncryptedPdfError, CorruptPdfError, MAX_PAGES } from "./extractPdf.ts"; +import { buildStrategies } from "./readingOrder.ts"; +import { detectSections } from "./sections.ts"; +import { extractFields } from "./fields.ts"; +import { runDiagnostics, sortFindings } from "./diagnostics.ts"; +import type { AtsReport, DocFacts, SourceKind, TextItem } from "./types.ts"; + +/** Matches the file size limit Greenhouse documents for resume uploads. */ +export const MAX_BYTES = 2.5 * 1024 * 1024; +/** Mirrors Textkernel code 411, "parsing had to be stopped". */ +export const PARSE_TIMEOUT_MS = 10_000; + +export { MAX_PAGES }; + +export class UnsupportedFormatError extends Error {} +export class TooLargeError extends Error {} +export class ParseTimeoutError extends Error {} + +/** Plain text has no layout, so synthesise one line per line of input. */ +function itemsFromText(text: string): TextItem[] { + return text + .split(/\r?\n/) + .map((line, index) => ({ + page: 1, + x: 0, + y: 1000 - index * 14, + width: line.length * 5, + height: 11, + fontName: "text", + bold: false, + hasEOL: true, + text: line, + })) + .filter((item) => item.text.trim().length > 0); +} + +function detectKind(fileName: string, data: Uint8Array): SourceKind { + const lower = fileName.toLowerCase(); + // Sniff magic bytes rather than trusting the extension. + const isPdf = + data[0] === 0x25 && data[1] === 0x50 && data[2] === 0x44 && data[3] === 0x46; + const isZip = data[0] === 0x50 && data[1] === 0x4b; + + if (isPdf) return "pdf"; + if (isZip && lower.endsWith(".docx")) return "docx"; + if (lower.endsWith(".txt") || lower.endsWith(".md")) return "text"; + if (lower.endsWith(".doc")) { + throw new UnsupportedFormatError( + "Legacy .doc files are not supported. Open it in Word or LibreOffice and save as .docx or PDF, then try again.", + ); + } + if (isZip) return "docx"; + throw new UnsupportedFormatError( + "Unrecognised file type. Upload a PDF, a DOCX, or paste the text directly.", + ); +} + +async function withTimeout(work: Promise): Promise { + let timer: ReturnType; + const timeout = new Promise((_, reject) => { + timer = setTimeout( + () => reject(new ParseTimeoutError("Parsing took too long")), + PARSE_TIMEOUT_MS, + ); + }); + try { + return await Promise.race([work, timeout]); + } finally { + clearTimeout(timer!); + } +} + +export interface AnalyseInput { + data: Uint8Array; + fileName: string; + /** Set when the user pasted text rather than uploading a file. */ + pastedText?: string; +} + +export async function analyseResume(input: AnalyseInput): Promise { + const started = Date.now(); + + if (input.data.byteLength > MAX_BYTES) { + throw new TooLargeError( + `File is larger than ${(MAX_BYTES / 1024 / 1024).toFixed(1)} MB.`, + ); + } + + return withTimeout(runAnalysis(input, started)); +} + +async function runAnalysis( + input: AnalyseInput, + started: number, +): Promise { + const { data, fileName } = input; + const kind: SourceKind = input.pastedText + ? "text" + : detectKind(fileName, data); + + let items: TextItem[]; + let facts: DocFacts; + + if (kind === "pdf") { + let extraction; + try { + extraction = await extractPdf(data); + } catch (err) { + if (err instanceof EncryptedPdfError) { + return emptyReport(kind, fileName, data.byteLength, started, true); + } + if (err instanceof CorruptPdfError) { + throw new UnsupportedFormatError( + "This PDF could not be opened. It may be damaged or incomplete.", + ); + } + throw err; + } + items = extraction.items; + facts = { + kind, + fileName, + byteSize: data.byteLength, + ...extraction.facts, + }; + } else if (kind === "docx") { + let extraction; + try { + extraction = await extractDocx(data); + } catch (err) { + if (err instanceof CorruptDocxError) { + throw new UnsupportedFormatError( + "This DOCX could not be opened. It may be damaged, or saved in an older format.", + ); + } + throw err; + } + items = extraction.items; + facts = { + kind, + fileName, + byteSize: data.byteLength, + pageCount: 1, + pagesParsed: 1, + truncated: false, + encrypted: false, + textItemCount: items.length, + fonts: [], + nonEmbeddedFonts: [], + imageCount: 0, + vectorOpCount: 0, + hadLigatures: false, + pageSizes: [], + docx: extraction.docx, + }; + } else { + const text = input.pastedText ?? new TextDecoder().decode(data); + items = itemsFromText(text); + facts = { + kind, + fileName: input.pastedText ? "pasted text" : fileName, + byteSize: data.byteLength, + pageCount: 1, + pagesParsed: 1, + truncated: false, + encrypted: false, + textItemCount: items.length, + fonts: [], + nonEmbeddedFonts: [], + imageCount: 0, + vectorOpCount: 0, + hadLigatures: false, + pageSizes: [], + }; + } + + const strategies = buildStrategies(items); + const visual = strategies.find((s) => s.id === "visual")!; + const column = strategies.find((s) => s.id === "column")!; + + // Column-aware reading only matters if it actually changes the text. + const columnsChangeReading = + kind === "pdf" && visual.text !== column.text; + + // Analyse the best available reading, which is what a good parser would use. + const primaryStrategy = columnsChangeReading ? column : visual; + const { sections, unrecognised } = detectSections(primaryStrategy.lines); + const { resume, scores } = extractFields(primaryStrategy.lines, sections); + + const findings = sortFindings( + runDiagnostics({ + facts, + lines: primaryStrategy.lines, + text: primaryStrategy.text, + sections, + unrecognised, + resume, + columnsChangeReading, + }), + ); + + return { + facts, + strategies, + primary: primaryStrategy.id, + sections, + resume, + findings, + scores, + columnsChangeReading, + parseMs: Date.now() - started, + }; +} + +/** Used when the document cannot be opened at all but we still owe a report. */ +function emptyReport( + kind: SourceKind, + fileName: string, + byteSize: number, + started: number, + encrypted: boolean, +): AtsReport { + const facts: DocFacts = { + kind, + fileName, + byteSize, + pageCount: 0, + pagesParsed: 0, + truncated: false, + encrypted, + textItemCount: 0, + fonts: [], + nonEmbeddedFonts: [], + imageCount: 0, + vectorOpCount: 0, + hadLigatures: false, + pageSizes: [], + }; + + return { + facts, + strategies: [], + primary: "visual", + sections: [], + resume: { + basics: { name: null, email: null, phone: null, location: null, urls: [] }, + work: [], + education: [], + skills: [], + }, + findings: sortFindings( + runDiagnostics({ + facts, + lines: [], + text: "", + sections: [], + unrecognised: [], + resume: { + basics: { name: null, email: null, phone: null, location: null, urls: [] }, + work: [], + education: [], + skills: [], + }, + columnsChangeReading: false, + }), + ), + scores: {}, + columnsChangeReading: false, + parseMs: Date.now() - started, + }; +} diff --git a/src/lib/ats/limits.ts b/src/lib/ats/limits.ts new file mode 100644 index 0000000..14c348c --- /dev/null +++ b/src/lib/ats/limits.ts @@ -0,0 +1,20 @@ +/** + * Rate limiters for the resume inspector. + * + * These live in their own module so they are created once per process. Astro + * re-runs page frontmatter on every request, so limiters declared there would + * reset on each call and enforce nothing. + * + * Parsing is CPU-bound and the endpoint is public and unauthenticated, so the + * limits are deliberately tight. There is no captcha: this is a utility, and + * requiring proof-of-work before every parse would break it for anyone + * browsing without JavaScript. + */ + +import { RateLimiter } from "@lib/RateLimiter.ts"; + +/** Per-client: enough to iterate on a resume, not enough to mine the endpoint. */ +export const perClientLimiter = new RateLimiter(12, 10 * 60 * 1000); + +/** Process-wide ceiling, so one busy hour cannot saturate the container. */ +export const globalLimiter = new RateLimiter(240, 60 * 60 * 1000); diff --git a/src/lib/ats/readingOrder.ts b/src/lib/ats/readingOrder.ts new file mode 100644 index 0000000..811996e --- /dev/null +++ b/src/lib/ats/readingOrder.ts @@ -0,0 +1,222 @@ +/** + * Three reconstructions of reading order. + * + * A PDF content stream is a list of "draw this glyph at these coordinates" + * instructions. It carries no reading order at all; only tagged PDFs have a + * logical structure tree, and resumes essentially never do. So every parser + * guesses, and different parsers guess differently. Rather than pick one and + * present it as truth, we compute all three and let the user see the divergence. + */ + +import { + groupByEOL, + groupByPosition, + linesToText, + typicalCharWidth, +} from "./groupLines.ts"; +import type { Line, Strategy, TextItem } from "./types.ts"; + +/** Bucket width in points for the vertical whitespace mask. */ +const BUCKET = 2; +/** A column separator must be at least this wide. */ +const MIN_GAP_PT = 25; +/** Each side of a split must span at least this share of the content width. */ +const MIN_SIDE_WIDTH_RATIO = 0.15; +/** + * Each side must also carry this share of the page's characters. This is what + * separates a real sidebar (lots of text) from a right-aligned date gutter + * (a handful of short dates), which must NOT be pulled out as its own column — + * doing so would detach every date from the job it belongs to. + */ +const MIN_SIDE_CHAR_RATIO = 0.15; +/** Each side of a split must hold at least this many distinct text lines. */ +const MIN_SIDE_LINES = 3; +const MAX_DEPTH = 3; + +interface Bounds { + minX: number; + maxX: number; +} + +function bounds(items: TextItem[]): Bounds { + let minX = Infinity; + let maxX = -Infinity; + for (const i of items) { + if (i.x < minX) minX = i.x; + if (i.x + i.width > maxX) maxX = i.x + i.width; + } + return { minX, maxX }; +} + +function distinctLineCount(items: TextItem[], tolerance = 3): number { + const ys: number[] = []; + for (const i of items) { + if (!ys.some((y) => Math.abs(y - i.y) <= tolerance)) ys.push(i.y); + } + return ys.length; +} + +/** + * Find an interior vertical whitespace channel that runs the full height of + * the content. Returns the x coordinate to cut at, or null when the block is + * a single column. + * + * This is the cheap, deterministic form of what Textkernel does with a trained + * gap classifier. Their rule-based version scored 60% on separator gaps before + * they replaced it with gradient boosting, so treat a null result as "probably + * one column" rather than as proof. + */ +function findColumnSplit(items: TextItem[]): number | null { + const real = items.filter((i) => i.text.trim().length > 0); + if (real.length < MIN_SIDE_LINES * 2) return null; + + const { minX, maxX } = bounds(real); + const contentWidth = maxX - minX; + if (!Number.isFinite(contentWidth) || contentWidth <= 0) return null; + + const bucketCount = Math.ceil(contentWidth / BUCKET) + 1; + const occupancy = new Uint32Array(bucketCount); + + for (const item of real) { + const from = Math.max(0, Math.floor((item.x - minX) / BUCKET)); + const to = Math.min( + bucketCount - 1, + Math.ceil((item.x + item.width - minX) / BUCKET), + ); + for (let b = from; b <= to; b++) occupancy[b]++; + } + + // Collect maximal runs of completely empty buckets, ignoring the margins. + let best: { start: number; end: number; width: number } | null = null; + let runStart: number | null = null; + + for (let b = 0; b <= bucketCount; b++) { + const empty = b < bucketCount && occupancy[b] === 0; + if (empty && runStart === null) { + runStart = b; + } else if (!empty && runStart !== null) { + const isInterior = runStart > 0 && b < bucketCount; + const width = (b - runStart) * BUCKET; + if (isInterior && width >= MIN_GAP_PT && (!best || width > best.width)) { + best = { start: runStart, end: b, width }; + } + runStart = null; + } + } + + if (!best) return null; + + const cutX = minX + ((best.start + best.end) / 2) * BUCKET; + const left = real.filter((i) => i.x + i.width <= cutX); + const right = real.filter((i) => i.x >= cutX); + + if ( + distinctLineCount(left) < MIN_SIDE_LINES || + distinctLineCount(right) < MIN_SIDE_LINES + ) { + return null; + } + + // Measure each side by the span it occupies, not by how wide its text + // happens to run: a skills sidebar holds short words in a wide column. + const leftSpan = cutX - minX; + const rightSpan = maxX - cutX; + if ( + leftSpan < contentWidth * MIN_SIDE_WIDTH_RATIO || + rightSpan < contentWidth * MIN_SIDE_WIDTH_RATIO + ) { + return null; + } + + const charsOf = (list: TextItem[]) => + list.reduce((sum, i) => sum + i.text.trim().length, 0); + const totalChars = charsOf(real); + if ( + totalChars === 0 || + charsOf(left) < totalChars * MIN_SIDE_CHAR_RATIO || + charsOf(right) < totalChars * MIN_SIDE_CHAR_RATIO + ) { + return null; + } + + return cutX; +} + +/** Recursively split a page into column blocks, ordered left to right. */ +function splitIntoBlocks(items: TextItem[], depth = 0): TextItem[][] { + if (depth >= MAX_DEPTH) return [items]; + const cutX = findColumnSplit(items); + if (cutX === null) return [items]; + + const left = items.filter((i) => i.x + i.width <= cutX); + const right = items.filter((i) => i.x >= cutX); + // Items straddling the cut belong with whichever side holds their left edge. + const straddling = items.filter( + (i) => i.x < cutX && i.x + i.width > cutX, + ); + for (const item of straddling) { + (item.x + item.width / 2 < cutX ? left : right).push(item); + } + + if (left.length === 0 || right.length === 0) return [items]; + return [...splitIntoBlocks(left, depth + 1), ...splitIntoBlocks(right, depth + 1)]; +} + +/** True when this page has a genuine column separator. */ +export function detectsColumns(items: TextItem[]): boolean { + const pages = new Set(items.map((i) => i.page)); + for (const page of pages) { + const pageItems = items.filter((i) => i.page === page); + if (findColumnSplit(pageItems) !== null) return true; + } + return false; +} + +function columnOrder(items: TextItem[], charWidth: number): Line[] { + const lines: Line[] = []; + const pages = [...new Set(items.map((i) => i.page))].sort((a, b) => a - b); + + for (const page of pages) { + const pageItems = items.filter((i) => i.page === page); + const blocks = splitIntoBlocks(pageItems); + for (const block of blocks) { + lines.push(...groupByPosition(block, charWidth)); + } + } + return lines; +} + +export function buildStrategies(items: TextItem[]): Strategy[] { + const charWidth = typicalCharWidth(items); + + const stream = groupByEOL(items, charWidth); + const visual = groupByPosition(items, charWidth); + const column = columnOrder(items, charWidth); + + return [ + { + id: "stream", + label: "Content-stream order", + blurb: + "Text in the order the PDF file stores it, split on the file's own line breaks. The simplest parsers read this way, so design-tool exports show up scrambled here.", + lines: stream, + text: linesToText(stream), + }, + { + id: "visual", + label: "Naive top-to-bottom", + blurb: + "Sorted by position, top to bottom then left to right. This is what most parsers do, and it is what merges the two halves of a column layout into one another.", + lines: visual, + text: linesToText(visual), + }, + { + id: "column", + label: "Column-aware", + blurb: + "Detects vertical whitespace channels and reads each column fully before moving right. This is what a good modern parser does.", + lines: column, + text: linesToText(column), + }, + ]; +} diff --git a/src/lib/ats/sections.ts b/src/lib/ats/sections.ts new file mode 100644 index 0000000..0c20cb3 --- /dev/null +++ b/src/lib/ats/sections.ts @@ -0,0 +1,212 @@ +/** + * Section heading recognition. + * + * No ATS vendor publishes its heading vocabulary, so this combines the best + * published open-source lists with the structural heuristics that matter more + * in practice: a heading is usually alone on its line, short, and visually + * distinct. Vocabulary is the fallback, not the primary signal. + * + * The important output is not "this heading was not recognised" but what + * happened as a result — which lines got absorbed into the wrong section. + */ + +import type { Line, Section, SectionId } from "./types.ts"; + +/** + * Base vocabulary from the ats-screener project, the best published list found, + * extended with aliases from code4goal's dictionary. + */ +const VOCABULARY: [SectionId, RegExp][] = [ + [ + "contact", + /^(contact\s*(info(rmation)?)?|personal\s*(info(rmation)?|details)|profiles?|social\s*(connect|profiles?)|links?)$/i, + ], + [ + "summary", + /^(summary|profile|about(\s*me)?|objective|professional\s*summary|career\s*summary|executive\s*summary|personal\s*statement)$/i, + ], + [ + "experience", + /^(experience|work\s*experience|professional\s*experience|employment(\s*history)?|work\s*history|relevant\s*experience|career\s*history|positions)$/i, + ], + [ + "education", + /^(education|academic(\s*background)?|educational\s*background|qualifications|academic\s*qualifications)$/i, + ], + [ + "skills", + /^(skills|technical\s*skills|core\s*competencies|competencies|areas?\s*of\s*expertise|proficiencies|technolog(y|ies)|tools?\s*(&|and)\s*technologies|skills\s*(&|and)\s*expertise)$/i, + ], + [ + "projects", + /^(projects|personal\s*projects|academic\s*projects|notable\s*projects|selected\s*projects|key\s*projects|side\s*projects)$/i, + ], + [ + "certifications", + /^(certifications?|licen[cs]es?(\s*(&|and)\s*certifications?)?|professional\s*certifications?|accreditations?)$/i, + ], + [ + "awards", + /^(awards?|honors?(\s*(&|and)\s*awards?)?|achievements?|recognition|scholarships?)$/i, + ], + ["publications", /^(publications?|research|papers?|presentations?)$/i], + [ + "volunteer", + /^(volunteer(ing)?(\s*experience)?|community\s*(service|involvement)|extracurricular(\s*activities)?)$/i, + ], + ["languages", /^(languages?|language\s*proficiency)$/i], + ["interests", /^(interests?|hobbies(\s*(&|and)\s*interests?)?|additional)$/i], + ["references", /^(references?)$/i], +]; + +export const SECTION_LABELS: Record = { + contact: "Contact", + summary: "Summary", + experience: "Work history", + education: "Education", + skills: "Skills", + projects: "Projects", + certifications: "Certifications", + awards: "Awards", + publications: "Publications", + volunteer: "Volunteer", + languages: "Languages", + interests: "Interests", + references: "References", + unknown: "Unrecognised", +}; + +/** Strip trailing punctuation parsers commonly ignore before matching. */ +function normaliseHeading(text: string): string { + return text + .replace(/[:\-_|]+/g, " ") + .replace(/\s+/g, " ") + .trim(); +} + +export function matchVocabulary(text: string): SectionId | null { + const cleaned = normaliseHeading(text); + for (const [id, re] of VOCABULARY) { + if (re.test(cleaned)) return id; + } + return null; +} + +/** + * Whether a line is shaped like a heading, independent of its wording. + * Deliberately excludes things that look like a person's name, which would + * otherwise swallow the top of the document. + */ +export function looksLikeHeading(line: Line, prevBlank: boolean): boolean { + const text = line.text.trim(); + if (text.length === 0 || text.length > 80) return false; + + const words = text.split(/\s+/); + if (words.length > 5) return false; + // Three or more consecutive digits means a date or metric, not a heading. + if (/\d{3,}/.test(text)) return false; + + const isBold = line.items.every((i) => i.bold); + const isAllCaps = /[A-Za-z]/.test(text) && text === text.toUpperCase(); + const endsWithColon = /:$/.test(text); + const alphaOnly = /^[A-Za-z\s&]+$/.test(text); + + // "Jane Q. Candidate" is alpha-only, short, and often bold. Exclude it. + const looksLikeName = words.length >= 2 && words.length <= 3 && + /^[A-Z][a-z]+\s+[A-Z]/.test(text); + + if (looksLikeName && !isAllCaps) return false; + if (isBold && isAllCaps) return true; + if (endsWithColon && words.length <= 4) return true; + if (isAllCaps && alphaOnly) return true; + if (prevBlank && alphaOnly && (isBold || isAllCaps)) return true; + return false; +} + +export interface UnrecognisedHeading { + lineIndex: number; + text: string; + absorbedInto: SectionId; + absorbedLines: number; +} + +export interface SectionResult { + sections: Section[]; + unrecognised: UnrecognisedHeading[]; +} + +/** + * Split lines into sections. + * + * A vocabulary match on a short line opens a new section. A line that is + * shaped like a heading but matches nothing is recorded as unrecognised, and + * its content stays in the previous section — which is precisely the failure + * the user needs to be told about. + */ +export function detectSections(lines: Line[]): SectionResult { + const sections: Section[] = []; + const unrecognised: UnrecognisedHeading[] = []; + + let current: Section = { + id: "contact", + heading: null, + headingLine: null, + lines: [], + inferred: true, + }; + + const pendingUnrecognised: { lineIndex: number; text: string }[] = []; + + lines.forEach((line, index) => { + const words = line.text.trim().split(/\s+/); + const isShort = line.text.trim().length <= 80 && words.length <= 6; + const vocab = isShort ? matchVocabulary(line.text) : null; + + if (vocab) { + // Close the section in progress and attribute anything that looked like + // a heading but was not recognised to the section it ended up inside. + for (const pending of pendingUnrecognised) { + unrecognised.push({ + ...pending, + absorbedInto: current.id, + absorbedLines: current.lines.length, + }); + } + pendingUnrecognised.length = 0; + + if (current.lines.length > 0 || current.heading) sections.push(current); + current = { + id: vocab, + heading: line.text.trim(), + headingLine: index, + lines: [], + inferred: false, + }; + return; + } + + const prevBlank = index === 0 || lines[index - 1].text.trim() === ""; + if (looksLikeHeading(line, prevBlank)) { + pendingUnrecognised.push({ lineIndex: index, text: line.text.trim() }); + } + current.lines.push(line); + }); + + for (const pending of pendingUnrecognised) { + unrecognised.push({ + ...pending, + absorbedInto: current.id, + absorbedLines: current.lines.length, + }); + } + if (current.lines.length > 0 || current.heading) sections.push(current); + + return { sections, unrecognised }; +} + +export function findSection( + sections: Section[], + id: SectionId, +): Section | undefined { + return sections.find((s) => s.id === id); +} diff --git a/src/lib/ats/subsections.ts b/src/lib/ats/subsections.ts new file mode 100644 index 0000000..28c958a --- /dev/null +++ b/src/lib/ats/subsections.ts @@ -0,0 +1,103 @@ +/** + * Splitting a section into individual entries (one job, one degree). + * + * Primary signal is vertical spacing: resumes put more space between entries + * than between the lines of one entry. When that fails — usually because the + * spacing is perfectly uniform — fall back to a bold line following a + * non-bold one, which is how most templates start a new entry. + */ + +import { startsWithBullet } from "./groupLines.ts"; +import type { Line } from "./types.ts"; + +/** Gap larger than this multiple of the typical line gap starts a new entry. */ +const GAP_MULTIPLIER = 1.4; + +function modalGap(lines: Line[]): number { + const gaps: number[] = []; + for (let i = 1; i < lines.length; i++) { + if (lines[i].page !== lines[i - 1].page) continue; + const gap = lines[i - 1].y - lines[i].y; + if (gap > 0) gaps.push(Math.round(gap)); + } + if (gaps.length === 0) return 0; + + const tally = new Map(); + for (const gap of gaps) tally.set(gap, (tally.get(gap) ?? 0) + 1); + + let best = gaps[0]; + let bestCount = 0; + for (const [gap, count] of tally) { + if (count > bestCount) { + best = gap; + bestCount = count; + } + } + return best; +} + +export function splitIntoSubsections(lines: Line[]): Line[][] { + if (lines.length === 0) return []; + + const typical = modalGap(lines); + // Default to treating the section as one entry, so a section short enough to + // have no measurable line gaps still produces a result. + let groups: Line[][] = [lines]; + let current: Line[] = [lines[0]]; + + if (typical > 0) { + groups = []; + for (let i = 1; i < lines.length; i++) { + const samePage = lines[i].page === lines[i - 1].page; + const gap = lines[i - 1].y - lines[i].y; + if (!samePage || gap > typical * GAP_MULTIPLIER) { + groups.push(current); + current = [lines[i]]; + } else { + current.push(lines[i]); + } + } + groups.push(current); + } + + // Spacing told us nothing; try the bold-start transition instead. Bullets + // are sometimes marked bold by the template, so ignore those lines. + if (groups.length <= 1) { + const fallback: Line[][] = []; + let group: Line[] = []; + for (let i = 0; i < lines.length; i++) { + const line = lines[i]; + const isBoldStart = + line.items[0]?.bold === true && !startsWithBullet(line.text); + const prevBoldStart = + i > 0 && + lines[i - 1].items[0]?.bold === true && + !startsWithBullet(lines[i - 1].text); + if (i > 0 && isBoldStart && !prevBoldStart) { + fallback.push(group); + group = []; + } + group.push(line); + } + if (group.length > 0) fallback.push(group); + if (fallback.length > 1) return fallback; + } + + return groups.filter((g) => g.length > 0); +} + +/** + * Index of the first description line in an entry: the first bullet, or + * failing that the first prose-looking line. The word-count fallback catches + * LinkedIn-generated resumes, which use no bullet glyphs at all. + */ +export function firstDescriptionLine(entry: Line[]): number { + const bulletIndex = entry.findIndex((l) => startsWithBullet(l.text)); + if (bulletIndex !== -1) return bulletIndex; + + const proseIndex = entry.findIndex((l) => { + const words = l.text.trim().split(/\s+/).filter((w) => !/^\d+$/.test(w)); + return words.length >= 8; + }); + return proseIndex === -1 ? entry.length : proseIndex; +} diff --git a/src/lib/ats/types.ts b/src/lib/ats/types.ts new file mode 100644 index 0000000..be6daa2 --- /dev/null +++ b/src/lib/ats/types.ts @@ -0,0 +1,214 @@ +/** + * Shared types for the ATS resume inspector. + * + * The pipeline is: bytes -> TextItem[] -> Line[] (per reading-order strategy) + * -> Section[] -> ParsedResume, with Finding[] accumulated throughout. + */ + +/** A single positioned run of text, normalised across PDF and DOCX sources. */ +export interface TextItem { + /** 1-based page number. DOCX and plain text are always page 1. */ + page: number; + /** Left edge in PDF user space (points, origin bottom-left). */ + x: number; + /** Baseline y in PDF user space. Higher is further up the page. */ + y: number; + width: number; + height: number; + /** pdf.js internal font id, e.g. "g_d0_f1". Not a real font name. */ + fontName: string; + bold: boolean; + /** + * pdf.js sets this when the run is followed by a line break. Note that it + * frequently lands on a zero-width empty item rather than on the text run + * itself, so empty items must not be discarded before line grouping. + */ + hasEOL: boolean; + text: string; +} + +/** A run of text items judged to sit on the same visual line. */ +export interface Line { + page: number; + items: TextItem[]; + text: string; + /** Baseline y of the first item. */ + y: number; + /** Left edge of the leftmost item. */ + x: number; +} + +/** + * The three reading orders we reconstruct. Showing the difference between + * these is the point of the tool: a PDF content stream carries no reading + * order, so each strategy is a defensible guess that real parsers actually make. + */ +export type StrategyId = "stream" | "visual" | "column"; + +export interface Strategy { + id: StrategyId; + label: string; + /** One-line description of which real-world parsers behave this way. */ + blurb: string; + lines: Line[]; + text: string; +} + +export type SourceKind = "pdf" | "docx" | "text"; + +/** Structural facts about the uploaded document, used to drive diagnostics. */ +export interface DocFacts { + kind: SourceKind; + fileName: string; + byteSize: number; + pageCount: number; + /** Pages actually parsed; lower than pageCount when the cap kicked in. */ + pagesParsed: number; + truncated: boolean; + encrypted: boolean; + textItemCount: number; + /** Real font names seen, e.g. "Helvetica-Bold". */ + fonts: string[]; + /** Fonts with no embedded file; these risk bad character mapping. */ + nonEmbeddedFonts: string[]; + /** Raster images at least 50x50px, which excludes glyph-sized artefacts. */ + imageCount: number; + /** Vector path ops; high counts with no text suggest outlined text. */ + vectorOpCount: number; + /** True when raw ligature codepoints (U+FB00..) were present pre-normalisation. */ + hadLigatures: boolean; + /** Page dimensions in points, used to lay out the position map. */ + pageSizes: { width: number; height: number }[]; + /** DOCX-only structural findings, from inspecting the raw zip parts. */ + docx?: DocxFacts; +} + +/** + * What a full DOCX reader can see that a typical ATS text extractor cannot. + * Diffing these against the extracted text is how header/footer and text-box + * losses are detected. + */ +export interface DocxFacts { + headerText: string[]; + footerText: string[]; + textBoxText: string[]; + smartArtText: string[]; + tableCount: number; + /** True when real Word column layout (w:cols) is used, which parses fine. */ + usesColumnSections: boolean; +} + +export type Severity = "fatal" | "major" | "minor" | "info"; + +/** + * A single diagnostic. `citation` names the real parser that documents this + * behaviour; the tool is only credible because these are not invented. + */ +export interface Finding { + id: string; + severity: Severity; + title: string; + /** Why this happens, mechanically. */ + cause: string; + /** What the user should change. */ + fix: string; + citation: string; + /** Evidence from this specific document. */ + detail?: string; +} + +export type SectionId = + | "contact" + | "summary" + | "experience" + | "education" + | "skills" + | "projects" + | "certifications" + | "awards" + | "publications" + | "volunteer" + | "languages" + | "interests" + | "references" + | "unknown"; + +export interface Section { + id: SectionId; + /** The heading text as written in the document, if one was found. */ + heading: string | null; + /** Index into the strategy's line array where the heading sits. */ + headingLine: number | null; + lines: Line[]; + /** True when the section had to be inferred without a recognised heading. */ + inferred: boolean; +} + +export interface DateRange { + raw: string; + start: string | null; + end: string | null; + isCurrent: boolean; + parsed: boolean; + /** Why parsing failed, when it did. */ + problem?: string; +} + +/** A field plus the evidence for it, so the UI can show its provenance. */ +export interface Field { + value: T; + confidence: number; + /** Index of the source line within the active strategy. */ + sourceLine: number | null; +} + +export interface WorkEntry { + title: Field | null; + organisation: Field | null; + dates: DateRange | null; + location: Field | null; + highlights: string[]; +} + +export interface EducationEntry { + institution: Field | null; + degree: Field | null; + dates: DateRange | null; + gpa: Field | null; +} + +/** Loosely JSON Resume shaped, with confidence and provenance added. */ +export interface ParsedResume { + basics: { + name: Field | null; + email: Field | null; + phone: Field | null; + location: Field | null; + urls: Field[]; + }; + work: WorkEntry[]; + education: EducationEntry[]; + skills: string[]; +} + +/** A candidate string and the features that scored it, for the score table. */ +export interface ScoredCandidate { + text: string; + score: number; + matched: string[]; +} + +export interface AtsReport { + facts: DocFacts; + strategies: Strategy[]; + /** Strategy used for section and field extraction. */ + primary: StrategyId; + sections: Section[]; + resume: ParsedResume; + findings: Finding[]; + /** Per-attribute candidate scores, shown to explain field extraction. */ + scores: Record; + /** True when `visual` and `column` disagree, i.e. columns changed the text. */ + columnsChangeReading: boolean; + parseMs: number; +} diff --git a/src/pages/index.astro b/src/pages/index.astro index 0916b4a..5f8f8a2 100644 --- a/src/pages/index.astro +++ b/src/pages/index.astro @@ -262,7 +262,7 @@ import Layout from "@layouts/BaseLayout.astro"; + ) + } + +