From 942429ed3b33a1db17b0e1ad786546f0a2357e58 Mon Sep 17 00:00:00 2001 From: technomancer Date: Mon, 14 Sep 2026 22:00:00 -0700 Subject: [PATCH 1/3] added triangle bench to gltest --- test/gltest/gltest | Bin 24316 -> 43300 bytes test/gltest/main.c | 292 ++++++++++++++++++++++++++++++++++++++++----- 2 files changed, 262 insertions(+), 30 deletions(-) diff --git a/test/gltest/gltest b/test/gltest/gltest index 1446dc546a502f9760d3c68ffd8e8454d856f2b7..74cf052ac7f76e702c20d58829f5952e4d62d2fe 100755 GIT binary patch literal 43300 zcmeHwe|(eIb@#oJ9$U6$TLc4kFwG;7u>so%a2kiYjffqGq!3&}Q@16xWJ|UcWJ#4| zFez;$FvLkr?51>u&TS0AU=oPgmTlzIWbvC{uq|D)j+Sg=!|AdY+NCR>x9LLnN#pl> z?sFemQba=Du7AC{pU>5~=f^$w+;h+U@!UtY9=L5?GdFOSNCsv`nOoWQF2Fj*o_J4X z%#M=Ea#=g2JhA;t{gJHQtiyqL_yH-nGFPl7Kp zkW!gNeUz#G8Sq6WU0N1xlZ@yP>|>M}Ypl{TQ#1~J#=z4uWBJRFJrG2{oNRe-B)r`3 zYhSgJK09h7!P*r<4L-Pv52R_N2d1ZgN3iIDwrQeSx!GhO@39-0NkGPPVA;n>xthU= zD&{o&*VEJQCz!H8?V@P1vfL|xij^7H0_Ip*!70FeuucLNAd@x=GNnTL2x-3JfC^(C zD{5CP!r%RzyP3w+5@}}ILXfjtOcw9Bs~NR0T%<70ai+gF%UTHVP(ihD;p>Q zzoz&v495YNP&@$US($n2T{Stl+a&{}bZa<*mzGHLO0UF<*el`~nE}Ts< zG}p==ZUJ11I@FJ9f-}i@$9?-Q75wv0o*zKf&uet!<`wrwHtq2Ls|_@-r1>GYU?|_w z2U=9pzIOW~Q{Sli@2x)pZMUG+M!kJeR_hD+qO7fLO}MLrwe@radjo_5?IaVvmNs{uC1sL>Yjb#;cktk&ag@9k>y1v-Oz zIn>)8aQAp4I_~!c5_0aaJJ2=5>I_3rmvgtbhrL@_-^O5+ZKsd>>7%1J9QFpH8`|&l zc0`#c?B2F1?1K`&mv#AX^>)GLE`MXt9}KVRMCl%H82uK-+X5aI>2O1kYHdVytr}sC zz3o(A*4vs?ajTHt0O4-0huqTP4|Z&6>+(jUKG+%x`vTEU_{tA5>+$w#=8gt^taU@k z8-T))-%ahdHg*St5$`>|NUz(^S{pH*QLoV1;|}Tg7kmLva2sn~?+)q3yQ9Id_bwVP z(jjXFM2vy3-o3@!sEriyZ}&zUgMm(87a>hvm=oL%f{@m6U2&ry=IHjdws->`U!aS% z-i7GBZ7bXb70uz^NVKWxAcZS>)jsN-UIq3L{r2M`)TA6EV6Vr4LXsUg)@31 zey=ygYTF|elJ2OR)%tzWI#wGFCa`)|8wqQe7SNp-2rY`S;p{==*F5ik;2LtK$ zb)c{ryeAcvYxE5N_H`ZbU#d_m*jq_5Nlz=fokYRp^f1b8USXZ~+2$3V)IQtVy#7ux zTfJe{)+spJXIrPJqkXn@QvbzATU#g`?84NOwc4U0*xK4WkzkugL`sLZhg^@4am13uR`8{T!fq{ ztpt(w0@^9(N`d)+<;b+(-YjVZoJr*mE7G2|5c$){v}bHWrnF82iO(-3N{ro^RHi+R z_K91O>5Ov^GNmtQKz!bY&thacFC?d0AUOB1&mhx&T!*Zu&+3IVh&?WmZkMHb67~SL zB)F-CFQ=j@m<3oa@+T#g_S#R$GJ*BTw3o~MjYx^4$Woq!?Gmm|NND^00^qbar>9a7 z9LU!rS0HyI-+;UpnG&4^Hi{Bs^?)ie_FA?GnUWoO88V$6mLOLlQxbc)R%A7RHzGGn zoc8V}S-uJIE6AM^C-6Dsdy)NwAbk>3c>tZGe14tHX08J|uOFN_>}wo7ZW$k;a44HzfZl4dpjT{QDYi zko8Ts;k-s`#T<=qT2JRU;4z7R8Td)yr4si5KZ9ePL*hO1oYx}pt=hTObW-Bq(9W~Q zNs0Gs=h>Voi4W5G5BdWVKS1X`Ou>-EAGfl(G2rKc(>#;UG*!nXj&ow8RoVx>aUneq zU^yz%>Kk!>93Z|)>U)aLt8~Pb{BkRMs2}`li7y3C{jHZco^2kw1iVq=vv3}D0I$<< z6Rm$D%{G#K#-CGTlr-ElAGlaUB>#6>e`o><`Tt7A8SJ^<1^X=5Y5m3X*L;e3i5t(O zzhXYg`u~=SoAR~(3;l>66VB%`R8Qb^o~F4;?GxXciW`5LhW{uPXEyB2b54>Bs=;~{=Y(}Eu!_g% zguWjDFG=FR0-WMQZ|}#z8urhu1@Ob*>-N0> z{1ogvFZq81{59}({bzwsfgh9np8)Y{NiJRe&e0L6S!~e zB@I6=>aiOYGMOvi9{;AG*Nq<((0G?elXFBR>?5d*OF82mPY`Z=ywWb z``rXtkETE8UJWYujf?hjR{QISpR-xE@7JN$Zq8=ee$HPD{md)j7Rmpz28*_B(Co_D z+$7p#ehqqE8-*SB7SroI2>Q!U2|dLRsiL2SBdQlNt@=rRAe4b9L&p^9515G?LrP(Z=!_sK9CZ@OX z=?t{bWT4%Wf#%FWTa$ryYX(|l23k`F+HDzV%^7I7XP~XkK%*E=_knmWPxIAhGsvyW zKwF=ICZ6BY^wIr9db=$dXrIeKqxCDjF0DD~G+JlUX`jzPqxqOF_k|3!)(kXSW76x2 zJC3yeihF}J+U5*(@6AB_N(LJ3mFexeh^D8Z!;DpqF_WE7Up5}RWUqW4aLQGILj%qz zIEee&gd)H4V8SyjNc{crn&zKY}|HlQlMoRguT8RInEvwvs7kYM@#T=9)oU ztnc_@6cc#d-%y6rXTurWaeE3KXP(Oy$FWKc-5lREJ<80q&WN_9{Lsy@STe1_g!&3{ zj;0UwTG)tM!=9_EVdot+gZ**;;8;8~_-d?gu%EHPF~~vZ;H#)JqjLvys$xx}wvY-d zpJ!O-aNnh(KQfU&rk`Wx=@*2}+-`fj-)Nh%+0lE9Q@?K4>A`rcDb~%JVj=SPfKG3U zx!CEw#1r(vtI!kj&@%w}flTtG|Ma8rSh9XjPf6+f3n}Yo{#k3wfp0J*YtC|e)T{yvA zd~dhyUrs>p)4ZEM8nXTFhm2JWLND>3;(Zlcpcim~OX}SpTU%Kgo18wyW0ezZa(aRf zK<}`<2zsC9`!Bq}hc6uDG2@$N+j$%G9^wPWpIL41twBsoQ@n%!EZ<*wGxP%9plW&% z_bG8PG0N4ULtL%874IeSYL@y0ntkXAZm;UV3gOc34l9qc$)R!3k8zddt&QBR0(XwIZy<^F^aWW9JSrbMqo=gWbURZJZgjR;?B1M!`D>o{crwjmf;y3S74?LYu1I7pjB)3}f$8 z(Pk6cTsLuqZJIdBHcyN*5A5Gut}-LZSs6P{aHt>aWdBa<&=_mNIBiD%NA><=+^vE( z2Js_RORb~gy9unaKmHcvJ>YG^Sm0W@ev+Mb*v@0WPOgt}vC46=Vq7*OUaW|brE}Qy zuc&S4c{@e^DYn^eVSJ8X`%ZHrsQUdFa7diT&b2EBMC~eDs6mec>g()VY}F zI~T3My2bJ~yo7!&MqV@v_fw*cr=aHqWDC1qQG&rm$RdyrL~8)H?7vr^3ElCw>S>B#l>z7%#-$ zg=4VgNsJ-+xCRl#ZkPoL+7kbusc+)6Q;EjXA!>-871G z;6Ya26z`L?uf2J0IR5gvZ^qA}?xbE9dMPe5>V+@uST_quf!$Q_RmigU(fL*k5iR_DQQc56)PQ8=3(!&;7uNzzN_^$S>Y>iV5z z?;D9a2V&>-GW-m`GDq^dLUk#n`}^Y4Oie#y&{B~5Xc zXtz1mp^EcnhPs)~p%lMOyJqTQT+5Tj)s=3;)y}76GyGZJ)Srln57b3pX3p+uzJtG9 zh|hY&XLMqW?Ld6?V?7?kdc1Gq7#o4^!3^sb!yIB*x5Q>6=T$e_NvzFy_ouBhV{9MR zWf3!#d*YX;k1!|3TCDMv560h_KFXZ3ejjv3F(-ClPV`?G2R%t2WTN(dtoH|EQ+ga= zz3#`HjbhI3;Gd=M5v;vV3C?-NxdhrUHY+krOsL2pUr%@rXF z%M;JXs6%UB|IiDp1Z&>X*;A;SR7XBfq0Ztd)Hx-1n49)n8K0naJ*n=zspcA7M=l%z{t^C@ei#XeD*!B?wrHj%@uMem0l-`2!h(1P(VX?l<}&>F{7^rfX~j|eq7*SMNF zGoFse2LTO$1At^lEA(hR$xcaJhxq93D9 zA&0%og)zc&3dRZN0gROm`;qG%+K;p`KSN_k@dRHv;6FxZ89Wb7((_RInIz*}!a2!) zk#Wp@I#VTLqGWV>2N@yTxpV<*by{5Ee71{K?vB5QdJJt3QmkrzS{Or|jLP~SoWF-5 zpSpk6V_hFjUe~ccwP1a6PF!Vuns|~O#@O!re_W#;7q(%(q^(gib4y#J8?I%ZnNa62>YC&{6F$Yfp|#1P$0lS(G{2>+Lsq$#Smip@E^Qu!yqRKQ zg2v$ck{k-0WI9XaU zrR~9F6ZXuM=dP>mnK(n5oA$cMB+ShY6=x?)(_pv6x41IhJ0;!MB=+8+3D|Oqnd$7< zG~|-?q3>zvd-TKgW!zu!Op?4e4?PS0FE9(8SDOx8T_60P*el^nEA|{d0Z8i?{7YXU zfnRY-L4xeex^NVCp4$9L8EecRZA~3#hjCS?t>?$sVS7HFBbLhb{1L3@PtB}@m4jH% zk7FI2S>z+dID?o1gjG9IF9cl zV$H%jM9(=<*i3O=!uF^(oc))U#1b)0=k!6uUnxDmOdJ>I_Z^65it|#wY^@E?{7dJk ziI^_E=9r$q{*xMmrO7dfJ){c$`snka7IP(9%(+B=#T+@sN}Kj4`749Ij4_>lW;}?# z9>SMA7Mg8M@vG~Dturx=Jp}iMd;$>u?WY*m{5t(G83@^#m=J zcmFi8$C(m)++Sjk`%Bm)=E1HOifg&Y)gxADk259jap(RF_Iw-r=#3*U(wf;6AD3;P zaD( ziDy~8y_8udPT@|b80WJI=EOZ*!Nk+Ja~Wl8k=NM^A$Nkc;6B5GxwFY$z^oHbu~w{0 z1=t@pL&k;dvFAbV80*HKZkZToetRz9Q5M4eLjhta3Yk9S9XPw0CdOC{bF2XCWj|;G z$b)nzH1P!P;tuhB7?WYd>Qbx9-V(k%%;`)%tl!JhyA}A=0$;wRpYgGW%@MLkQU3@V zrr5<@&Qj~7Tz@1DwhRwt&<9;ZK{S%)oC_W!=}5h<|o8R zNt!+4`4TZn>l)qtG+nK(9OKEf`q-Kg6CaGR*@FJ@d|ZeA)S2E)>^t;#bE?pj!N2GW z-Tx%w9s8%$0puU-x11CAG2pQgR!Z>vYbaX#>ytcsDwTw)Wv2Oh3zY#OU6Y<*ZL&SV&za=9AMg|U0RQi&z7D=h*4EQly_4jJQJiZ=u|A(~ zurR}ogE5XZ`gN?Q&rlzzF1~hOLG|orqiWAHwBk<(Dsb)@*}Da2_hQV037iF=2L5+A z|KSX6FRU^{UrFV0HV$1Bb61kuyCn6}_`avp6SO^87hm0TdFVi#rSV3{4(thtgSbiO zA$HzQ)IEaofjF$x-oc{}mnClm^VNd)G^E#}^R;n%(2PEUJ`>Nt9e765#`Xxij2KDs z_k(J(tagvUL-@WqZU!>lX{O#);@M^&o=NuOoI3J>yGq>G;vQy^rU$-O46PV9ya&V? zaGX`+yy?KRRUO*L{$GuIk%DTHr9M)OlYf%?FZ)hDu_pSgjVXK|FG-;th?nX(-ld^G z9B*l%8~4ofr^MgAFgv_1ZBk)VHEeRgCTUNV)xz>ke?WV11ED3+uLJS&guRD&pVi54 z)!W9qt1IZAs`Jr@D>%c>ozWgli%U^?icHO@8`*Qf+Eca^E6|GvTrecyB+ijGdG8-Ga{9r}4fqhB5eiNq>M~opE|2-rvXYth)?* zjApk9v@y_jtIJAa3H^;iK6mO{hTl5#rK!$il^_PzT#cdGW1omrxdhjdCXWgxeO20dVg;h`uX(dWxxJd>bY6a zjnf^}uaZXgZArgZ(3R=w3x^YZ)AZ+nb`&&i>>iTxZ9+b8`h@J)E=d>qhaQP?O z^l04m?>`oNpWq*eyH4K-{&!G+-U;W>{|DAK_@wn&W zZw3F0g1-y=uY>fi7y=g`i$=Ph^u^*<;0yW_1dkAfeM@25M}q5gR5TQR}EL+}UU zt*;>JPw$S8H!C(a^dR_eL2eJ~y$Jd))H7Q3XFxGOPVb5DA9^U>`tOT{o|}Z8N8;Vs zr&z@noaGC;UgG|i#e7}Mq7`@ITw2if65@X`@*?=N`^0^s-V#x7FY28@y%Uft-1ZWG z+O?QJ>RPm7knrG3m?Mjk7s1cnCytAHB98v2*W&cE75pdmnM`Y7iSzwQvfXZ4!>~sr zzt4J3UoV9BY`k*83Nbo6M-o{~9evj(|{QYsy zX^Y0EdpK=xOR68oxK(5P9*?)4{w3f6Ks@CF)+O558Sfs%Dvs|b4Or!!51nT&JOSK8 zp@Y8VFay?x0qYU*cZLkl6-LaR0qZ4p9q}QAC>4607M3$bWza-jwAEx`xuud;KoGRG zplz1497$UT8lp{9)ck!QUa#9}H+?3CdJQ^mW$%l%gZ53#`Ot;uxcClA`uA^0Uve@| zY3oZeEnCidAnSpw2eKZ>dLZk8tOv3l$a)~_fvg9z9>{ti>w&BXvL48KAnSpw2eKZ> zdLZk8tOv3l$a)~_fvg9z9>{ti>w&BXvL48KAnSpw2eKZ>dLZk8tOv3l$a)~_fvg9z z9>{ti>w&BXvL48KAnSpw2eKZ>dLZk8tOv3l$a)~_fvgAq5FU6T#u(*WWKJ$;q7h%u z#7{1#|4gO{ZsqI{y}_C;?GVCi?qt1F{jB$_g_5ojB2`O#gcz7nx|{{}2F@ zEYS#(9PtPzKGhL_Q?AgtG#Ydq+0dRFSMh1(XcAa+U5@T^oGc~3JCP#$KBCBw3D46qB7aK zs#A@?ZZxf~5A`h<3R--9UVmgcgdpMfd$it2JHq%SbLgO?|Jr?;ZTMk%Gzi;5eX7^p z(JjiF!6{~Dw8PPYlpSq@yYKyNRiHP32lz!HAZ2m+7z?$_o=)Yk1vC;-77=<<8jXm>c++tnTFjjjeW(&NU^>dc5bb^*hnJHsI0jB-_EGfWSaOHW-eo2-D?SfG+QG_X#aB zf@Yj>mmWqU=G9JL7=Ekrd!rHb%d6gb$0oHS*c0;ky_y-?a3@@-s@?$ooV$!x^13h3 zrB;2ZroFeb(;N1Bpc4KdFO!0*Kv2~NWr?ms{OG&8BkJ4gRU;u^AOHn59lh;dwI}HD z=I1B*#HSf`d-EO58)_onK*T3%1q0qsCB_#rVRr<-BM*m%Yubao0Z(l{Yjg(!!6>>C z2zWc9YBZ>}su6E^t2eC0;mW$2c3)KXLNi_6>WlQc{VLV#BsJ9-)$P=~6`v&G+FIz< ziq~3i-NK)ih+zpo?BOkoDQE2z3dp>QySphP@xy#3ZSy8Xg-eZFr9 zc>}kvQ&-eF)YcU%)J+&KF?H?{eO)aL!bC?0Y2mnSi&?oX*z5PG5rkAA>ff%0+>wat#Yk*dJHzg-9#Bx*=Y@aqf*dsH_p&<*Vk)atc2dOLd~G;SDmIWyc? ziQN=velHpi^!BuS!x&9%zbXC;!*Tj&ELO1t3io{E$H$c~{_`Db@t@z{S3K6&R6NGc z7LHzdxp4Himcr5ZUoRZ}?Rn+xMtMhkZ}sfD}h`ik2hdA+#fQ?6py7p=vvCRTV*d9S!L4w!Qi zIflF+x<`sn{^k_@?k@e-F8#hP{iZJcjxPOnF8y9Ee$&>0jQ{!ga6NJZvJ=^r>`;Br ze<8-06O6_Oa7K9TwI6;O`IfQ+-<=!<9r=0SpP5x#aB1;RzkSOT1e4MWmo>yG{<9Ot zQQ~WA`(wlr`!XPn@do@4s$gd)N)aYDNR+VbVuNB#?h=3^U;(?m{t)Qh5aRzrhXj7R ziA|gE|EiIQxV+E5^~M{QZ|U{i-&^Ajg;%Ux9u5Yh%X=f?<-S0Nzt`hs*C4YzvORJw za@QjBKdy^)grh5>-o9w9*r>zaF6>UV-E_uElyQEHbooG!XpBxAfutB%?nsYD)(Ujk zWD5PHv&;Av<5p|QEJKblpBs$EZG4u28*}li#m24lVWN**Ze0Lg4t~7Yct5wIXe9F2 z=|f4BKrB?f>a2`#AlQI zEQi6{W$Srd*dLV*7p2-?Xj4j{9`Q2M19Y*_Q?A%6m1V_B{d{HphcyGeqLW3`$&goc z(j(itfopwSXjAg4u}+p)!KPKQuC5MU%rTlFIGa3P*rHgQl)_EgXM?g}o>Dklv6g_8 zjMOXWu1+czd=EU8&k@EZqc_nk2BTDkPfc)Q9R`z-C7FPbaFmqJ@~BIyl|q`I{>icQc65sXtgRj;?JaK9o}aG-T5s8uPfvMIBKM#N`bgK}MkQd+LevMWmrUpE@K za?>JZNy&$OmSMrq;dXo)5emyHm9kQlO}t@FgMic?W-iL>IpXVPzM!#4RX zHL8W!h_qT8L@iY3bS8PYtt}FDhs9z1L|ab>o;O7K^|rRK-_sUszpt$u7YdMMm)hFa zguAv<@ro#RP?6Gnv3rraKS6$^tqq4{sgu~>6zm`1rh~3|7WZuoiY7j}T^5?}mxX&G z9d5sP;_m9`4To_c-_U*^9-Bx>%oBER+aw=-iExMr8^IM1!$f(7D2;f4BjhbYHh|mh z^@M}?Z=j6-oECE8F+PhM&GXH(m0Js}I1w0XnbACpU&mJ$e5&9Ui!-0faVI#&I&cj6 z1)zS6T0(t=d3J$RU&l!C%>_ntgM}19t4=WF6TwuQAfz;}%+QmJ$`MqB^d(VqXQ(MU z&{_0hrn9s#*kwNvg8A8mnEE8Etu zTHX=qU7mDo%Wwq!Xay`r0S^20@T3D9c2__1a>HJn?j?o;wn2|tVk7sl)7%hZ2@L<8 z+n5&2iV=?jEhY}Sh-iVkbi9KJ+-cHGI3Yao$=}Npaz6eZuHbLN6>?f>&p3@LD}pc3 zQF`J5#u&~jaG6mU+-BnP+`_0I#zuqTzlxwD*p6ku$St{I8j7AefEVV9*(mTj;6=G& zQVP5tcyX>aH;FJ{1;MJMPEjG*z)k$RoRpc$jgKwBc!?*qRv5D1#ODi%nW-y9#Dg3? z?o66^g%A;QnVTM4Fp5^iGu|NBmrT4#u#;vt30=nViyi`PClS9^g}aLb!EJ33FMf?y zm|#0;#jjp~SFiAprK2a*<_Y%V3W?Y%Sg>;84zODlzO7JuUZ=G&8eyayeqI=}vI_W^ zM-_g!P~7T4eY7`38XVw-tZeaF@b@eHImxFx)wcVkz~U*Z_>~9#oWlRQ5O<<&+KnUi zkQ2+ws*j`oR1W94g0J7okqIXuQhgE+xg!{4Q!eMXD`G*@9V8xpE#OtbNsK)x!r+&` zN8IY$$nO!h=DB3bI$_FtCXU(M1xh{V>BL9mo5cOKX5NyO0(a9r(C$d z7vXAUbK&{x5j_SPf5yu3dv(*CncgEvi97S0H{`AnOnj73fz}B4(qdU)&2ws@Sc1@9W&4^)*;xv7=h?&dv zs`SgFBvUlO(0S8-^jL3c1S$U2a#B9Q6V1 z(gb0xvR%b#*jOZcMgRR5{DoFPzu>E_MJQ8Jk%<>W8*d9<2zHbh{!*_zBU1r;xkt!2 zAY(Cn{(~}$ry%1s;IO7~N|b9vz5dHOO0T1C>b z1GFh+nE|0Qc~t?vN5kbR0>aUmRRh?ogeBGp(KaGl%r^O&O3nfrSv?6p&o>Dik&PZ; zvdkon{{x**(Ojv`J55ILjW6navGZVAO(elLt=IXQXLPPRvU>GRwJU2^tWF#e`fx{+ zLKM4YA41TxYNgI#?OrO@)vXr0n)rR@lxo^yg<5I#l5kY^dursXnboUnaM<*>X*U2W z-ha5|wSJr57ucfKc>SGfP21gHynE9ZTW)K+`<{(;YE6$X6j`O#;Gm<{1of9aHS5H^ z0nS0IRxTG3%fl-iNa_bUDI~SVPj}%peq4;#_Y(+K&|VdO#Ny~q!i#SR0G!_MRX47|TAcCuecK3o=i2G}XI z``l1jY*H_y#>*(5fnBcwz7G5>G`s=1^Hj6n2uezyVW;Rj+7ys*>J3T9jgP`QK4tD&{_>^0QWB%hJ(J(j_y7t3OAeXA^X3BF5#k4YRO^%h1- z@Cd@1x8PYCp&hXJZSonF!MjXlj7LUDG8o-=;KwVFdGEopI72oen+M9`7SKkadlWJ> zu3|hNg}v9c?19AYvbYtwq`xd)It*&z$#g1olq0Lqk%*T+DvR5p>(nD<@hZbFc$yxv z2wKX~8;Y9eP_Bjshn^oQi`PMe6YVx3uSH&mTn`%>9zx8}T7utG#*GFF_yt>q)r#|3 zh}EvYqvh^exr(w{+_n1xok3PBS1MMEh0u>D6+wwa!>kss=(;s*dylLh+1>*&R@;Mu zPQqC)!fNTVpW5=_GaU4|qi$3cpIx{%fbwXa_^fBOk+8;SkD#8%+uqwnOSe}qhkDzw zc6t*$cYAx-yH)2(qeU&e{_9Cp>+B75q}IdE&_!zSKAK(kXcO15YYGM zd61)5AS8DZa>)4I%9%v^2v?DHxigTXOgf21iR4J<66m7@eTFLl$gBmK8p={`0Cgz~ zdq{>7*+wajjOQxSH-Pb>%tL~seUKdGJ;+k-p?=0Fvub2L>2h>F)a6b>4w=>KLV)CN z%8wzF9LD3JOOQimjmUZ;8A{(prZVh%*nwNhWhmDn>j^4EdOXCK{M|glNzS}ZV~J1l z1*NA_PrBR>Qsut&4ajNFdzz#!_t&X%`@RQvAt#b9_Y;!SJH^-m#J6EI zAqC~K{IgVjfBa9NqwPe}<$j$iH_{4;(S%e&AII}YLN|&dSAc#gNsh)1i5Ea$!Hj+1 z-=xtKpCku1FNd5hRnA0ywn#~7AFoZ7J2VRZEabv8IVW&h4`e!odt-j5j_L6K0%@KF AzyJUM delta 9083 zcma)C4^&jwng8y4Gs6Ib^VC5Y7@>b$;?lJXpeqD+?CkcPE@S0RjE`Z zRrKnrw6r`MD(FJK)vPPcx(>=dg%&8NrPdGIvr)?hn|{@gIsyDH)Qc3VY}coy%)7H6 z>lQt~{_ve{C>6p3ihL`0lP9rGzWUC&QHoA=kRe+b^! zCIa$TMbcw!4>oV>XzpCqs{h!sO1D~Xv9)h&-O}->zS`PjtK72jk+$ZElhy($9X?c0 zYt)zVo5H`kVqclVk=$YMh*D82APH0lS_8TTbdTO>Uz1XSR;-liTa$D35A62f*TDL6 zjE(xPxHdN*h{HT@v9bV-e9#v`Hqe(qTyBe2#_wAEIzb!ac-$-k$)MYb7LWxEWX4>n z(zl8AhiD<% zi$MWH>x)TBY7@{>&^i#8T+n*GBk8UnBR{AXWG*kQM*AVqMo_HOpmCoG8e&Y;kHodP z`4v#T$-CREW7SvDxIYF^e*p`;UAV=g`pp$i3? zdEi{uL@~V)h4d#A|5Fr8t|+!FQAoJJOEvK_;9_9a#3jIofYVKU2k;4Oqhb?RDI{Gc z%uub#_+oTVTK1Uu%hA0mkDB=5=w7w;nz)%a9y(|>@iyLY7)Z5=I~7Vj1>6aY%}YWd zK1CLBfQ+$0i%#(hafDz<|HlUi48lg#j=c@ z&Hel{}yXsWKGj@BkFzFfC^p3Jkc2pBoj~{ARY|@J&4p-UU8QP#y&4 z5!y|R0i|96#tazc4+@{hNKzC3B?yKOj&Y7~5jg%9m7-x9h?{_UbD9RC5V$c;|8C$8 z$fr$vB<-T^IQb*M`{VpQ0emP92fqQv@i@jF;InafFEA%rETC5uGLEwt#uTUVHjH7c zaavv+eh0WL4!;LnivUB>fP>4`=7>H9#x4-~CjL+0exkJRnE21Y zZ($Q{ied@7AngS7yG^_h_#E(viL-!b;Qy?FgCylElx_jT3_wDXq^ASdmir;2zWY9eggPX9R3Bip*;@&8rTzuk(BOY_~RKi_CNM@MmrQRCA_bNL!r!C z)W~83XDH-pM4%oM->%RS71(R;^HmD@fRk%YydL`;Q%)9B-xfQ_u22&X_~%`9 zkHPpc7(9i%FInNI!1Zw$BdlzQ!zsXxCPqb7z^dN6pgdWcD;E9Z1%(sy7JgM^Qd&?ReNZ&%n)%_vs;5jm5El#ohWPmru`2&9fvtV;$0KQX4FEJov6Yn)>0yw-!Au z>vb8c7QC*=2c`?k5@39#v><$5?a=pS+;nTl=rUvnCXUMN1>r2MzRZn$E<2Gm>br)I z5zOR_<`6}0)BO5O#%gyN?e`YYk-P#rS6mPdY4zbjts(p#{Q5KR%Q&dWS<_iQ3t1w4 zYDo58PDB?p8P?i#FtGS?zVv1PSEL&)9IM((K_^Fg_)^s*Gbw8iBjaz z1D?W+kz%uxgd25YM{n%u^ZQ|MZQM`g4f`=83&%xfSbT2899P}q!jxgJ?3>Pt>{J83 zCchq8TxSX7+~e2Jb0at1Ke2X6lSmn)t4ihi_)N8nB@9a^c z@d0v;50VE{dc)CTGkCE)c{!Mh{#e8TwUKy?*GC+Ivm5CK)E1h2LAz!)ST}->>H+1Z?ByWZ6t*g3=If%o0_y=$elev^4jUF@c|JU9~5ECsn_Qw%lO-r=SwHY_+iTTIVpX7oQi$( zD0%!%q-sBvg35d;&>JJa&jGzr3iy&q8GnPSeM!)J9ckQ8wV*m*BJ>VYz0XcT2Xec? zmjHudYV_HlI7FME)BX`&*)HuErsY&Dd0c+u&n=1!2jO6dRNs3fXtH>Pa*T{WVPiZ5 zqu$6NoSlw2L-KgB;A+lt1cf>>0-wVO;6mvN*YxZ;4$9d4ed^gbn^5F+3Z*Xe^GgC_ zop|BfdGYkN+uXrH3WWzL3`e2xc?wZNv5Ek*$aXW)P+k^fcFasSG8^N|%gG(0JlLL# z>mZClgk7-#MH8-<{l|3TZ2i~U1~G+USHC$MQR5SzV_bi|@n>x-=CaZl?A4%oz}E&H z?RS3MtW?x;$C#DOUC>_CU?F-VW7LZje80>=(yd`tAPtWpV+MGv zY>Pjm@F*zF@!4gsFHyp6q&6q=28DYbCkN76AgwKtH-S&$IC&Gilsr59xpGHo-}q?I zIO{)&>+r1U_$&gs5aZABcH>-ouc8yOi`U6;^)1dp+Vg2S6O&;8=jLpO(C$+&#bm}w z8|+grUu8s(+8D-zrxOpJEhF3QMqU#X*g5wk zY{V&I3GCpTGCu^j_gBq(yyEDr&{gV=-wX;)m)=dixz|54DIy|4O_^ zGG3x7M9;8gN%Cyw)a#-}+*`Xc)gQe$lsLc}1FxrXWvWWJq#AGAe#Z6Go1&%QBo(9{ zBfj3!)Hu9P-;r6_lIEr}%z@`!Tw`MIm^vz2FuDI`@{QFlk!SCldR4Sgg|ZUIrZJup z3wVR@ew zsd#N>kfYie?4E$ZWCMr`1f}3i)?opB5bRNA_to=drGs_>G2w zrqv$DpeA2|gUplu$N{bP{QnyAhYa~n$iD^o0XSItM}K68R{IGC3;Qn_@?Bcsct7Nt z)>S1CP)G}Wq8jqM4SBZ~co%6mxl0RIt+i%R#}h{0Rg*VJItQtXzpd*0`SB zt#w8AXo272k&2G^-6CP8>hx}Z?)hv5r0d^?*5-q*r(M_I>?ao-x1McH_m~E8O}%G2eiQCf1&O}U1-$a zxlVRyb>VeHIQ`lcfAqTQvyhm24R{kzKDHt5MG`h#yE0_FNG4pIWTbSva)B1`=%Iy- zG92eS$hP0)fo<@L!8>j8l1*NTgA&qBp3UTyg4b{IqBGv7`Q%&7^Vw&)Q^g4KVv#HK zH-0xve6C|J)=%LUi4&wc7hhIzwM?JzR(O>ksEQ=k>Cf)Stl%elUF#FgUXfLTU<-xCQ?QCw_Zt&aMHYHJ2Yg?Ol+s2M&R=oL9@A~%6A}CLs-?doiO84gAH#(g^ z?D9GXI|I%^I-N55tCEzl9h5RQL1=G6TTK~zi<~2aqbXx!;7ygJjQ(Z-q=HVQjD9qe zGWze}ebftTOBwyO-8uA?5$9ls$2qvU-8tCqan?8*Q+D3f;jH(?|o?|DdZ-V;D)zv*z+YTRia+Luu)=)mPX^q0f@ zMuy+Q@S7KY+rn>H_^k@RNx{nw@GI*;curO}fS`jHAjO(XI;!d~@5x5)M9a~cXuZ~~ z%L_v)bNt4UK~zxiNb{DaiRIs}7m~YnV(0VIqM+!MoGjduAc3`&{)HsyKWw2{OAcdEssGg zTZ zx$NynxQbP5bZ&X_=?&YPn<(0mw;!_0kd<0RYYJy(%Z9B_VZtExDwKsWXI+H2-YT}G zYAA#UfB*qp6?VPwp01WV?>tERK!C|4d?O!4lozlbm zm)ym#k5e1jm03LOQC?%*5im{2S&Ck=u93eNlc%17b_Y$&?B6VcjgHq9Ero=%&T%l> zSnU`oHaHcIVXx76PB;e1XbnsAy#1O}b%p0;5!Wx6SX$#a5LJ0jnwK@hBBS%@uF8`K zGl(?*F8r!$FSpBfbGysA!=}laRkJ}+cqNOz^}u4i<-nHUZnqi&b?t&-2YCr$UD@T`T8Ldq8y1;kcdSGPxZ9IO@qGf}JFHMnFI zf69%1&IQZ3_n>Ao{-cAhGn_^^HaK&Q?CsRwA6!z%=LFLt72;T7uWb4R$TTa08TL;4 zP%4)rMc$tmAzS^qvhT&}K-WPNg zoSK)Q_a}XHINP=!J@Z$6G{1zKrR37Dpn3lxD6G+aBiYjZf6%K(mPvJCy&cs97xaB2 a%WMx^fa+fTIOO%u=ocWU|F*6ibp1c%QA*4J diff --git a/test/gltest/main.c b/test/gltest/main.c index 06904638..60b9671e 100644 --- a/test/gltest/main.c +++ b/test/gltest/main.c @@ -100,23 +100,68 @@ static double now_sec(void) { return ts.tv_sec + ts.tv_nsec * 1e-9; } -/* Draw one full-screen Gouraud-shaded quad. - Projection must already be set to ortho 0..w, 0..h. */ -static void bench_quad(int w, int h) { - glBegin(GL_QUADS); - glColor3f(1.0f, 0.0f, 0.0f); glVertex2i(0, 0); - glColor3f(0.0f, 1.0f, 0.0f); glVertex2i(w, 0); - glColor3f(0.0f, 0.0f, 1.0f); glVertex2i(w, h); - glColor3f(1.0f, 1.0f, 0.0f); glVertex2i(0, h); - glEnd(); +/* --- repeat-run statistics ------------------------------------------------- + + A single timing is not a measurement. Guest-side results here vary a lot run + to run -- JIT warmup, host scheduling, and (for queue-bound work) how the CPU + and painter threads happen to interleave. Reporting min/median/max across N + repeats makes that visible instead of letting one lucky peak stand in for the + truth, and the per-iteration list shows whether the first run is the slow one + (JIT cold) or the spread is genuinely random. */ + +static int cmp_double(const void *a, const void *b) { + double x = *(const double *)a, y = *(const double *)b; + return (x > y) - (x < y); } -static void run_bench(Display *dpy, Window win, GLXContext glc, int w, int h, int n) { - double t0, t1, elapsed; - long long total_px; +/* Print min/median/max of `vals` (n entries), labelled with `unit`, at + `prec` decimal places. Higher is better for every rate we report. */ +static void report_stats(const char *label, const char *unit, double *vals, int n, int prec) { + double *sorted; + double med; int i; - /* Set up orthographic projection matching window pixels */ + if (n <= 0) return; + + printf("%s (%d runs):", label, n); + for (i = 0; i < n; i++) printf(" %.*f", prec, vals[i]); + printf("\n"); + + if (n == 1) { + printf(" %s: %.*f\n", unit, prec, vals[0]); + return; + } + + sorted = (double *)malloc((size_t)n * sizeof(double)); + if (!sorted) return; + memcpy(sorted, vals, (size_t)n * sizeof(double)); + qsort(sorted, (size_t)n, sizeof(double), cmp_double); + med = (n % 2) ? sorted[n / 2] + : (sorted[n / 2 - 1] + sorted[n / 2]) / 2.0; + + printf(" %s: min %.*f median %.*f max %.*f (spread %.1f%%)\n", + unit, prec, sorted[0], prec, med, prec, sorted[n - 1], + sorted[0] > 0.0 ? (sorted[n - 1] / sorted[0] - 1.0) * 100.0 : 0.0); + free(sorted); +} + +/* Shared projection/state setup for both benchmarks. + + `depth` enables the depth test. On Indy that is a split between CPU and REX3: + the GL driver performs the depth comparison in software on the MIPS side, + then hands REX3 the 32-bit result as a ZPATTERN coverage mask for a 32-pixel + span. REX3 draws only the pixels whose bit is set (see process_pixel_zpattern + in src/rex3.rs). All rendering goes out in 32-pixel spans, which is why + ZPATTERN is a 32-bit rotate reset per row and why LENGTH32 exists. + + That makes --depth interesting for GFIFO work specifically: each span now + costs an extra register write (the ZPATTERN mask) on top of the draw command, + so depth testing *raises* GFIFO traffic per rasterized pixel rather than + diluting it with rasterizer work. Expect it to widen queue differences, not + compress them. It also costs guest CPU time for the software compare, so a + slower queue and a busier CPU are both in play -- read the two benchmarks + together rather than either alone. */ +static void bench_setup(int w, int h, int depth) { glViewport(0, 0, w, h); glMatrixMode(GL_PROJECTION); glLoadIdentity(); @@ -124,36 +169,168 @@ static void run_bench(Display *dpy, Window win, GLXContext glc, int w, int h, in glMatrixMode(GL_MODELVIEW); glLoadIdentity(); - glDisable(GL_DEPTH_TEST); + if (depth) { + glEnable(GL_DEPTH_TEST); + glDepthFunc(GL_LEQUAL); + } else { + glDisable(GL_DEPTH_TEST); + } glShadeModel(GL_SMOOTH); - /* Clear once so we start clean */ glClearColor(0.0f, 0.0f, 0.0f, 1.0f); - glClear(GL_COLOR_BUFFER_BIT); + glClear(GL_COLOR_BUFFER_BIT | (depth ? GL_DEPTH_BUFFER_BIT : 0)); glFinish(); +} - printf("Benchmark: %d x %d, %d quads\n", w, h, n); +/* Draw one full-screen Gouraud-shaded quad. + `z` varies per iteration so a depth-tested run actually exercises the depth + comparison instead of rejecting/accepting every fragment identically. */ +static void bench_quad_z(int w, int h, float z) { + glBegin(GL_QUADS); + glColor3f(1.0f, 0.0f, 0.0f); glVertex3f(0.0f, 0.0f, z); + glColor3f(0.0f, 1.0f, 0.0f); glVertex3f((float)w, 0.0f, z); + glColor3f(0.0f, 0.0f, 1.0f); glVertex3f((float)w, (float)h, z); + glColor3f(1.0f, 1.0f, 0.0f); glVertex3f(0.0f, (float)h, z); + glEnd(); +} + +/* Fill-rate benchmark: few GFIFO commands, then ~w*h rasterized pixels per + quad. Rasterizer-bound, so it is nearly blind to the GFIFO implementation -- + use --tribench for that. With --depth it measures Z-buffered fill rate, where + each pixel also costs a depth read and conditional write. */ +static void run_bench(int w, int h, int n, int repeat, int depth, int warmup) { + double *rates; + long long total_px; + int r, i; + + rates = (double *)malloc((size_t)repeat * sizeof(double)); + if (!rates) { printf("out of memory\n"); return; } + + printf("Fill benchmark: %d x %d, %d quads, depth %s\n", + w, h, n, depth ? "ON" : "off"); fflush(stdout); - t0 = now_sec(); - for (i = 0; i < n; i++) { - bench_quad(w, h); + total_px = (long long)w * h * n; + + /* Untimed warmup. The first timed run otherwise measures the emulator's + JIT compiling the draw paths, not the draws themselves -- observed as a + ~25%% low outlier in the min column that recovers on later runs. */ + for (r = 0; r < warmup; r++) { + bench_setup(w, h, depth); + for (i = 0; i < n; i++) bench_quad_z(w, h, 0.0f); + glFinish(); } - glFinish(); - t1 = now_sec(); + if (warmup > 0) { printf(" (%d warmup run%s, untimed)\n", warmup, warmup == 1 ? "" : "s"); fflush(stdout); } - elapsed = t1 - t0; - total_px = (long long)w * h * n; - printf("Time : %.3f s\n", elapsed); - printf("Pixels : %lld\n", total_px); - printf("Fill rate: %.1f Mpx/s\n", total_px / elapsed / 1e6); + for (r = 0; r < repeat; r++) { + double t0, t1, elapsed; + + bench_setup(w, h, depth); + + t0 = now_sec(); + for (i = 0; i < n; i++) { + /* Sweep z back to front across the run so the depth test has real + work to do; harmless when depth is off. */ + float z = depth ? (float)i / (float)(n > 1 ? n - 1 : 1) * 2.0f - 1.0f + : 0.0f; + bench_quad_z(w, h, z); + } + glFinish(); + t1 = now_sec(); + + elapsed = t1 - t0; + rates[r] = elapsed > 0.0 ? (double)total_px / elapsed / 1e6 : 0.0; + printf(" run %d: %.3f s %.1f Mpx/s\n", r + 1, elapsed, rates[r]); + fflush(stdout); + } + + printf("Pixels/run: %lld\n", total_px); + report_stats("Fill rate", "Mpx/s", rates, repeat, 1); fflush(stdout); + free(rates); +} + +/* Triangle-throughput benchmark. + + run_bench() above measures fill rate: a handful of GFIFO commands per + full-screen quad, then hundreds of thousands of rasterized pixels. That is + rasterizer-bound, so it barely moves when the GFIFO implementation changes. + + This instead draws many *small* triangles: the per-primitive GL work (and so + the REX3 register writes queued through the GFIFO) dominates, and the pixel + count per primitive is small. That is the workload whose cost actually lands + on the queue, so this is the one to use when comparing GFIFO backends. */ +static void run_tribench(int w, int h, int n, int tri_px, int repeat, int depth, int warmup) { + double *rates; + int r, i; + + rates = (double *)malloc((size_t)repeat * sizeof(double)); + if (!rates) { printf("out of memory\n"); return; } + + printf("Triangle benchmark: %d tris, %d px each, %d x %d, depth %s\n", + n, tri_px, w, h, depth ? "ON" : "off"); + fflush(stdout); + + /* Untimed warmup -- see run_bench. */ + for (r = 0; r < warmup; r++) { + int wx = 0, wy = 0; + bench_setup(w, h, depth); + for (i = 0; i < n; i++) { + wx += tri_px; + if (wx + tri_px >= w) { wx = 0; wy += tri_px; if (wy + tri_px >= h) wy = 0; } + glBegin(GL_TRIANGLES); + glColor3f(1.0f, 0.0f, 0.0f); glVertex3f((float)wx, (float)wy, 0.0f); + glColor3f(0.0f, 1.0f, 0.0f); glVertex3f((float)(wx + tri_px), (float)wy, 0.0f); + glColor3f(0.0f, 0.0f, 1.0f); glVertex3f((float)wx, (float)(wy + tri_px), 0.0f); + glEnd(); + } + glFinish(); + } + if (warmup > 0) { printf(" (%d warmup run%s, untimed)\n", warmup, warmup == 1 ? "" : "s"); fflush(stdout); } + + for (r = 0; r < repeat; r++) { + double t0, t1, elapsed; + int x = 0, y = 0; + + bench_setup(w, h, depth); + + t0 = now_sec(); + for (i = 0; i < n; i++) { + /* Walk across the window so successive triangles touch different + pixels (no degenerate all-same-address case), wrapping at the + edge. With depth on, z cycles so fragments are a mix of passes + and fails rather than a uniform accept. */ + float z = depth ? (float)(i & 255) / 255.0f * 2.0f - 1.0f : 0.0f; + x += tri_px; + if (x + tri_px >= w) { x = 0; y += tri_px; if (y + tri_px >= h) y = 0; } + glBegin(GL_TRIANGLES); + glColor3f(1.0f, 0.0f, 0.0f); glVertex3f((float)x, (float)y, z); + glColor3f(0.0f, 1.0f, 0.0f); glVertex3f((float)(x + tri_px), (float)y, z); + glColor3f(0.0f, 0.0f, 1.0f); glVertex3f((float)x, (float)(y + tri_px), z); + glEnd(); + } + glFinish(); + t1 = now_sec(); + + elapsed = t1 - t0; + rates[r] = elapsed > 0.0 ? (double)n / elapsed : 0.0; + printf(" run %d: %.3f s %.0f tris/s %.3f us/tri\n", + r + 1, elapsed, rates[r], elapsed * 1e6 / (double)n); + fflush(stdout); + } - (void)dpy; (void)win; (void)glc; + report_stats("Triangles", "tris/s", rates, repeat, 0); + fflush(stdout); + free(rates); } int main(int argc, char *argv[]) { - int bench_n = 0; /* 0 = interactive mode */ + int bench_n = 0; /* 0 = interactive mode */ + int tribench_n = 0; /* >0 = triangle-throughput mode */ + int tri_px = 8; /* triangle leg length, --trisize */ + int repeat = 1; /* --repeat: timed runs, reported as min/median/max */ + int depth = 0; /* --depth: enable depth testing (Z fill rate) */ + int warmup = 0; /* --warmup: untimed runs before timing starts */ int i; Display *dpy; Window root; @@ -171,6 +348,31 @@ int main(int argc, char *argv[]) { for (i = 1; i < argc; i++) { if (strcmp(argv[i], "--bench") == 0 && i + 1 < argc) { bench_n = atoi(argv[++i]); + } else if (strcmp(argv[i], "--tribench") == 0 && i + 1 < argc) { + tribench_n = atoi(argv[++i]); + } else if (strcmp(argv[i], "--trisize") == 0 && i + 1 < argc) { + tri_px = atoi(argv[++i]); + if (tri_px < 1) tri_px = 1; + } else if (strcmp(argv[i], "--repeat") == 0 && i + 1 < argc) { + repeat = atoi(argv[++i]); + if (repeat < 1) repeat = 1; + } else if (strcmp(argv[i], "--depth") == 0) { + depth = 1; + } else if (strcmp(argv[i], "--warmup") == 0 && i + 1 < argc) { + warmup = atoi(argv[++i]); + if (warmup < 0) warmup = 0; + } else if (strcmp(argv[i], "--help") == 0 || strcmp(argv[i], "-h") == 0) { + printf("usage: gltest [options]\n" + " --bench N fill-rate benchmark: N full-screen quads\n" + " --tribench N triangle throughput: N small triangles\n" + " --trisize PX triangle leg length for --tribench (default 8)\n" + " --repeat N run the benchmark N times, report min/median/max\n" + " --warmup N N untimed runs first (lets the JIT compile)\n" + " --depth enable depth testing (Z-buffered fill rate)\n" + " (no option) interactive spinning-cube mode\n" + "\n" + "--tribench is the GFIFO-sensitive one; --bench is rasterizer-bound.\n"); + exit(0); } } @@ -211,13 +413,43 @@ int main(int argc, char *argv[]) { glc = glXCreateContext(dpy, vi, NULL, GL_TRUE); glXMakeCurrent(dpy, win, glc); + /* Report the depth buffer we actually got, not the one we asked for. + The visual fallback chain above ends at att_fb3, which requests NO depth + buffer -- and GLX_DEPTH_SIZE is a minimum, so the other three can return + more bits than requested. If that last fallback is what matched, then + glEnable(GL_DEPTH_TEST) is a silent no-op: the state turns on, there is + no depth buffer behind it, and every fragment passes. A --depth run would + then report a perfectly ordinary number that measures nothing at all, so + check the granted config rather than trusting the request. */ + { + int granted = 0; + glXGetConfig(dpy, vi, GLX_DEPTH_SIZE, &granted); + printf("Visual: depth %d bits\n", granted); + if (depth && granted == 0) { + printf("ERROR: --depth requested but this visual has no depth buffer;\n" + " the depth test would silently pass every fragment and the\n" + " result would be indistinguishable from a no-depth run.\n" + " Refusing to report a meaningless number.\n"); + return 1; + } + } + // Init OpenGL state glEnable(GL_DEPTH_TEST); glShadeModel(GL_SMOOTH); glClearColor(0.4f, 0.1f, 0.6f, 1.0f); // Purple background + if (tribench_n > 0) { + run_tribench(800, 600, tribench_n, tri_px, repeat, depth, warmup); + glXMakeCurrent(dpy, None, NULL); + glXDestroyContext(dpy, glc); + XDestroyWindow(dpy, win); + XCloseDisplay(dpy); + return 0; + } + if (bench_n > 0) { - run_bench(dpy, win, glc, 800, 600, bench_n); + run_bench(800, 600, bench_n, repeat, depth, warmup); glXMakeCurrent(dpy, None, NULL); glXDestroyContext(dpy, glc); XDestroyWindow(dpy, win); From 972b3d1817ca1d5fd0e0279dea5ed598256d07fd Mon Sep 17 00:00:00 2001 From: technomancer Date: Tue, 15 Sep 2026 02:45:13 -0700 Subject: [PATCH 2/3] make gfifo push retryable --- src/rex3.rs | 103 ++++++++++++++++++++++++++++++++++++++++++++-------- 1 file changed, 88 insertions(+), 15 deletions(-) diff --git a/src/rex3.rs b/src/rex3.rs index 488b2fcb..45531e29 100644 --- a/src/rex3.rs +++ b/src/rex3.rs @@ -3,7 +3,7 @@ use parking_lot::Mutex; use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, AtomicUsize, Ordering}; use std::thread; use crossbeam_utils::CachePadded; -use crate::traits::{BusRead8, BusRead16, BusRead32, BusRead64, BUS_OK, BUS_ERR, BusDevice, Device, Resettable, Saveable}; +use crate::traits::{BusRead8, BusRead16, BusRead32, BusRead64, BUS_OK, BUS_ERR, BUS_BUSY, BusDevice, Device, Resettable, Saveable}; use crate::devlog::{LogModule, devlog_is_active, devlog}; use crate::snapshot::{get_field, u32_slice_to_toml, u16_slice_to_toml, u8_slice_to_toml, load_u32_slice, load_u16_slice, load_u8_slice, toml_u32, toml_u64, toml_u8, hex_u32, hex_u64, hex_u8}; use std::cell::{Cell, UnsafeCell}; @@ -1015,26 +1015,40 @@ impl GFifo { tail.wrapping_sub(head) & GFIFO_MASK } - /// Push an entry. Spins if full. Safe to call from multiple producers concurrently. + /// Try to push an entry without blocking. Returns `false` if another + /// producer holds the lock or the queue is full — the caller should report + /// back-pressure and retry rather than spin here. + /// + /// Spinning inside the CPU's store path is what this exists to avoid. That + /// spin runs with no interrupt servicing, so a sustained full queue starves + /// IP7 delivery — and because the guest's own clock is driven by IP7, it + /// also *dilates guest time*: wall-clock advances while guest-visible time + /// does not. Any guest-side benchmark then reports inflated throughput, and + /// inflated most for whatever configuration spins most. Returning `false` + /// lets the bus write report `BUS_BUSY` (== `EXEC_RETRY`), so the CPU leaves + /// the store, re-enters `step()` — sampling interrupts in `step_preamble!` + /// — and re-dispatches the same instruction. Nothing is lost by not making + /// progress here: if the queue is full the CPU cannot retire this store + /// anyway. #[inline] - pub fn push(&self, addr: u32, val: u64) { - // Acquire the spinlock — uncontested in the common case (one active producer). - while self.lock.compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed).is_err() { - while self.lock.load(Ordering::Relaxed) { - std::hint::spin_loop(); - } + pub fn try_push(&self, addr: u32, val: u64) -> bool { + // Acquire the spinlock — uncontested in the common case (one active + // producer: IRIX only drives DMA for pixmap blits, never while the CPU + // is writing REX3 registers), so a failure here is rare. + if self.lock.compare_exchange(false, true, Ordering::Acquire, Ordering::Relaxed).is_err() { + return false; } - // Spin if full — consumer will drain it. let tail = self.tail.load(Ordering::Relaxed); let next_tail = tail.wrapping_add(1) & GFIFO_MASK; let mut cached_head = self.shadow_head.get(); if next_tail == cached_head { cached_head = self.head.load(Ordering::Acquire); self.shadow_head.set(cached_head); - while next_tail == cached_head { - std::hint::spin_loop(); - cached_head = self.head.load(Ordering::Acquire); - self.shadow_head.set(cached_head); + if next_tail == cached_head { + // Full — release the lock and let the caller retry once the + // consumer has drained something. + self.lock.store(false, Ordering::Release); + return false; } } // SAFETY: we hold the lock; no other producer touches this slot. @@ -1046,6 +1060,20 @@ impl GFifo { // Release: consumer's Acquire on tail sees the slot write above. self.tail.store(next_tail, Ordering::Release); self.lock.store(false, Ordering::Release); + true + } + + /// Push an entry, spinning until it fits. Safe to call from multiple + /// producers concurrently. + /// + /// For callers with no way to report back-pressure: shutdown sentinels, and + /// MC's VDMA worker thread, which has no EXEC_RETRY mechanism of its own. + /// The CPU store path uses `try_push` instead — see its doc comment. + #[inline] + pub fn push(&self, addr: u32, val: u64) { + while !self.try_push(addr, val) { + std::hint::spin_loop(); + } } /// Peek at the next entry without advancing head. Returns `None` if empty. @@ -3409,6 +3437,34 @@ impl Rex3 { } } + /// Non-blocking `gfifo_push`: returns `false` when the queue is full or a + /// producer holds the lock, so a bus write can report `BUS_BUSY` + /// (== `EXEC_RETRY`) instead of spinning with interrupts unserviced. + /// + /// Only safe for callers that commit no other state first: the CPU + /// re-executes the entire store on retry, so anything done beforehand would + /// be applied twice. + #[must_use] + fn gfifo_try_push(&self, addr: u32, val: u64) -> bool { + #[cfg(feature = "developer")] + { + let len = self.gfifo.len() + 1; + let _ = self.gfifo_hwm.try_update(Ordering::Relaxed, Ordering::Relaxed, |hwm| { + if len > hwm { Some(len) } else { None } + }); + } + if !self.gfifo.try_push(addr, val) { + return false; + } + #[cfg(feature = "idle-pause")] + if self.processor_parked.load(Ordering::Acquire) { + if let Some(t) = self.processor_unparker.get() { + t.unpark(); + } + } + true + } + fn wait_idle(&self) { loop { // Acquire load: when gfxbusy goes false, all execute_go() writes become visible. @@ -5184,6 +5240,11 @@ impl BusDevice for Rex3 { } } + // Blocking push, deliberately: `result` is already computed above and a + // HOSTRW read has already called note_hostrw_read(), advancing the + // read-then-advance pipeline. Returning BUS_BUSY here would re-run that + // on retry. Unlike write32's default arm, this path cannot be made + // retryable without moving the push ahead of the read side effects. if is_go { self.gfifo_push(GFIFO_PURE_GO, 0); } result } @@ -5281,7 +5342,15 @@ impl BusDevice for Rex3 { } REX3_DCBRESET => { *self.dcb.lock() = Rex3DcbState::default(); } _ => { - self.gfifo_push(offset, val as u64); + // The push is this write's ONLY effect, so a full queue can + // safely report BUS_BUSY (== EXEC_RETRY): the CPU re-executes + // the store from scratch, having sampled interrupts in + // step_preamble!, and nothing was half-applied. Spinning here + // would starve IP7 — and with it the guest clock — for as long + // as the queue stays full. + if !self.gfifo_try_push(offset, val as u64) { + return BUS_BUSY; + } return BUS_OK; } } @@ -5363,7 +5432,11 @@ impl BusDevice for Rex3 { if reg_offset64 == REX3_HOSTRW0 { // Encode as REX3_HOSTRW64 (0x0231) + GO bit if present. // addr bit 0 = is_64bit, bit 11 = GO. - self.gfifo_push(REX3_HOSTRW64 | (offset & 0x0800), val); + // Sole effect of this write, so a full queue reports BUS_BUSY and + // the CPU retries the store (see write32's default arm). + if !self.gfifo_try_push(REX3_HOSTRW64 | (offset & 0x0800), val) { + return BUS_BUSY; + } return BUS_OK; } From 93d92d6be6450c7e3be6b55c23119e4b694556d1 Mon Sep 17 00:00:00 2001 From: technomancer Date: Tue, 15 Sep 2026 07:53:20 -0700 Subject: [PATCH 3/3] rex3 benchmarking tests jit vs interp comparisons and gfifo overhead measures --- src/rex3_tests.rs | 226 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 226 insertions(+) diff --git a/src/rex3_tests.rs b/src/rex3_tests.rs index 020ff49b..f5889f78 100644 --- a/src/rex3_tests.rs +++ b/src/rex3_tests.rs @@ -3160,6 +3160,232 @@ mod jit_tests { ); } + /// GFIFO pressure sweep: how much can we draw per second, as primitives shrink? + /// + /// `jit_timing_shade_scanlines_fullscreen` already goes through the GFIFO + /// (`reg()` calls `rex.write32()`, the real bus entry point), but it draws + /// 1280-pixel scanlines from 5 register writes -- about 0.004 queue entries + /// per pixel. Guest GL drawing small triangles is nothing like that: tens of + /// entries per primitive covering ~32 pixels, call it 1 entry/pixel, some + /// 250x denser. So the fullscreen number says the queue is fast at *low* + /// entry density and nothing about high density. + /// + /// This runs each configuration for a fixed wall-clock budget and reports + /// how much it managed, rather than timing a fixed amount of work: at + /// ~2000 Mpx/s a fixed-work run finishes in milliseconds and measures + /// mostly noise. Short spans mean more GOs and more register writes for the + /// same fill, so the Mpx/s curve across span lengths isolates what queue + /// traffic costs. Flat means the queue is free at any density; collapsing + /// means per-entry cost dominates once primitives get small -- the regime + /// real GL content lives in. + /// + /// Both plain Gouraud and ZPATTERN-masked spans are measured. ZPATTERN is + /// Indy's depth path (the GL driver compares in software and hands REX3 a + /// 32-bit coverage mask per 32-pixel span), so it costs an extra register + /// write per span *and* a per-pixel mask test -- exactly what depth-tested + /// content pays, and the guest-side numbers show depth is expensive. + /// + /// Not an assertion test: it prints a table. Run with + /// `cargo test --release --features rex-jit gfifo_pressure_sweep -- --nocapture` + /// (add `--ignored`; it is ignored by default since it burns real seconds). + #[test] + #[ignore = "benchmark: runs for several seconds of wall clock"] + fn gfifo_pressure_sweep() { + /// Minimum wall-clock per sample. The loop runs whole batches and stops + /// once this has elapsed, so a sample is always *at least* this long and + /// usually a little over — which is why every rate below divides the + /// pixels actually drawn by the nanoseconds actually measured, never by + /// an assumed budget. + const BUDGET: std::time::Duration = std::time::Duration::from_millis(1000); + /// Samples per cell; the median is reported, with min/max as spread. + /// Three at >=1s each, rather than one: the short-span rows varied 2.6x + /// run to run on single samples (span 32 read 163 / 429 / 229 Mpx/s), + /// and one number gives the reader no way to see that. + const SAMPLES: usize = 3; + const SPAN_LENS: [i32; 6] = [1280, 256, 64, 32, 16, 8]; + /// Spans per timing check — checking the clock every span would itself + /// cost more than the draw at short lengths. + const BATCH: u64 = 256; + + let dm1 = DM1_RGB24_SRC; + let dm0_plain = DM0_DRAW_SPAN | (1 << 18); // shade + stoponx + // Depth mode as the GL driver actually drives it: ENZPATTERN (bit 12) + // for the coverage mask AND LENGTH32 (bit 15), which hard-caps the draw + // at 32 pixels (see execute_go's `length32 && pixel_count > 32`). That + // cap is why ZPATTERN is a *fixed* 32-pixel row below rather than part + // of the span sweep: a longer span would not draw longer, it would just + // recycle the same 32-bit mask over pixels it never reaches. + let dm0_zpat = DM0_DRAW_SPAN | (1 << 18) | (1 << 12) | (1 << 15); + + // Draw spans of `len` pixels for BUDGET, through the GFIFO. + // Returns (pixels drawn, queue entries pushed, elapsed nanos). + let run = |rex: &Rex3, len: i32, zpat: bool| -> (u64, u64, u64) { + reg(rex, REX3_DRAWMODE0, if zpat { dm0_zpat } else { dm0_plain }); + reg(rex, REX3_DRAWMODE1, dm1); + reg(rex, REX3_WRMASK, 0xFFFFFF); + reg(rex, REX3_SLOPERED, 2u32 << 11); + reg(rex, REX3_SLOPEGRN, 1u32 << 11); + reg(rex, REX3_SLOPEBLUE, 0); + + // 5 writes + 1 GO per span, plus the ZPATTERN mask when enabled -- + // the extra queue entry per span that depth content actually pays. + let per_span = if zpat { 7 } else { 6 }; + let mut spans = 0u64; + let start = std::time::Instant::now(); + loop { + for _ in 0..BATCH { + let i = spans; + // Walk across the framebuffer so successive draws touch + // different lines rather than rewriting one hot row. + let y = (i % 1024) as i32; + let x0 = ((i / 1024) as i32 * len) % (1280 - len).max(1); + if zpat { + // A fresh mask per span, as the GL driver emits after + // each 32-pixel software depth compare. Varying it (not + // a constant) keeps the per-pixel test honest and stops + // the value being hoisted; the alternating-ish patterns + // reject roughly half the pixels. + reg(rex, REX3_ZPATTERN, (0xAAAA_AAAAu32 ^ (i as u32).wrapping_mul(2654435761))); + } + reg(rex, REX3_COLORRED, ((i * 7 % 200) as u32) << 11); + reg(rex, REX3_COLORGRN, ((i * 3 % 180) as u32) << 11); + reg(rex, REX3_COLORBLUE, ((i * 5 % 160) as u32) << 11); + reg(rex, REX3_XYENDI, xy(x0 + len - 1, y)); + rex.write32(go_addr(REX3_XYSTARTI), xy(x0, y)); + spans += 1; + } + if start.elapsed() >= BUDGET { break; } + } + rex.wait_idle(); + let ns = start.elapsed().as_nanos() as u64; + // LENGTH32 caps the draw at 32 pixels however long the span is, so + // count what was actually rasterized, not what was requested. + let drawn = if zpat { len.min(32) } else { len } as u64; + (spans * drawn, spans * per_span, ns) + }; + + let rex_interp = make_rex3(); + rex3init(rex_interp); + let rex_jit = make_rex3_jit(); + rex3init(rex_jit); + + // Force both shader variants compiled before timing. + for dm0 in [dm0_plain, dm0_zpat] { + reg(rex_jit, REX3_DRAWMODE0, dm0); + reg(rex_jit, REX3_DRAWMODE1, dm1); + reg(rex_jit, REX3_XYENDI, xy(63, 0)); + reg_go(rex_jit, REX3_XYSTARTI, xy(0, 0)); + if let Some(ref jit) = rex_jit.rex_jit { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + while !jit.compiled_pairs().contains(&(dm0, dm1, 0)) { + assert!(std::time::Instant::now() < deadline, + "JIT compile timed out for dm0={dm0:#010x}"); + jit.request_compile(dm0, dm1, 0); + std::thread::sleep(std::time::Duration::from_millis(5)); + } + } + } + + // Median of `SAMPLES` runs, plus the observed min/max, in Mpx/s. + let sample = |rex: &Rex3, len: i32, zpat: bool| -> (u64, u64, u64, u64, u64, u64) { + let mut rates = Vec::with_capacity(SAMPLES); + let mut entries = 0u64; + let mut last_ms = 0u64; + let mut last_spans = 0u64; + for _ in 0..SAMPLES { + let (px, e, ns) = run(rex, len, zpat); + entries = e; + last_ms = ns / 1_000_000; + last_spans = px / if zpat { len.min(32) } else { len } as u64; + // Measured pixels over measured nanos — never an assumed budget. + rates.push(px * 1000 / ns.max(1)); + } + rates.sort_unstable(); + (rates[SAMPLES / 2], rates[0], rates[SAMPLES - 1], entries, last_ms, last_spans) + }; + + // --- self-validation ------------------------------------------------- + // A throughput number from draws that never touched a pixel is worse + // than no number: it looks like a fast configuration. Likewise a "JIT" + // column that is really the interpreter. Check both before reporting, + // on both engines and both modes, rather than trusting the setup. + for (rex, engine) in [(rex_interp, "interp"), (rex_jit, "jit")] { + for (zpat, mode) in [(false, "plain"), (true, "zpat")] { + // Clear a known region, draw one span into it, confirm it moved. + let probe_y = 700; + { + let fb = unsafe { &mut *rex.fb_rgb.get() }; + for x in 0..64usize { fb[probe_y as usize * 2048 + x] = 0; } + } + let go_before = rex.jit_go_count.load(Ordering::Relaxed); + let int_before = rex.interp_go_count.load(Ordering::Relaxed); + + reg(rex, REX3_DRAWMODE0, if zpat { dm0_zpat } else { dm0_plain }); + reg(rex, REX3_DRAWMODE1, dm1); + reg(rex, REX3_WRMASK, 0xFFFFFF); + reg(rex, REX3_SLOPERED, 2u32 << 11); + reg(rex, REX3_SLOPEGRN, 1u32 << 11); + reg(rex, REX3_SLOPEBLUE, 0); + if zpat { reg(rex, REX3_ZPATTERN, 0xFFFF_FFFF); } + reg(rex, REX3_COLORRED, 200u32 << 11); + reg(rex, REX3_COLORGRN, 180u32 << 11); + reg(rex, REX3_COLORBLUE, 160u32 << 11); + reg(rex, REX3_XYENDI, xy(31, probe_y)); + reg_go(rex, REX3_XYSTARTI, xy(0, probe_y)); + + let changed = { + let fb = unsafe { &*rex.fb_rgb.get() }; + (0..32usize).filter(|&x| fb[probe_y as usize * 2048 + x] != 0).count() + }; + assert!(changed > 0, + "{engine}/{mode}: draw mutated no pixels — the benchmark would be timing nothing"); + + let jit_gos = rex.jit_go_count.load(Ordering::Relaxed) - go_before; + let int_gos = rex.interp_go_count.load(Ordering::Relaxed) - int_before; + println!(" validate {engine:>6}/{mode:<5}: {changed:>2}/32 px written, \ + GOs jit={jit_gos} interp={int_gos}"); + if engine == "jit" { + assert!(jit_gos > 0, + "{engine}/{mode}: no GO dispatched through the JIT (jit={jit_gos} \ + interp={int_gos}) — the 'jit' column would just be the interpreter"); + } + } + } + + for (label, zpat) in [("plain Gouraud", false), ("ZPATTERN-masked (LENGTH32: 32px draws)", true)] { + println!("\n=== GFIFO pressure sweep: {label} ({} ms per cell) ===", + BUDGET.as_millis()); + println!(" {:>6} {:>10} {:>21} {:>21} {:>8} {:>11} {:>17}", + "span", "entries/px", "interp Mpx/s [min-max]", "jit Mpx/s [min-max]", + "jit x", "ms i/j", "spans i/j"); + // ZPATTERN draws are capped at 32 pixels, so sweeping span length + // past that measures nothing new -- one row is the whole story. + let lens: &[i32] = if zpat { &[32, 16, 8] } else { &SPAN_LENS }; + for &len in lens { + let (i_mpx, i_lo, i_hi, entries, i_ms, i_spans) = sample(rex_interp, len, zpat); + let (j_mpx, j_lo, j_hi, _, j_ms, j_spans) = sample(rex_jit, len, zpat); + let drawn = if zpat { len.min(32) } else { len } as u64; + // entries/px uses pixels actually rasterized (LENGTH32 caps + // ZPATTERN draws at 32), not the span length requested. + let per_px = entries as f64 + / (entries / if zpat { 7 } else { 6 }).max(1) as f64 + / drawn as f64; + // Ratio must come from WORK DONE, not elapsed time: every + // sample runs for the same wall-clock budget, so i_ns/j_ns is + // ~1.00 by construction and says nothing. (It printed a + // reassuring "1.00x" next to cells where the JIT was doing + // half the interpreter's work.) + println!(" {:>6} {:>10.3} {:>6}[{:>5}-{:>6}] {:>6}[{:>5}-{:>6}] {:>7.2}x {:>5}/{:<5} {:>8}/{:<8}", + len, per_px, + i_mpx, i_lo, i_hi, + j_mpx, j_lo, j_hi, + j_mpx as f64 / i_mpx.max(1) as f64, + i_ms, j_ms, i_spans, j_spans); + } + } + println!("\n For comparison: guest-side gltest --bench ~44 Mpx/s,\n \x20 --bench --depth ~20 Mpx/s.\n"); + } + /// Verify Gouraud interpolation pixel-by-pixel: R ramps from 255 down to 0 across 256 pixels. /// /// slope = (0 - 255) / 255 = -1 per pixel = -1 << 11 in o12.11 fixed-point.