From 0742bb5969f828c681c9484d6aae0aa564000a13 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Wed, 15 Jul 2026 16:51:42 +0100 Subject: [PATCH 1/7] Literals --- grammar/grammar.bnf | 117 ++++ grammar/graph.pdf | Bin 0 -> 15720 bytes grammar/railroad.svg | 1238 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 1355 insertions(+) create mode 100644 grammar/grammar.bnf create mode 100644 grammar/graph.pdf create mode 100644 grammar/railroad.svg diff --git a/grammar/grammar.bnf b/grammar/grammar.bnf new file mode 100644 index 0000000..104caed --- /dev/null +++ b/grammar/grammar.bnf @@ -0,0 +1,117 @@ + +%axiom literal + +## Data types: + +bit_datatype -> 'BOOL' + | 'BYTE' + | 'WORD' + | 'DWORD' + ; + +integer_datatype -> /U?[DS]?INT/ + ; + +float_datatype -> /L?REAL/ + ; + +## Literals + +# identifier -> /([A-Za-z]|_[A-Za-z0-9])(_?[A-Za-z0-9])/ +# ; + +#number -> /[0-9]+/ +# ; + + +literal -> integer + | floating_pointer_number + | time_literal + | character_string + ; + +# FIXME: Note that - is only valid for some datatypes. Validate through grammar or semantic? +integer -> _datatype? ( ('+'|'-')? decimal_digit_string | binary_digit_string | octal_digit_string | hex_digit_string) + ; + +decimal_digit_string -> /[0-9]+(_[0-9]+)*/ + ; + +binary_digit_string -> '2#' /[01]+(_[01]+)*/ + ; + +octal_digit_string -> '8#' /[0-7]+(_[0-7]+)*/ + ; + +hex_digit_string -> '16#' /[0-9A-F]+(_[0-9A-F]+)*/ + ; + +floating_pointer_number -> (float_datatype '#')? ('+'|'-')? decimal_digit_string ('.' decimal_digit_string) _exponent + ; + +time_literal -> date + | time_of_day + | date_and_time + | duration + ; + +date -> /D(ATE)?#/ date_information + ; + +duration -> /T(IME)?#/ (decimal_representation | sequence_representation '_'? decimal_representation?) + ; + +time_of_day -> /(TIME_OF_DAY|TOD)#/ time_of_day_information + ; + +date_and_time -> /(DATE_AND_TIME|DT)#/ date_information '-' time_of_day_information + ; + +# FIXME: We might want to limit 0001 to 2300; 01 to 12 and 01 to 31 -- Or create then rules for that +date_information -> /[0-9][0-9][0-9][0-9]-[0-9][0-9]-[0-9][0-9]/ + ; + +# FIXME: is the two chars required or can one use only one? 4:3:5.3 is valid!? +time_of_day_information -> /([0-9]|1[0-9]|2[0-3]):[0-5]?[0-9]:[0-5]?[0-9](\.[0-9][0-9]?[0-9]?)?/ + ; + +# FIXME: at least one is required +sequence_representation -> _days _hours? _minutes? _seconds? _milliseconds? + | _hours _minutes? _seconds? _milliseconds? + | _minutes _seconds? _milliseconds? + | _seconds _milliseconds? + | _milliseconds + ; + +_days -> decimal_digit_string 'd' '_'? + ; + +_hours -> /[01]?[0-9]|2[0-3]/ 'h' '_'? + ; + +_minutes -> /[0-5]?[0-9]/ 'm' '_'? + ; + +_seconds -> /[0-5]?[0-9]/ 's' '_'? + ; + +_milliseconds -> /[0-9][0-9]?[0-9]?/ 'ms' + ; + +# the entry to decimal representation is only possible for time units that are not yet defined +decimal_representation -> decimal_digit_string '.' decimal_representation? ('d'|'h'|'m'|'s'|'ms') + ; + +character_string -> 'STRING#'? "'" characters* "'" + ; +characters -> /$[0-9A-F][0-9A-F]/ + | /[\x20-\x23\x25\x26\x28-\x7E\x80-\xFF]/ + | /$['$LINnPpRrTt]/ + ; + +_exponent -> ('E'|'e') ('+'|'-')? decimal_digit_string + ; + +_datatype -> (bit_datatype | integer_datatype) '#' + ; + diff --git a/grammar/graph.pdf b/grammar/graph.pdf new file mode 100644 index 0000000000000000000000000000000000000000..36e4d66cfbf7fc711201f1dfe795bcd2eccd6de7 GIT binary patch literal 15720 zcmd73byQr<7CndrC%6V{9D;R2;|cEW8r+`6^**-a@YB*Fkf?Z4|76duJczy}eSacY+y9t2s1B5ELJg=R&ol#TQteLs{*l0qSXs=0@Ur z5iQm;v+!0Cb`egIAl9;?>3$=$Eetpq{j{G+BT#hUk_hXXj>~&6Fa(zd^ z5eW~Y&B|v>wFnTSeGA2WlL+tY z=eI%B(WuAOpkDUX)dvH2SGW7Uue=#vPu4kfo6XIi-VeZp)y}HrW66Ef!N@~#uBt|I z7U!vzJ{Xl9OH=oZq^9gfyx+Nh{Bgq@v8^;s*qIz|Qe)(;&=H3S6hXd$4O&99ULGq*R1q)K2Xx25-NkZ619cC>6%IhVxK`<6UrET zBRJpBeep@2)>hfMEc+MCHYH>>J~vJxF7ipqc9_BkY5S~?E?hDm9>ueOw57Grdv8@j zU2<}E?+5Dc#_Aip3f52V{e|?dr-G zn?#!y_2tK_qc?9Y1-oZ{TC}=dMNJ+(oqr|zK&<6@y`6i(Vbes?E?Yt3M*MVXI@o!2 zFX^TF;lYPwqH}MQqcD?XW%mBq1>(SwFV#tor7E=B1qt3?9fGVqQTSfh{sT+&R^{3F zSfXCpqFz$ZI0V;&hN~LUkC@6N?8*TzySg#P>F6pGB?S%WzNGekc}iV57$(U0UfMQW z1u04n3m>;n1;<=?V(nXc(+hD<31kZqHz|46yrq7pjYHO2rDx72zPi&l4Mh%UvbZv9 z8#%97G2?zP^#5dz82y-}L5?jNgzN^Fw3K?td{i01o}7nL5}3H%`ku=nsz1re!|ube zg44A^5(_Te!h}l>KEqd$39})Lj9i@(NmX0D&viN01P7^T{c@iZBNG<FydI8+yp5W=85SQvkt(NbnIZWb&c zn2cWF=+NI4n7IC4m-MYq1IT}c2*GZ(S$_H-y9KKDyS=oPVg4V zGg5F>f>c8vkDkv(`s*zY@3+K(aoV>gCRIB9!`DCkQBdKI++P!~g|y_iC(D@MClEhI zJoN+}WumGDgAwhGPZ4`&e+HW}P^4#soV~AF>?Y(-qdfh7Vq!<+fDQI7ZWm6CiZ6V^x_=wXm4EbDYP%y(+ z;$ho;w$m2J=9WAIC+|4CkrAV@qQ zsidC>fLH9>)XYE6e}rQGszUD$$k!c29MR-^C$v8`bf#KmVC{5a-uFo!tg*PGKqDTY zXRtx6A58-RlKcdWW)tE%-b#u*Th53T6!>H}IYyJO83Lc}Ii@URb6Elv9xX4+jj4-> zw)GayQw1|C-kO3d%*^DR!Q3Pe?{k)dGs>OQVt8rvB7^bP5!7c@4V)9E;Y#rOS;#Ad zFc@W{oBD?LLuR zF?YNY*SEx2dnKN;T{ocjuJ{&KZ>M2c^&#^dI}{B&M^UM7M87qBA#N|-+BeuxK;N|h zw|VgJ1HV{aRXBk^MTUVI58w;)E0OUEjgg5u;FX@`_Da3@}m|I@O9%Ybpb~5Q>9kLPkx+h_Zb~f^2 z6^zfRbJmR{Pqnq+>>|oXI>7VjN+Q;Db-6N%Av)sY!?Xy3_sa-r(p`kE^!+~S&)E3Pbc-YONX8GT|odOYDgG*saRqkPry|B$Lk%!TZ}rQ8<)-E=)rO;sX+7^vV1}n zlg?B@i^x0dCG1Rsh?MU@#;jl{oGp6C>jd#!;*Z3NkC@ z$LDUw(s z!o?9PdpREuE8(ldY{j~B;b`BYqQ}!5;tw0-Yb1RcSJ|SQ8~WsW416~c!C5N(^yvxm z*Cq6GEE+OCjH5%!bckQ+Xj|YsDBL%Y04&by^(=Y4di-5#6ntGp`;A;925qR|F?3l& zS~G~w)QL~OeADqmP;Eb}(Vr>kC_+?MTim}g_0A|(m4NHjwNtp>38U&z!XwJ1!?Sdj zZ2rQ-C%Mnmv7ZdV^F#*gphBTzW}@TG)t1%4%p7_}Z1N7b=!%n;(>>>n==bW=zFVC& zPh?~oXyr_G$L}gXdn(@+MVV-QjbD9|Y70Ya*?e0^W>78E6z9MH#F4;vOs{T5UM{}7 zJXYwc#;MGN-)uY+1q-R+3589AO{Av`ouNR8} zpA@VSTAX25T_4N*ehz)Iiddt{Q`!+mADV^t8hon_f1cEL5n3u#KHFIRvCcnSp5vW7 z#_a@WjmEG5qStD7I})5*5jOYCgwXh-9#y>Tm4o(ON?vw@j}P+d(vOmg$E=SeHCP#2 z0M!vhzBnLcR$lhk5hG|w&l!K8e{J=AZK58MrcT+h^aVem*S|PaoE#w?i^=b@AlFz~ zb@OZ3i|-)ZbbN2ON%Un4ha1e%XHl>wMg0it{6NO{t^-bkAZVC1av6^LQ&ND6(uzet zj}k3}gd+~OK%^pMjPsGLF-bhwanBwpOzAH7u{8i*P!xx(06m0ZFWuM}jY=p!ulK9n z9G}n%D{0OtP56+TSG~@ZR?=(uJ@2rD9d>n!?_P8X8xB}nZT(w4fgxSn!}`9MMOAc= z{qOB^M)k=!(dW>EAV zU`)`}KhSEk<>C1X?lZtKav+eyp!OC)2=x>MIPlS!6=GNyxs_$_D-A=UAv|Wan<7Jr zJxrI-=#dGC7Z3IrK_fgbk^euIh^#ve(Yw|8;d~t(w{)yMhh@xmcvdvAfRK(&0b1zM->a2~WEU4Po$nIfO;`aYQk~u{&@}G&#E^S{-MrrHP@v%Z>^Z(@i;Q zjF^8e15XeOuy&MCQ@k9;3hbZQ&xb#NG6Uj8&2IqVb7KSwEaHT=z30#63*im=(5f6} znbt0gGiE3*zeTJGEV(GAH1VD}!t2=QNJa$HHxAQ(7FQ3PDXAo(41KVjk_eb#9+%L5 zgW_6c3xf=jV9h*uO`k57Jd5Eq$%UuA5vmzVS8z2q#JmPJq*dF zG;4=Fj{Y| zLzqU9YI=JjZFQb5znT3^Q+pW?U0yu1l$?p({Crid;31ZAxV#-)+EVP{K&;j`n3kEf z71@ed3eqvV;bG9d^g=z;%`XAut_<~Zi5$qH67|kYx|LGR<80Q?5)$G<#%A@$v4ot--cpF3~Bv@v%!%xoBn(cgbm}?VRVb#k$4F|nF(XG*(dS9v6XxEf<++Z?ogT^E+ zm(NzMRiYy>)!mnRA%u|Y`*vE7LPus|0nTN#`pxMYjgh*lhEK#;+J#03012aF77*ns zuxkau_!X-EP47$|PC(_9sDjiE|vyoS&~%VlfR+=ibd<9!3BRA7ROCzI3rVBQZ9*}zb#%&jZb>utXNb`W-=oY zl22I?l!7b;`qUj`WymGA?hhKscN=Hv3p;(_kEmbq4d;gxm#8#Bx27KQE6%g);7S_i zpK=-YD?T;N;-Z)qyZET=axpOFILQBy@Q4dK7+o0*AQP?A6nVUu|CHzqmem<@H!!Ch zG&?3|dc+og+?o0x6)|Imv@v3P;l$POK`26 zE88VT+x(~)&Mkk!L>ts6jR=v@{40~D>3T1WQUg_EpW~oxkpGe#*lcZvj^VtBoe9*q2~t!o@9oUdMWj>na4OMkh%^ z$FW=2qK7pLVU8JKDM}hkb1GY-y>jm@^3fLJ=y9*SDw9t1g>_C(Iww}o`bg1X{&{iy z1Sy|h)kwIszN=%?-6&Wli#23iG-AtbI#*mz`(}ctTyN3=UVg~p4CyX~9*c_Z0-6!= z7EKF6e|WIgblG?$E<-UYVCJ>n&xhJz0C*{N4yU`|*_+iiIfs*FIlz5A)E1jKEE8nF6b5&Rkcpee? z_hO@b_C66?-4P#oUfrg;-DNC@g4n9rNzdMbCDeD#v9Z?ee|p-NM1>p*AX!T!rhRCB zCDhh#7j=^_HwqD{h_80t`T7n}guzmB`Ln>BIH8{b2jd&yV3H84I7@2|xdRqS;*G@q z2R1dlqj_DuzFY%G^cSe*qRg+@jE}U^o)M%?+zR18(JHvHTGBE%)+JE5xH4Y2tzBYB z&$J?MIE|_hzjxvO9%BSZTTnE^j%LUP2F>bQ2-5|asIW=LhMlju&z%7=@gT@FyKVJ# zqm|rXxXsrrvnsnAp<(T7y|CjZynAqN0V&PG`QC+>md8^!EixW%2;Ako-hpnUK z@Q$rxfbawqG~&uMQfz#9HuDgAVl2Jkl*o)oU{b9UB3H)U&J!3~qr4JxX35S%k3T`>ca^<>fC6w0{^P6wov5>6OEeut#} zX7)f(H>CvyA&P74mMw|tgO_>J9?hS$BnbG|rA#sa(C-ov zGkXU|0Qk@E|My%aiz4u6=vL?^6a*{>q}G30=6Cadng;Z@VL;6P7};L~d9l|2CP-iA z`bSL=fSsKK{BMluu)2q<(p25bU1`ns1IHN$OG=6qvv@MHI8{H8l+V>4S_n-p&j)Wz zEFQ1p)vEyy0IYlm9VryZM2ZvrOWw$4dod_!^Xo8vnU34iIjsl%OZV9ELGoMFC<9XXf~wl42M=e;MYSB zdb{?XPVKSkps%%G?5lLdm^UR<7zagbFq}wGJ!0p$1?;N9^u~1@vV89hwk^1lvgy(r ztb}2)#GiWiO9!FlQFcS6Y;*7FKAS2?jv3Z1E#D5~eGJ7{s&dWtI1bTHio5_|7@q&x z`|-yNy}U~H`4s}rY0JFYPcpK>4g;Cgfs$H{zK~(y0XF*|-*jV=z8%R5%J}Ab zxFg@pffELz4HS(wyZ#dEDHSe6p-9QTb6{tZrW>jWZ#tx>*r)e;fvkMP4)NEnugP*3 z`I0KpFNG*aJ#pe+csw~YyBMESONT6bH`jGil+=J``g*H*Pp+qH+aa&Y^ym-=1`rXc zW$v9#>0xWkv0IotkxA=6HHt@UxDW*ET05U|Va(WCjSOIw>E`OaM}nl4LLZ4bh`3C5 zt9i9KHyL$d&3{;*5HA!>QzM{+7-QRs6>DE4NhyJmlK4xgtZ;DJcn|UYMTYSxe^hoy zOuz>A9CH_P+N3fT5vaOPAm(rlmBjO0olEBskLPASWKm|RdiXqBx+&QW=wCI z9!y9)T_5~@F|0Cab5MH=XnEhVY>t>Z+mx)3+X`Kc++8?{V$_Nqe(BtFiR8JuXUY`z z6OPFmS!Kv$4_IU}+C4`R31PJ+ADS|q?zOTK@rxX!7MLwZO6f=phD)e`Z&uwAxwyIz z-b~WLx%E%7qGY4+i06Ct))or>TiYvnTLf#;{4qN85Q+3x7)({n^7i7^w8CAvoJRIO zupH9e*n201)9dZAwZ!Iw3Y=J0Nz$V9`KEHmD&fP)Eml_xxUx=JH{l$FqdI-teL73P zbs>Sv9E4%@`}=B*!7LePvM5;Q_(e+WTjp?LvSic68r7NozS)GS`04beokAhO?RVjZ z8z!lp+H?r`DeRL8Y2@igBB&7}^6YFL;vpR_QHxhoUj$ZEJx%rNF@y~S{TGdz_5z(Z zPfxP7YQy$ES+}G)Su}TBt2TV4%O9NFWTmsjq5D>7?pR1jIgF%5k?ZPOpVMEuM&S{l zV`(#6Z*HWxO?ordcsZZj`pYBA-lJi$NnK6u=6n%#bOPV|1Ex;B8}56%lNEisFVZ~Y zkGUoV^9CD?uOc`t6JOES%llvJM$wgKO9_m!QDbrdYvN>Ex`XRe?mL zXQn*+Q~qMfVR@7nA6ptk+dk1p9qc)X-rZrLH%gL?VinAMp`KU`ou5#u%hwYxF559z ze!VMWBxg1x4^}y!jTP!8P2p8*@=_H1x^_Cs)O2>SqV|Rqly| zw2Ddxo+82Xx@GNiw(O&;LIK0R2t*~PLTVTReoy`!AJG=ex7q=rVH44Q=<9_!)a9;? zM00lJI^h1W;y(SB!l%Wx@>2;9!KK_gk#{ZGcWD(m^T*fl8stYS3yl_6hguOHt3pr1 znIGdmKLn)?=s2>gGGvi6eaIrmQF8h*5J9c;w9|?Dc&>qqUps|&YA*C+Fpe;9F3(Iw zmmKDsehM)^bQOxaI9cEr?u}bp%~0ofC}0947UzLRC`7$#u5o!_VK&#`ePnX_<%h!| z;YFVbwfFL*!{G=nuAUwK5p57r3xb9zpCp8B|;2SJ+ik+u2a zLdduj`qPY#OYU{TagWy5#=SaT&*N3m3SO-R`8qzYhYpyVIvC~Z3Y8Eha{1|_Gt#cq zDTRWC0#M55MI2AWSlRjWgm(yELCyZZOeNLeQxLz4IgGl6GludO^25@KK!}Op=dQtM zvkth1xBv$3zbz*n+a-POD&Y!IO-byk4=(}m*(DhcwZFY$WAIk+%Gw#9&y8{!gF#x` zA^lkBt*|d-IcITXOIe3|Iv*I^+8j4&-&IlvN)T@@C_29_iE+fkf*;fez5wq^tXVm5?(>zepUth2FzJ z_FM|1LvQc|ShyhM_!ttBXrXE0d}m4xuBMX%N^RQXB~m2C=8nmk(mUc7q_d^7WjJTb z{0iNr+gT9}Y&H7;i_W8VpUX|c`-9ZsAYqVgafxvcwj~{r3RTr=-{*A7LiR}oCi%qC==S2T#=HHTOEz?H zle9sjk$76~b?d3IAKoVa6n$_PuiBu5TO8J1Hfdg&F17KO$fR7^pp>bgnhB!Fa621< z0H4#%32n?)D^apYZXrR*w^;x)s1y;FNp3v__+gJsT)w%Bmd!ejKHOc^l`n1PucML( z+5(bxDm#Mjjn)flGQDo+dYb9%U^cg@2lFatida#F60o1H3Uivrc?nf4pTyMUKZ{@= zES9WE)R-gpUpHc$4I0qQ5T0g$Ay+-cM2h;86Gv{K^F`uL+;eH8+rRFR=Y3`k`VbX- zpe?i}XJp-GI&i$Vc$3rmqZlb9F39#q#a_BRmHM3GpriaaPU)3EdMr&=t5l{|IX1nH zu5!ua=Oj1!=%C%00l{C{d^F_o_yegn$u{XWsWuszJiI_&121DY6+w5wEYe)k8qyNh z@2q2~7rfa<>ZItw4(2*KT~(dmB$^%(Ne!pL{tFT z13qP=<4izllv1r*pH!pgO@dR0hE;p63y28VnK&wq?gCfd^KLfuVl`XbymTGJU z=h$(aqHf#|QbXP1wowLAS2VwWAy2=X%TcZ|mcHe;_R^cI_T)v18$8)thg@mqA8PNLvjY&SSD#3=gE0y7?lPhq+E=wIKoS?&TPP|kn-3Vy zh2u1@yH?<2;xw((2#1yN9@DSeK{0ol{gJTi6rU^AX(q`Tsac7ReP)g}>_{nnG-5 zinXau023_XnzCqw>9d?Ib-Zg%F39cWI%QvESZdhv z;6vvJgV$vpXH$#2@VrU(^h&3GNsh_#%hRAb#%u$iJ*r8qV6gYx>- zXYOCQu_+(<*#~6!39#zlo(zs$Ur{_Xc%Ct%M@l(hi8bScXqCJ3l<1*P4#+d`=*dph zKT*su2T0W|3RzE#)WnBP2(r8%qwtohSfCBl9QCRf)SeLXJ3YG^yV`3ys4c8l(WZ+E?=~&ECwsKYZtXg^ z_<7g_RYjIi8eLt-sL$TzM3G=~S&AlQ_oaL|hkf$nr(NV2bQC&~wO|sa+xPPRq1Kef zG^$lJ%lzZMK-wz0sz%O(VjZ&k;(~j%*wkBL%A&l*Yj80zUWkuZx%}3zNMs!1T}SA` z&OQOoa5m>D_jlY(F$#=P9@Zi~vNNg7D*da=`A6z09h!PV5326CZ<#~NN7Rf%ekPJ` z>KZ`=Vp-jMYchcd0S&^YV5JUYlLa)MeP>evvY@u@t5k=e!ZuuKvXv{tcIa1LPec@U zCuh{Hd#5K1C1v@UAGhlSBLcGwpOOd97Lo4@&>=nO^N3k|=FZt(rGDKZJ~TS?<8uaJ z;1Ri!k)KUaZn85?$e4^v5LOW2x%_nP-#j_;if?>`^z^A852~s~jZdojCZY4NobW^v z1*!A;)P~Zz12bkvjyRo5ASA^!gxJQEu_O3(zM;b1=dKgkyepl{Y}jfH${F$Hwo{zx za&l1hCrio$wI_c|xjUVoP7>`ISINv*3=`rGw!~ksG{mr&?kEox{S7jI-kV!C#4Svn zAlXVVptjLkGpy=TO&b@}&5Kiquexz0e4J9YiMHF|v{x&tQOv{M9ws}|O$q)gOHSSA zaZYR>+#wY;p#t^U51P^B}5Adl?N!=Ts)aL~vn ziNb=-K2lLsa>L#}9E^3kD*qn}1*b_HzUyp1Z=mmE6&D#kppfJQoe*w@D>gpqF{At{ z6OpLsPEyygXS~#o3?E@cxBSOq;oeWVTt$oODeR^cyiW3dB6B+C3V5ab|v@4Fu zWsU=hfg*v>0-um%$|S^iX;8m$F_H7$x28h!6Np8pAe`OuZRbm}47!n|CFvF+!0%#w z1nxzxuwM7xAD^de`gU2&#J}Emmo~z`9A@k-p6IaNZ!MtLc^LM!ZE$`Z4V4eVQR{tV z7oRC|J{u2MNksN~mcguJfAwxIwrqs#1Jb|Dm~U zMeyCCYQGb-3C)wLD^v)6*illS7XQxB3TpH`8%1jS~c7gS%Jl+|urf&K2cuB5NDux+bRWbjrS(K9@z`!m8iYGii8JD3bzMFPJV zxk>~YFcV5k>4`G&=_3-xtoCzVry$F!{rnOnz5hjq!}3$6&}y&pvFjVc&<=wrm8xDt z*IGnkv${&rPh%}3ZS#J%wp{P%+XP|8>KP*H`j3A?LwSs$qQcwiwgwQvcX*{YtU%pS zdXX?LZu@(tU8^9~O58v2n>Mj`Xc*3Z?QOq=)f=^DSy=aZ|5+*a{QmF`*|GuO*+qd& zdh>{-%~R{+SvA-6M?&W%E5omrm6B^|mDAuiZ(N8?BAE9V<=dT!gFdHS6D{~zio)zi zUmu9gTmmc?%ncSMbTzm08Gm84PQhv(?=UFa+PpfnYu{37T!2q2E}o^r-%WxqU50j< zQlHh5FRW?nd96}E4oOgoW)pD1re@+(X8~n-0HapJl~9JzwyUHT5T=sfE`jew(;2#H zTrUu~X{^S0GPwJ5BtL(DL_9om_=kwH}enrVqMnW zY-7yULasg>_7B!i`#i3x1lY0Aj&;)z_RT#WI5V?~3mX0v-;nh_b+~L*jVl3g&C%apQ-OR{L_LZwLG@)u=666uQfZ5(=O$n#sJ2y4B&w$1QG~wQ$57G8ql;7)ai#YzaC?I;NSZf_NJ%@2jdpJR|VjKMl6@K_E zP$Jj7-NymhDXRkN^K_4+%-B3%XO5I29P&&R&2b}sCTb(C2?Q5GFmi32ij`T8+fy~_ zjS1%QcT|ytRtT@{Ad~;XjsC?CKXaR`%pCt>NB@g@6!iaGKB^TvY~2Y!?mc=J5Ve#& zAc||6Pc!jD%YHT}@cpmjn_yep5Buq%dEg=XmcQ}y z&LQo$j1WmKO*N6EkHhu}Px`cj3Os%-S@4n@(z>psHu45ok#s)3<6#eyD{yM3{MG$) z*FB=80T;_DgIzAlfP5^~+ry}zw6n0)gUP%o*CIm{^QP-#XelwKDCR}zMd*3#MeO;f zi%)ndF=`;XwHQtuMb#nfssT<}nV}o(&7{DG*Pu+wNp-j_=#!Wmv8vLNJcr>}bCr*{ z35Okzq4)WQB;s5%B(GU_r_s&vhc9`MJTG`6N#FbY@ZP;QJb4sZT%&pG*?J(Dz3^X_ z>K_b03p4mHf*r*64<7puhW&-H7cz1%us5@Hw6Xt#Lx0Ya(X)D{-}%KA#YI%5=`2tiXl`@bw8rT1qV#QbNt;@>O|_`N27 z#1Vw8pS9Iy)+PWZRWoaTYX`HxGJbchYG&wY>hNsHmk7xJcQ01pOW&-VY>XgI5E#e? zU;_ggK`d>n&3PDT!9F!OU`FWvkj$qD9Qd@1;S>EHF)fXvT50@*o0|K8x=LjtpNFanv`S-@Zb zD~N>=#KH11IxrB#$id1Ae(v!fo&TN~8w(>R8<^vzJ{y>k6a0IQzy0*tMQoI<&0h5H z0MN^x_fJm({nM5H3K9Js=K8N1`hSLqn8BQE{~jVzGP6{g!EUi>@#tBfyU*Chw`{w_ z^Drm0>BNJF70igB4MVU4<`)>Xwdryja3_1(QKiuPGZ-IL`sc~;nOnSJHdE$muVpPl zh#-@+f`)q5qhSgR*U9LSVNS9cne%J<=Xvj!{O8*XXmoTc?QkG(5m*?h3{U{R|Qtn zt<2WwpAwP6Oj!-86J#7*%=E)o0GD~F9tr-x!e<%@tt9t!jvg2 zF8EYkS*KNjuT~ywmo><+K$%VUrOkcVY2k?5Y70Bz2aF5ORg97~9o6r9ohHa91JaE1 zuydvQ=^aW&^eHu&2v`O(zj$do&;cDQjCT^<2fyVjtIA>}u)09Ih{6qh-Cg1{cB*?y zPt{3TNkXM81b5Zous?}yyvN%pcBAi6IyfnkOn@qTTcT^9pu(}7T7qTxfGJa+`pztL zb4KFm>lJU94);Z|g5@%Y$CJy6>;09Ar=d89RF9;MKiA>s*FjZTGEYShl5JkvH|8SxJ}wUhx2>@@gyA!0O?geKyl3jJW56N9ZH{S~W78>2;P zmHc~hJeht3WX*bbq&MmoQby+?uL>kXR)b{U?iqjt({UffKL-kjuj7PR&^rzB=zKt9 zk5B40+bj+a)~@2O*4%jO+&u8zzx!=-YAkQCQPF_7W^(zd_(ceC)XjHJmbaFUa)7Kh zE1~r>aYh!2!(6kVzY&t8oy4x<+57Z%{a>rMQCrxZiAISS>ZRunpA94QwvKRF8dp>f z9T|BO4WF*1&fEKmw#I}(yjCq!MQW2|`XCPvx%~9Xtxz2Fc70>DzIyE{xuzavf5s7Y z@s6Be=888B*INN=J(I)BnG6gxOn2-%%9G#h#_?RN&ZvmrnOY_qn3<*bbg$NQ=T{t0 zK5g9Bek;wcnj?5Jb$%e9o@2PMv@kbj4VfLfoV(b(_oWy$cU$W;s(3UO78%wN@;ZHs zXy>qjJN4x1ZTIj9zGyx$K9xD8ka+9kJ=r|;fwXp!seENmrGmHm$wHmt2#-O0`OA-D z*+;l{r6%II`D<#S(%8K><%qxJ9*5MZ=;hb9j4(;uZOsaU83SzlqXmVpi;ze@&=;Tz zG^&NS1H3TWVc=j1Ii6Z%wxL~hM9F+eQaMc4;HX+`);y?|!$z~OM9?z*T%^+@eiC7> zIIT;y#*OUkxU6wQe&qJI*<5vm__B;LPK#+1bgVKYZF4i$FeT&3^RWifrr0$XirQs! zTpY6^Q|{Nrhjcl+`{cC9JKON}$R@j?Xi?6S!pK~CdJ3<$7DzVk?`#9bDy+Fx9a0$j{z+w)!gk3&hAqBadD^rD9T}Z%bmUhG)$HZ6g`%@ zgLrR{^PNHL0NONapk9Tf4rbiAPlaUKb}SX$UqJ11fHIXzs;Rm>KLeGDl5SH-R*2E- zYxh?=ELF9{!&EKi`uz|s`Z=`z!dw|BStmuv=HWk!km*ZVcPra`Z%Bs-Y;xT|L$7sF zlv3B*lUP6t`d{Wa3o|#fZ&8Md70nXioa#`D73D4R6w9DRxWYGh*!rpUlW7Odw&IfI zb0@=Le@HbbMnt|5lxZ-~aH5h%%5g6inMz46UbVw5NKSqy{-vLaO4-dcHnP~~bMZTP zDXA|$M7}}WEFhxC0n5{w+-kD0w>9#dG<2u?N-+VUh4iwh@@~jR`CJLCS@A(MQ}X70 zy^R;W0yHTtRRnx$XN!&3%FM~=0P)7A!dPH4Nh1jPke3{m{ce%!gVfJ~z`2gexZyMCF>`eLyGYT=*4EO<>IDk@#yQUef4Sn{G2An}`ui?W z1!w}F5d$0W`CQJ)&c^zDT4!Tp|r#(&_e|Kp7R zLM{NNzb3&T_Pen^pq!!JGbDSq;R~GkZF7Y`kl;TK*e|#Dza6mu(^X~vLmkHaFWN!o zZ$;RPI_%lye_QJR@KaXhm~nGIAXea!XXumh5ObmUk9xX^Fbx_54J}X}iol^IZ;A={edTY35?ErPeOZ-HE1Z`<8V0#-cv7~YIkL1x|AM35( zUvL|qwT-8?2+I4D{V?d`qjsb+H`;4PE3n4&@@@E@Q5@Tha{E>O)^QO|Y{ECQmgT$h zJn3haQE=h?&fyNh&DxPE^a!79!RMs2EZk)c>-Z{DWR#X6O8SH~CL`K~2Hv z56<8(NCf=fI=z<#QZ#cn0x+}xUaV*51p$}=%zwCsf{o4dGX4R(UTVJFO3%ysmzVs# zu^Z^>UO#1q_@?rEfLWn8i6HtQcT6(4b3lf0= z6KxfmqV~uy*oIGl7fmd=3<^ES$NvTnpa;p^tw5-cP9CglTw4VlI_J*}gn$@ENBJ#T e{C7>IgQK3kqpSV%bV00OFe@S@rLe3B;{OHaRRp^L literal 0 HcmV?d00001 diff --git a/grammar/railroad.svg b/grammar/railroad.svg new file mode 100644 index 0000000..21e6c76 --- /dev/null +++ b/grammar/railroad.svg @@ -0,0 +1,1238 @@ + + + + +bit_datatype + + + + + + + + +'BOOL' + + + + + + + + +'BYTE' + + + + + +'WORD' + + + + + +'DWORD' + + + + + + + + + +integer_datatype + + + + + + +/U?[DS]?INT/ + + + + + + + + +float_datatype + + + + + + +/L?REAL/ + + + + + + + + +literal + + + + + + + + + +integer + + + + + + + + + + +floating_pointer_number + + + + + + + +time_literal + + + + + + + +character_string + + + + + + + + + + +integer + + + + + + + + + + +_datatype + + + + + + + + + + + + + +'+' + + + + + + +'-' + + + + + + + +decimal_digit_string + + + + + + + + + + + + +binary_digit_string + + + + + + + +octal_digit_string + + + + + + + +hex_digit_string + + + + + + + + + + + + +decimal_digit_string + + + + + + +/[0-9]+(_[0-9]+)*/ + + + + + + + + +binary_digit_string + + + + + + + +'2#' + + + + +/[01]+(_[01]+)*/ + + + + + + + + + + +octal_digit_string + + + + + + + +'8#' + + + + +/[0-7]+(_[0-7]+)*/ + + + + + + + + + + +hex_digit_string + + + + + + + +'16#' + + + + +/[0-9A-F]+(_[0-9A-F]+)*/ + + + + + + + + + + +floating_pointer_number + + + + + + + + + + + +float_datatype + + + + + +'#' + + + + + + + + + + + +'+' + + + + + + +'-' + + + + + + + +decimal_digit_string + + + + + + +'.' + + + + + +decimal_digit_string + + + + + + + + +_exponent + + + + + + + + + + + + + + +time_literal + + + + + + + + + +date + + + + + + + + + + +time_of_day + + + + + + + +date_and_time + + + + + + + +duration + + + + + + + + + + +date + + + + + + + +/D(ATE)?#/ + + + + + +date_information + + + + + + + + + + + +duration + + + + + + + +/T(IME)?#/ + + + + + + + +decimal_representation + + + + + + + + + +sequence_representation + + + + + + + +'_' + + + + + + + + +decimal_representation + + + + + + + + + + + + + + + + +time_of_day + + + + + + + +/(TIME_OF_DAY|TOD)#/ + + + + + +time_of_day_information + + + + + + + + + + + +date_and_time + + + + + + + +/(DATE_AND_TIME|DT)#/ + + + + + +date_information + + + + + +'-' + + + + + +time_of_day_information + + + + + + + + + + + + + +date_information + + + + + + +/[0-9][0-9][0-9][0-9]-[0-9][0-9]-[0-9][0-9]/ + + + + + + + + +time_of_day_information + + + + + + +/([0-9]|1[0-9]|2[0-3]):[0-5]?[0-9]:[0-5]?[0-9](\.[0-9][0-9]?[0-9]?)?/ + + + + + + + + +sequence_representation + + + + + + + + + + +_days + + + + + + + + +_hours + + + + + + + + + +_minutes + + + + + + + + + +_seconds + + + + + + + + + +_milliseconds + + + + + + + + + + + + + + + + + + +_hours + + + + + + + + +_minutes + + + + + + + + + +_seconds + + + + + + + + + +_milliseconds + + + + + + + + + + + + + +_minutes + + + + + + + + +_seconds + + + + + + + + + +_milliseconds + + + + + + + + + + + + +_seconds + + + + + + + + +_milliseconds + + + + + + + + + + +_milliseconds + + + + + + + + + + +_days + + + + + + + + +decimal_digit_string + + + + + +'d' + + + + + + +'_' + + + + + + + + + + + + +_hours + + + + + + + +/[01]?[0-9]|2[0-3]/ + + + + +'h' + + + + + + +'_' + + + + + + + + + + + + +_minutes + + + + + + + +/[0-5]?[0-9]/ + + + + +'m' + + + + + + +'_' + + + + + + + + + + + + +_seconds + + + + + + + +/[0-5]?[0-9]/ + + + + +'s' + + + + + + +'_' + + + + + + + + + + + + +_milliseconds + + + + + + + +/[0-9][0-9]?[0-9]?/ + + + + +'ms' + + + + + + + + + + +decimal_representation + + + + + + + + +decimal_digit_string + + + + + +'.' + + + + + + + +decimal_representation + + + + + + + + +'d' + + + + + + + + + +'h' + + + + + +'m' + + + + + +'s' + + + + + +'ms' + + + + + + + + + + + + + +character_string + + + + + + + + + +'STRING#' + + + + + +'\'' + + + + + + + + + + + +characters + + + + + + + +'\'' + + + + + + + + + + + + +characters + + + + + + + + +/$[0-9A-F][0-9A-F]/ + + + + + + + +/[\x20-\x23\x25\x26\x28-\x7E\x80-\xFF]/ + + + + + +/$['$LINnPpRrTt]/ + + + + + + + + + +_exponent + + + + + + + + + +'E' + + + + + + +'e' + + + + + + + + + +'+' + + + + + + +'-' + + + + + + + +decimal_digit_string + + + + + + + + + + + + +_datatype + + + + + + + + + + +bit_datatype + + + + + + + + +integer_datatype + + + + + + +'#' + + + + + + + + + + \ No newline at end of file From 1ba383a1a037843d8ad2f3621f23de97748ce1bf Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Thu, 16 Jul 2026 11:48:59 +0100 Subject: [PATCH 2/7] Add tests --- .gitignore | 2 + Makefile | 41 +++++++ grammar/{grammar.bnf => literals.bnf} | 12 +- tests/literals.txt | 161 ++++++++++++++++++++++++++ 4 files changed, 210 insertions(+), 6 deletions(-) create mode 100644 .gitignore create mode 100644 Makefile rename grammar/{grammar.bnf => literals.bnf} (90%) create mode 100644 tests/literals.txt diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..0187b3d --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +build/ +*~ diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..e38e547 --- /dev/null +++ b/Makefile @@ -0,0 +1,41 @@ +GRAMMAR_NAME := literals +GRAMMAR_SRC := grammar/$(GRAMMAR_NAME).bnf +BUILD_DIR := build/tree-sitter-$(GRAMMAR_NAME) +TESTS_DIR := tests + +TS_BNF_TOOL := ts-bnf-tool +TREE_SITTER := tree-sitter + +.DEFAULT_GOAL := help + +.PHONY: help check grammar test test-update clean + +help: ## Show this help message + @echo "StructuredCheck" + @echo + @echo "Usage: make " + @echo + @awk 'BEGIN {FS = ":.*##"} /^[a-zA-Z_-]+:.*##/ {printf " %-12s %s\n", $$1, $$2}' $(MAKEFILE_LIST) + +check: ## Run ts-bnf-tool static checks on the BNF grammar + $(TS_BNF_TOOL) check $(GRAMMAR_SRC) + +grammar: $(BUILD_DIR)/src/parser.c ## Generate the tree-sitter parser from the BNF grammar + +$(BUILD_DIR)/src/parser.c: $(GRAMMAR_SRC) + rm -rf $(BUILD_DIR) + $(TS_BNF_TOOL) convert --generate --name $(GRAMMAR_NAME) --output-dir $(BUILD_DIR) $(GRAMMAR_SRC) + +test: grammar ## Run the corpus tests in tests/ against the generated parser + mkdir -p $(BUILD_DIR)/test/corpus + cp $(TESTS_DIR)/*.txt $(BUILD_DIR)/test/corpus/ + cd $(BUILD_DIR) && $(TREE_SITTER) test + +test-update: grammar ## Run the tests, update their expected trees, and copy them back to tests/ + mkdir -p $(BUILD_DIR)/test/corpus + cp $(TESTS_DIR)/*.txt $(BUILD_DIR)/test/corpus/ + cd $(BUILD_DIR) && $(TREE_SITTER) test --update + cp $(BUILD_DIR)/test/corpus/*.txt $(TESTS_DIR)/ + +clean: ## Remove generated build artifacts + rm -rf build diff --git a/grammar/grammar.bnf b/grammar/literals.bnf similarity index 90% rename from grammar/grammar.bnf rename to grammar/literals.bnf index 104caed..3e7db9f 100644 --- a/grammar/grammar.bnf +++ b/grammar/literals.bnf @@ -58,10 +58,10 @@ time_literal -> date date -> /D(ATE)?#/ date_information ; -duration -> /T(IME)?#/ (decimal_representation | sequence_representation '_'? decimal_representation?) +duration -> /T(IME)?#/ (decimal_representation | sequence_representation decimal_representation?) ; -time_of_day -> /(TIME_OF_DAY|TOD)#/ time_of_day_information +time_of_day -> /(TIME_OF_DAY|TOD)#/ time_of_day_information ; date_and_time -> /(DATE_AND_TIME|DT)#/ date_information '-' time_of_day_information @@ -95,7 +95,7 @@ _minutes -> /[0-5]?[0-9]/ 'm' '_'? _seconds -> /[0-5]?[0-9]/ 's' '_'? ; -_milliseconds -> /[0-9][0-9]?[0-9]?/ 'ms' +_milliseconds -> /[0-9][0-9]?[0-9]?/ 'ms' '_'? ; # the entry to decimal representation is only possible for time units that are not yet defined @@ -104,9 +104,9 @@ decimal_representation -> decimal_digit_string '.' decimal_representation? ('d'| character_string -> 'STRING#'? "'" characters* "'" ; -characters -> /$[0-9A-F][0-9A-F]/ - | /[\x20-\x23\x25\x26\x28-\x7E\x80-\xFF]/ - | /$['$LINnPpRrTt]/ +characters -> /\$[0-9A-F][0-9A-F]/ + | /[\x20-\x23\x25\x26\x28-\x7E\x80-\xFF]/ + | /\$['$LINnPpRrTt]/ ; _exponent -> ('E'|'e') ('+'|'-')? decimal_digit_string diff --git a/tests/literals.txt b/tests/literals.txt new file mode 100644 index 0000000..0189a73 --- /dev/null +++ b/tests/literals.txt @@ -0,0 +1,161 @@ +================== +Decimal integer +================== + +42 + +--- + +(literal + (integer + (decimal_digit_string))) + +================== +Typed negative decimal integer +================== + +INT#-42 + +--- + +(literal + (integer + (integer_datatype) + (decimal_digit_string))) + +================== +Binary integer with separators +================== + +2#1010_0110 + +--- + +(literal + (integer + (binary_digit_string))) + +================== +Hexadecimal integer with separators +================== + +16#FF_00 + +--- + +(literal + (integer + (hex_digit_string))) + +================== +Floating point number with exponent +================== + +3.14e5 + +--- + +(literal + (floating_pointer_number + (decimal_digit_string) + (decimal_digit_string) + (decimal_digit_string))) + +================== +Typed negative floating point number +================== + +REAL#-1.0E-2 + +--- + +(literal + (floating_pointer_number + (float_datatype) + (decimal_digit_string) + (decimal_digit_string) + (decimal_digit_string))) + +================== +Duration in days +================== + +T#5d + +--- + +(literal + (time_literal + (duration + (sequence_representation + (decimal_digit_string))))) + +================== +Date +================== + +D#2024-01-15 + +--- + +(literal + (time_literal + (date + (date_information)))) + +================== +Time of day +================== + +TOD#12:30:00.5 + +--- + +(literal + (time_literal + (time_of_day + (time_of_day_information)))) + +================== +Date and time +================== + +DT#2024-01-15-12:30:00 + +--- + +(literal + (time_literal + (date_and_time + (date_information) + (time_of_day_information)))) + +================== +Character string with escape +================== + +'hello $N' + +--- + +(literal + (character_string + (characters) + (characters) + (characters) + (characters) + (characters) + (characters) + (characters))) + +================== +Typed character string +================== + +STRING#'a' + +--- + +(literal + (character_string + (characters))) From 279a032673dc9b0293ecb07ab690a93808f40339 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Thu, 16 Jul 2026 14:18:25 +0100 Subject: [PATCH 3/7] Fragments of the grammar --- grammar/control.bnf | 65 +++++++++++++++++++++++++++++++++++ grammar/data_types.bnf | 74 ++++++++++++++++++++++++++++++++++++++++ grammar/decl_blocks.bnf | 24 +++++++++++++ grammar/literals.bnf | 25 +++----------- grammar/pou.bnf | 17 +++++++++ grammar/st.bnf | 16 +++++++++ grammar/stmt_section.bnf | 11 ++++++ 7 files changed, 212 insertions(+), 20 deletions(-) create mode 100644 grammar/control.bnf create mode 100644 grammar/data_types.bnf create mode 100644 grammar/decl_blocks.bnf create mode 100644 grammar/pou.bnf create mode 100644 grammar/st.bnf create mode 100644 grammar/stmt_section.bnf diff --git a/grammar/control.bnf b/grammar/control.bnf new file mode 100644 index 0000000..e3dd18e --- /dev/null +++ b/grammar/control.bnf @@ -0,0 +1,65 @@ + +if_stmt -> 'IF' expression 'THEM' stmt_section _elsif* _else? 'END_IF' ';' + ; + +_elsif -> 'ELSIF' expression 'THEN' stmt_section + ; + +_else -> 'ELSE' stmt_section + ; + +case_stmt -> 'CASE' (variable | expression) 'OF' _case_value+ _else? 'END_CASE' ';' + ; + +_case_value -> value_list ':' stmt_section + ; + +value_list -> _value_list_value (',' _value_list_value)* + ; + +_value_list_value -> constant ('..' constant)? + ; + +rept_and_jump_stmt -> for_stmt + | while_stmt + | repeat_stmt + | exit_stmt + | return_stmt + | waitforcondition_stmt + | goto_stmt + ; + +goto_stmt -> 'GOTO' jump_label ';' + ; + +exit_stmt -> 'EXIT' ';' + ; + +return_stmt -> 'RETURN' ';' + ; + +waitforcondition_stmt -> 'WAITFORCONDITION' expression_identifier _params? _with? 'DO' stmt_section 'END_WAITFORCONDITION' ';' + ; + +repeat_stmt -> 'REPEAT' stmt_section 'UNTIL' expression 'END_REPEAT' ';' + ; + +while_stmt -> 'WHILE' expression 'DO' stmt_section 'END_WHILE' ';' + ; + +for_stmt -> 'FOR' variable_identifier _start_value? 'TO' expression _by_expression? 'DO' stmt_section 'END_FOR' ';' + ; + +_start_value -> ':=' expression + ; + +_by_expression -> 'BY' expression + ; + +_params -> '(' fc_parameter ')' + ; + +_with -> 'WITH' expression + ; + + diff --git a/grammar/data_types.bnf b/grammar/data_types.bnf new file mode 100644 index 0000000..ba4e405 --- /dev/null +++ b/grammar/data_types.bnf @@ -0,0 +1,74 @@ + + +datatype -> elementary_datatype + | udt_identifier + | system_datatype + | to_datatype + ; + +elementary_datatype -> bit_datatype + | numeric_datatype + | time_type + | string_datatype + ; + +bit_datatype -> 'BOOL' + | 'BYTE' + | 'WORD' + | 'DWORD' + ; + +numeric_datatype -> integer_datatype + | float_datatype + ; + +integer_datatype -> /U?[DS]?INT/ + ; + +float_datatype -> /L?REAL/ + ; + +time_type -> /TIME(_OF_DAY)?/ + | /DATE(_AND_TYPE)?/ + | 'TOD' + | 'DT' + ; + + +string_datatype -> 'STRING' ('[' constant_expression ']')? + ; + +udt_identifier -> 'TYPE' _udt_type+ 'END_TYPE' + ; + +_udt_type -> identifier ':' (_dt_with_init | struct_datatype) ';' + ; + +_dt_with_init -> (datatype | array_datatype | enum_datatype) (':=' initialization)? + ; + + +array_datatype -> 'ARRAY' '[' constant_expression '..' constant_expression ']' 'OF' datatype + ; + + +enum_datatype -> '(' identifier (',' identifier)* ')' + ; + +struct_datatype -> 'STRUCT' component_decl+ 'END_STRUCT' + | 'STRUCT' 'OVERLAP'? component_decl_rel_address+ 'END_STRUCT' + ; + +component_decl -> identifier ':' _component_type ';' + ; + +component_decl_rel_address -> identifier 'AT' relative_address ':' _component_type ';' + ; + +_component_type -> (datatype | array_datatype) (':=' initialization)? + ; + +relative_address -> '%B' /[0-9]+/ + ; + + diff --git a/grammar/decl_blocks.bnf b/grammar/decl_blocks.bnf new file mode 100644 index 0000000..b55d842 --- /dev/null +++ b/grammar/decl_blocks.bnf @@ -0,0 +1,24 @@ + +constant -> 'VAR_CONSTANT' constant_decl+ 'END_VAR' + ; + +global_constant -> 'VAR_GLOBAL_CONSTANT' constant_decl+ 'END_VAR' + ; + +global_var -> 'VAR_GLOBAL' (variable_decl | symbolic_pi_access | instance_decl)+ 'END_VAR' + ; + +global_retain -> 'VAR_GLOBAL_RETAIN' variable_decl+ 'END_VAR' + ; + +fc_temp_var -> 'VAR' variable_decl+ 'END_VAR' + ; + +fb_temp_var -> 'VAR_TEMP' variable_decl+ 'END_VAR' + ; + +static_var -> 'VAR' (variable_decl | symbolic_pi_access | instance_decl)+ 'END_VAR' + ; + + + diff --git a/grammar/literals.bnf b/grammar/literals.bnf index 3e7db9f..fae85b4 100644 --- a/grammar/literals.bnf +++ b/grammar/literals.bnf @@ -1,28 +1,13 @@ %axiom literal -## Data types: +%include "data_types.bnf" -bit_datatype -> 'BOOL' - | 'BYTE' - | 'WORD' - | 'DWORD' - ; - -integer_datatype -> /U?[DS]?INT/ - ; - -float_datatype -> /L?REAL/ - ; - -## Literals - -# identifier -> /([A-Za-z]|_[A-Za-z0-9])(_?[A-Za-z0-9])/ -# ; - -#number -> /[0-9]+/ -# ; +identifier -> /([A-Za-z]|_[A-Za-z0-9])(_?[A-Za-z0-9])/ + ; +number -> /[0-9]+/ + ; literal -> integer | floating_pointer_number diff --git a/grammar/pou.bnf b/grammar/pou.bnf new file mode 100644 index 0000000..b6d4cd9 --- /dev/null +++ b/grammar/pou.bnf @@ -0,0 +1,17 @@ + +expression -> 'EXPRESSION' identifier expr_decl_section? stmt_section 'END_EXPRESSION' + ; + + +function -> 'FUNCTION' identifier ':' ('VOID' | datatype) fc_declaration_section? stmt_section 'END_FUNCTION' + ; + + +function_block -> 'FUNCTION_BLOCK' identifier fb_declaration_section? stmt_section 'END_FUNCTION_BLOCK' + ; + + +program -> 'PROGRAM' identifier program_declaration_section? stmt_section 'END_PROGRAM' + ; + + diff --git a/grammar/st.bnf b/grammar/st.bnf new file mode 100644 index 0000000..36994b2 --- /dev/null +++ b/grammar/st.bnf @@ -0,0 +1,16 @@ + +%extras line_comment, block_comment, /\s/ + +line_comment -> /\/\/.*/ + ; + +# FIXME: Need to confirm greedyness +block_comment -> /\(\*/ characters* /\*\)/ + ; + +%include 'control.bnf' +%include 'literals.bnf' +%include 'pou.bnf' +%include 'decl_blocks.bnf' +%include 'data_types.bnf' +%include 'stmt_section.bnf' diff --git a/grammar/stmt_section.bnf b/grammar/stmt_section.bnf new file mode 100644 index 0000000..8d3212f --- /dev/null +++ b/grammar/stmt_section.bnf @@ -0,0 +1,11 @@ + +stmt_section -> ((identifier ':')? statement? ';')+ + ; + + +statement -> value_assignments + | subroutine_execution + | control_stmt + ; + + From 0ad9ec6d14c95a50ce9485e6aa1bf12f214e5848 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Fri, 17 Jul 2026 16:34:54 +0100 Subject: [PATCH 4/7] Full gramar from IEC --- Makefile | 2 +- grammar/configuration.bnf | 108 ++++++++++++++++++ grammar/control.bnf | 65 ----------- grammar/data_types.bnf | 230 ++++++++++++++++++++++++++++++-------- grammar/decl_blocks.bnf | 24 ---- grammar/literals.bnf | 203 ++++++++++++++++++++++----------- grammar/pou.bnf | 109 ++++++++++++++++-- grammar/sfc.bnf | 66 +++++++++++ grammar/st.bnf | 157 ++++++++++++++++++++++++-- grammar/stmt_section.bnf | 11 -- grammar/variables.bnf | 217 +++++++++++++++++++++++++++++++++++ 11 files changed, 956 insertions(+), 236 deletions(-) create mode 100644 grammar/configuration.bnf delete mode 100644 grammar/control.bnf delete mode 100644 grammar/decl_blocks.bnf create mode 100644 grammar/sfc.bnf delete mode 100644 grammar/stmt_section.bnf create mode 100644 grammar/variables.bnf diff --git a/Makefile b/Makefile index e38e547..496d719 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -GRAMMAR_NAME := literals +GRAMMAR_NAME := st GRAMMAR_SRC := grammar/$(GRAMMAR_NAME).bnf BUILD_DIR := build/tree-sitter-$(GRAMMAR_NAME) TESTS_DIR := tests diff --git a/grammar/configuration.bnf b/grammar/configuration.bnf new file mode 100644 index 0000000..2c6ad5a --- /dev/null +++ b/grammar/configuration.bnf @@ -0,0 +1,108 @@ + +# B.1.7 Configuration elements + +configuration_name -> identifier + ; + +resource_type_name -> identifier + ; + +configuration_declaration -> 'CONFIGURATION' configuration_name global_var_declarations? + _resources access_declarations? instance_specific_initializations? 'END_CONFIGURATION' + ; + +_resources -> single_resource_declaration + | resource_declaration+ + ; + +resource_declaration -> 'RESOURCE' resource_name 'ON' resource_type_name global_var_declarations? single_resource_declaration 'END_RESOURCE' + ; + +single_resource_declaration -> (task_configuration ';')* (program_configuration ';')+ + ; + +resource_name -> identifier + ; + +access_declarations -> 'VAR_ACCESS' (access_declaration ';')+ 'ENV_VAR' + ; + + +access_declaration -> access_name ':' access_path ':' non_generic_type_name direction? + ; + +access_path -> (resource_name '.')? direct_variable + | (resource_name '.')? (program_name '.')? (fb_name '.')* symbolic_variable + ; + +global_var_reference -> (resource_name '.')? global_var_name ('.' structure_element_name)? + ; + + +access_name -> identifier + ; + +program_output_reference -> program_name '.' symbolic_variable + ; + +program_name -> identifier + ; + +direction -> 'READ_WRITE' | 'READ_ONLY' + ; + +task_configuration -> 'TASK' task_name task_initialization + ; + +task_name -> identifier + ; + +task_initialization -> '(' ('SINGLE' ':=' data_source ',')? ('INTERVAL' ':=' data_source ',')? 'PRIORITY' ':=' integer ')' + ; + +data_source -> constant + | global_var_reference + | program_output_reference + | direct_variable + ; + +program_configuration -> 'PROGRAM' /(NON_)?RETAIN/? program_name ('WITH' task_name)? ':' program_type_name _prog_conf_elements? + ; + + +_prog_conf_elements -> '(' prog_conf_elements ')' + ; + +prog_conf_elements -> prog_conf_element (',' prog_conf_element)* + ; + +prog_conf_element -> fb_task | prog_cnxn + ; + +fb_task -> fb_name 'WITH' task_name + ; + +prog_cnxn -> symbolic_variable ':=' prog_data_source + | symbolic_variable '=>' data_sink + ; + +prog_data_source -> constant + | enumerated_value + | global_var_reference + | direct_variable + ; + +data_sink -> global_var_reference + | direct_variable + ; + +instance_specific_initializations -> 'VAR_CONFIG' (instance_specific_init ';')+ 'ENV_VAR' + ; + +instance_specific_init -> resource_name '.' program_name '.' (fb_name '.')* _instance_specific_init; + +_instance_specific_init -> variable_name location? ':' located_var_spec_init + | fb_name ':' function_block_type_name ':=' structure_initialization + ; + +# This syntax does not reflect the fact that location assignments are only allowed for references to variables which are marked by the asterisk notation at type declaration level. diff --git a/grammar/control.bnf b/grammar/control.bnf deleted file mode 100644 index e3dd18e..0000000 --- a/grammar/control.bnf +++ /dev/null @@ -1,65 +0,0 @@ - -if_stmt -> 'IF' expression 'THEM' stmt_section _elsif* _else? 'END_IF' ';' - ; - -_elsif -> 'ELSIF' expression 'THEN' stmt_section - ; - -_else -> 'ELSE' stmt_section - ; - -case_stmt -> 'CASE' (variable | expression) 'OF' _case_value+ _else? 'END_CASE' ';' - ; - -_case_value -> value_list ':' stmt_section - ; - -value_list -> _value_list_value (',' _value_list_value)* - ; - -_value_list_value -> constant ('..' constant)? - ; - -rept_and_jump_stmt -> for_stmt - | while_stmt - | repeat_stmt - | exit_stmt - | return_stmt - | waitforcondition_stmt - | goto_stmt - ; - -goto_stmt -> 'GOTO' jump_label ';' - ; - -exit_stmt -> 'EXIT' ';' - ; - -return_stmt -> 'RETURN' ';' - ; - -waitforcondition_stmt -> 'WAITFORCONDITION' expression_identifier _params? _with? 'DO' stmt_section 'END_WAITFORCONDITION' ';' - ; - -repeat_stmt -> 'REPEAT' stmt_section 'UNTIL' expression 'END_REPEAT' ';' - ; - -while_stmt -> 'WHILE' expression 'DO' stmt_section 'END_WHILE' ';' - ; - -for_stmt -> 'FOR' variable_identifier _start_value? 'TO' expression _by_expression? 'DO' stmt_section 'END_FOR' ';' - ; - -_start_value -> ':=' expression - ; - -_by_expression -> 'BY' expression - ; - -_params -> '(' fc_parameter ')' - ; - -_with -> 'WITH' expression - ; - - diff --git a/grammar/data_types.bnf b/grammar/data_types.bnf index ba4e405..83d8d9c 100644 --- a/grammar/data_types.bnf +++ b/grammar/data_types.bnf @@ -1,74 +1,208 @@ +# B.1.3 -datatype -> elementary_datatype - | udt_identifier - | system_datatype - | to_datatype - ; +# FIXME: Not used, only in library_element_name (not sure what that is) +# data_type_name -> non_generic_type_name +# | generic_type_name + #; -elementary_datatype -> bit_datatype - | numeric_datatype - | time_type - | string_datatype - ; +non_generic_type_name -> elementary_type_name + | derived_type_name + ; -bit_datatype -> 'BOOL' - | 'BYTE' - | 'WORD' - | 'DWORD' - ; +# B.1.3.1 Elementary data types -numeric_datatype -> integer_datatype - | float_datatype - ; +elementary_type_name -> numeric_type_name + | date_type_name + | bit_string_type_name + | /W?STRING/ + | 'TIME' + ; -integer_datatype -> /U?[DS]?INT/ - ; +numeric_type_name -> integer_type_name + | real_type_name + ; + +integer_type_name -> signed_integer_type_name + | unsigned_integer_type_name + ; + +signed_integer_type_name -> /[SDL]?INT/ + ; + + +unsigned_integer_type_name -> /U[SDL]?INT/ + ; + +real_type_name -> /L?REAL/ + ; -float_datatype -> /L?REAL/ +date_type_name -> 'DATE' + | 'TIME_OF_DAY' + | 'TOD' + | 'DATE_AND_TIME' + | 'DT' ; -time_type -> /TIME(_OF_DAY)?/ - | /DATE(_AND_TYPE)?/ - | 'TOD' - | 'DT' - ; +bit_string_type_name -> 'BOOL' + | 'BYTE' + | /[LD]?WORD/ + ; + + +# B.1.3.2 Generic data types + +# FIXME: used by data_type_name that is used by library_element_name, that I am not sure what it is about +# generic_type_name -> 'ANY' + #| /ANY_(DERIVED|ELEMENTARY|MAGNITUDE|NUM|REAL|INT|BIT|STRING|DATE)/ + #; + + +# B.1.3.3 Derived data types +derived_type_name -> single_element_type_name + | array_type_name + | structure_type_name + | string_type_name + ; -string_datatype -> 'STRING' ('[' constant_expression ']')? +single_element_type_name -> simple_type_name + | subrange_type_name + | enumerated_type_name + ; + +#FIXME: later we need to merge all these + +simple_type_name -> identifier + ; + +subrange_type_name -> identifier + ; + +enumerated_type_name -> identifier + ; + +array_type_name -> identifier ; -udt_identifier -> 'TYPE' _udt_type+ 'END_TYPE' - ; +structure_type_name -> identifier + ; + -_udt_type -> identifier ':' (_dt_with_init | struct_datatype) ';' - ; +data_type_declaration -> 'TYPE' (type_declaration ';')+ 'END_TYPE' + ; -_dt_with_init -> (datatype | array_datatype | enum_datatype) (':=' initialization)? - ; - +type_declaration -> single_element_type_declaration + | array_type_declaration + | structure_type_declaration + | string_type_declaration + ; -array_datatype -> 'ARRAY' '[' constant_expression '..' constant_expression ']' 'OF' datatype - ; +single_element_type_declaration -> simple_type_declaration + | subrange_type_declaration + | enumerated_type_declaration + ; + +simple_type_declaration -> simple_type_name ':' simple_spec_init + ; + +simple_spec_init -> simple_specification (':=' constant)? + ; - -enum_datatype -> '(' identifier (',' identifier)* ')' - ; +simple_specification -> elementary_type_name + | simple_type_name + ; -struct_datatype -> 'STRUCT' component_decl+ 'END_STRUCT' - | 'STRUCT' 'OVERLAP'? component_decl_rel_address+ 'END_STRUCT' +subrange_type_declaration -> subrange_type_name ':' subrange_spec_init + ; + + +subrange_spec_init -> subrange_specification (':=' signed_integer)? + ; + +subrange_specification -> integer_type_name '(' subrange ')' | subrange_type_name + ; + +subrange -> signed_integer '..' signed_integer + ; + +enumerated_type_declaration -> enumerated_type_name ':' enumerated_spec_init + ; + +enumerated_spec_init -> enumerated_specification (':=' enumerated_value)? + ; + +enumerated_specification -> '(' enumerated_value (',' enumerated_value)* ')' + | enumerated_type_name + ; + +enumerated_value -> (enumerated_type_name '#') identifier + ; + +array_type_declaration -> array_type_name ':' array_spec_init + ; + +array_spec_init -> array_specification (':=' array_initialization)? ; -component_decl -> identifier ':' _component_type ';' - ; +array_specification -> array_type_name + | 'ARRAY' '[' subrange (',' subrange)* ']' 'OF' non_generic_type_name + ; -component_decl_rel_address -> identifier 'AT' relative_address ':' _component_type ';' +array_initialization -> '[' array_initial_elements (',' array_initial_elements)* ']' + ; + +array_initial_elements -> array_initial_element + | integer '(' array_initial_element? ')' + ; + +array_initial_element -> constant + | enumerated_value + | structure_initialization + | array_initialization + ; + +structure_type_declaration -> structure_type_name ':' structure_specification ; -_component_type -> (datatype | array_datatype) (':=' initialization)? - ; +structure_specification -> structure_declaration + | initialized_structure + ; + +initialized_structure -> structure_type_name (':=' structure_initialization)? + ; + +structure_declaration -> 'STRUCT' (structure_element_declaration ';')+ 'END_STRUCT' + ; + +structure_element_declaration -> structure_element_name ':' _structure_element_declaration + ; + +_structure_element_declaration -> simple_spec_init + | subrange_spec_init + | enumerated_spec_init + | array_spec_init + | initialized_structure + ; + +structure_element_name -> identifier + ; + +structure_initialization -> '(' structure_element_initialization (',' structure_element_initialization)* ')' + ; + +structure_element_initialization -> structure_element_name ':=' _structure_element_initialization + ; + +_structure_element_initialization -> constant + | enumerated_value + | array_initialization + | structure_initialization + ; -relative_address -> '%B' /[0-9]+/ +string_type_name -> identifier ; - +string_type_declaration -> string_type_name ':' /W?STRING/ ('[' integer ']')? (':=' character_string)? + ; + diff --git a/grammar/decl_blocks.bnf b/grammar/decl_blocks.bnf deleted file mode 100644 index b55d842..0000000 --- a/grammar/decl_blocks.bnf +++ /dev/null @@ -1,24 +0,0 @@ - -constant -> 'VAR_CONSTANT' constant_decl+ 'END_VAR' - ; - -global_constant -> 'VAR_GLOBAL_CONSTANT' constant_decl+ 'END_VAR' - ; - -global_var -> 'VAR_GLOBAL' (variable_decl | symbolic_pi_access | instance_decl)+ 'END_VAR' - ; - -global_retain -> 'VAR_GLOBAL_RETAIN' variable_decl+ 'END_VAR' - ; - -fc_temp_var -> 'VAR' variable_decl+ 'END_VAR' - ; - -fb_temp_var -> 'VAR_TEMP' variable_decl+ 'END_VAR' - ; - -static_var -> 'VAR' (variable_decl | symbolic_pi_access | instance_decl)+ 'END_VAR' - ; - - - diff --git a/grammar/literals.bnf b/grammar/literals.bnf index fae85b4..468f80a 100644 --- a/grammar/literals.bnf +++ b/grammar/literals.bnf @@ -1,102 +1,175 @@ -%axiom literal +%include 'data_types.bnf' -%include "data_types.bnf" +# B.1.1 Identifiers -identifier -> /([A-Za-z]|_[A-Za-z0-9])(_?[A-Za-z0-9])/ - ; +identifier -> /([A-Za-z]|_[A-Za-z0-9])(_?[A-Za-z0-9])*/ + ; -number -> /[0-9]+/ - ; - -literal -> integer - | floating_pointer_number - | time_literal - | character_string - ; +# B.1.2 Constants + +constant -> numeric_literal + | character_string + | time_literal + | bit_string_literal + | boolean_literal + ; + +# B.1.2.1 Numeric literals + + +numeric_literal -> integer_literal + | real_literal + ; + +integer_literal -> (integer_type_name '#')? _integer_literal + ; -# FIXME: Note that - is only valid for some datatypes. Validate through grammar or semantic? -integer -> _datatype? ( ('+'|'-')? decimal_digit_string | binary_digit_string | octal_digit_string | hex_digit_string) +_integer_literal -> signed_integer + | binary_integer + | octal_integer + | hex_integer + ; + +signed_integer -> ('+'|'-')? integer + ; + +integer -> /[0-9](_?[0-9])*/ ; -decimal_digit_string -> /[0-9]+(_[0-9]+)*/ - ; +binary_integer -> '2#' /[01](_?[01])*/ + ; + +octal_integer -> '8#' /[0-7](_?[0-7])*/ + ; + +hex_integer -> '16#' /[0-9A-F](_?[0-9A-F])*/ + ; -binary_digit_string -> '2#' /[01]+(_[01]+)*/ - ; +# FIXME: If spaces are possible around exponents, for instance, we need to check the regex +real_literal -> (real_type_name '#')? ('+'|'-')? /[0-9](_?[0-9])*\.[0-9](_?[0-9])*([Ee][+-]?[0-9](_?[0-9])*)?/ + ; -octal_digit_string -> '8#' /[0-7]+(_[0-7]+)*/ +bit_string_literal -> ( 'BYTE' | /[DL]?WORD/ )? _unsigned_integer_literal ; -hex_digit_string -> '16#' /[0-9A-F]+(_[0-9A-F]+)*/ +_unsigned_integer_literal -> integer + | binary_integer + | octal_integer + | hex_integer + ; + +boolean_literal -> 'BOOL#'? /[10]/ | 'TRUE' | 'FALSE' + ; + + +# B.1.2.2 Character Strings + +character_string -> single_byte_character_string + | double_byte_character_string ; -floating_pointer_number -> (float_datatype '#')? ('+'|'-')? decimal_digit_string ('.' decimal_digit_string) _exponent - ; +single_byte_character_string -> "'" single_byte_character_representation* "'" + ; + +double_byte_character_string -> "'" double_byte_character_representation* "'" + ; + +single_byte_character_representation -> common_character_representation + | "$'" + | '"' + | /\$[0-9A-F][0-9A-F]/ + ; -time_literal -> date +double_byte_character_representation -> common_character_representation + | "$'" + | '"' + | /\$[0-9A-F][0-9A-F][0-9A-F][0-9A-F]/ + ; + +common_character_representation -> _printable_character + | /\$[$LNPRTlnprt]/ + ; + +# any printable character except '$', '"' or "'" +_printable_character -> /[\x20-\x23\x25\x26\x28-\x7E\x80-\xFF]/ + ; + + +# B.1.2.3 Time Literals + +time_literal -> duration | time_of_day + | date | date_and_time - | duration ; -date -> /D(ATE)?#/ date_information - ; +# B.1.2.3.1 Duration -duration -> /T(IME)?#/ (decimal_representation | sequence_representation decimal_representation?) +duration -> /T(IME)?/ '#' '-'? interval ; -time_of_day -> /(TIME_OF_DAY|TOD)#/ time_of_day_information +interval -> days + | hours + | minutes + | seconds + | milliseconds + ; + +days -> fixed_point 'd' | integer 'd' '_'? hours + ; + +fixed_point -> /[0-9]+(\.[0-9]+)?/ ; -date_and_time -> /(DATE_AND_TIME|DT)#/ date_information '-' time_of_day_information - ; +hours -> fixed_point 'h' | integer 'h' '_'? minutes + ; -# FIXME: We might want to limit 0001 to 2300; 01 to 12 and 01 to 31 -- Or create then rules for that -date_information -> /[0-9][0-9][0-9][0-9]-[0-9][0-9]-[0-9][0-9]/ - ; +minutes -> fixed_point 'm' | integer 'm' '_'? seconds + ; -# FIXME: is the two chars required or can one use only one? 4:3:5.3 is valid!? -time_of_day_information -> /([0-9]|1[0-9]|2[0-3]):[0-5]?[0-9]:[0-5]?[0-9](\.[0-9][0-9]?[0-9]?)?/ - ; +seconds -> fixed_point 's' | integer 's' '_'? milliseconds + ; -# FIXME: at least one is required -sequence_representation -> _days _hours? _minutes? _seconds? _milliseconds? - | _hours _minutes? _seconds? _milliseconds? - | _minutes _seconds? _milliseconds? - | _seconds _milliseconds? - | _milliseconds - ; +milliseconds -> fixed_point 'ms' + ; -_days -> decimal_digit_string 'd' '_'? - ; + +# B.1.2.3.2 Time of day and date -_hours -> /[01]?[0-9]|2[0-3]/ 'h' '_'? - ; +time_of_day -> ('TIME_OF_DAY' | 'TOD') '#' daytime + ; -_minutes -> /[0-5]?[0-9]/ 'm' '_'? - ; +daytime -> day_hour ':' day_minute ':' day_second + ; -_seconds -> /[0-5]?[0-9]/ 's' '_'? +# FIXME: should we make regex to validate ranges? +day_hour -> integer ; -_milliseconds -> /[0-9][0-9]?[0-9]?/ 'ms' '_'? - ; - -# the entry to decimal representation is only possible for time units that are not yet defined -decimal_representation -> decimal_digit_string '.' decimal_representation? ('d'|'h'|'m'|'s'|'ms') - ; +day_minute -> integer + ; -character_string -> 'STRING#'? "'" characters* "'" - ; -characters -> /\$[0-9A-F][0-9A-F]/ - | /[\x20-\x23\x25\x26\x28-\x7E\x80-\xFF]/ - | /\$['$LINnPpRrTt]/ +day_second -> fixed_point ; -_exponent -> ('E'|'e') ('+'|'-')? decimal_digit_string - ; -_datatype -> (bit_datatype | integer_datatype) '#' - ; +date -> /D(ATE)?/ '#' date_literal + ; + +date_literal -> year '-' month '-' day + ; + +year -> integer + ; + +month -> integer + ; + +day -> integer + ; + +date_and_time -> ('DATE_AND_TIME' | 'DT') '#' date_literal '-' daytime + ; + diff --git a/grammar/pou.bnf b/grammar/pou.bnf index b6d4cd9..679feca 100644 --- a/grammar/pou.bnf +++ b/grammar/pou.bnf @@ -1,17 +1,104 @@ -expression -> 'EXPRESSION' identifier expr_decl_section? stmt_section 'END_EXPRESSION' - ; +%include 'variables.bnf' - -function -> 'FUNCTION' identifier ':' ('VOID' | datatype) fc_declaration_section? stmt_section 'END_FUNCTION' - ; +# B.1.5 Program organization units - -function_block -> 'FUNCTION_BLOCK' identifier fb_declaration_section? stmt_section 'END_FUNCTION_BLOCK' +# B.1.5.1 Functions + +function_name -> standard_function_name + | derived_function_name + ; + +# FIXME: Complete standard function names from 2.5.1.5 +standard_function_name -> 'FIXME' + ; + +derived_function_name -> identifier + ; + +function_declaration -> 'FUNCTION' derived_function_name ':' _function_type_name? _function_decls function_body 'END_FUNCTION' + ; + +_function_type_name -> elementary_type_name + | derived_type_name + ; + +_function_decls -> io_var_declarations + | function_var_decls + ; + +io_var_declarations -> input_declarations + | output_declarations + | input_output_declarations + ; + +function_var_decls -> 'VAR' 'CONSTANT'? (var2_init_decl ';')+ 'END_VAR' + ; + +# FIXME: Confirm we can simplify this way +function_body -> sequential_function_chart | statement_list + ; + + +var2_init_decl -> var1_init_decl + | array_var_init_decl + | structured_var_init_decl + | string_var_declaration ; - -program -> 'PROGRAM' identifier program_declaration_section? stmt_section 'END_PROGRAM' - ; +# FIXME: NOTE 1 This syntax does not reflect the fact that each function must have at least one input declaration. + +# FIXME: NOTE 2 This syntax does not reflect the fact that edge declarations, function block references and invocations are not allowed in function bodies + +# B.1.5.2 Function blocks + +function_block_type_name -> standard_function_block_name + | derived_function_block_name + ; + +standard_function_block_name -> 'FIXME' # as defined in 2.5.2.3 + ; + +derived_function_block_name -> identifier + ; + +function_block_declaration -> 'FUNCTION_BLOCK' derived_function_block_name _function_decls? function_block_body 'END_FUNCTION_BLOCK' + ; + +other_var_declarations -> external_var_declarations + | var_declarations + | retentive_var_declarations + | non_retentive_var_declarations + | temp_var_decls + | incompl_located_var_declarations + ; + +temp_var_decls -> 'VAR_TEMP' (temp_var_decl ';')+ 'END_VAR' + ; + +non_retentive_var_declarations -> 'VAR' 'NON_RETAIN' (var_init_decl ';')+ 'END_VAR' + ; + +function_block_body -> sequential_function_chart | statement_list + ; + +# B.1.5.3 Programs + +program_type_name -> identifier + ; + +program_declaration -> 'PROGRAM' program_type_name _program_decls function_block_body 'END_PROGRAM' + ; + +_program_decls -> io_var_declarations + | other_var_declarations + | located_var_declarations + | program_access_decls + ; + +program_access_decls -> 'VAR_ACCESS' program_access_decl ';' (program_access_decl ';')* 'END_VAR' + ; + +program_access_decl -> access_name ':' symbolic_variable ':' non_generic_type_name direction? + ; - diff --git a/grammar/sfc.bnf b/grammar/sfc.bnf new file mode 100644 index 0000000..1ea9b19 --- /dev/null +++ b/grammar/sfc.bnf @@ -0,0 +1,66 @@ + +%include "pou.bnf" + +# B.1.6 Sequential function chart elements + +sequential_function_chart -> sfc_network+ + ; + +sfc_network -> initial_step (step | transition | action)* + ; + +initial_step -> 'INITIAL_STEP' step_name ':' (action_association ';')* 'END_STEP' + ; + +step -> 'STEP' step_name ':' (action_association ';')* 'END_STEP' + ; + +step_name -> identifier; + +# FIXME: this would allow a comma right after the parenthesis, what seems wrong - to confirm +action_association -> action_name '(' action_qualifier? (',' indicator_name)* ')' + ; + +action_name -> identifier + ; + +action_qualifier -> /{NRSP]/ + | timed_qualifier ',' action_time + ; + +timed_qualifier -> 'L' | 'D' | 'SD' | 'DS' | 'SL' + ; + +action_time -> duration + | variable_name + ; + +indicator_name -> variable_name + ; + +transition -> 'TRANSITION' transition_name? _priority? 'FROM' steps 'TO' steps transition_condition 'END_TRANSITION' + ; + + +_priority -> '(' 'PRIORITY' ':=' integer ')' + ; + +transition_name -> identifier + ; + +steps -> step_name + | '(' step_name (',' step_name)+ ')' + ; + +transition_condition -> +# FIXME: this one's for IL +# ':' simple_instruction_list + ':=' expression ';' +# FIXME: I suppose this is only for ladder diagrams and function block diagrams +# | ':' (fbd_network | rung) + ; + +action -> 'ACTION' action_name ':' function_block_body 'END_ACTION' + ; + +# NOTE 1 The non-terminals simple_instruction_list and expression are defined in B.2.1 and B.3.1, respectively. diff --git a/grammar/st.bnf b/grammar/st.bnf index 36994b2..8adb730 100644 --- a/grammar/st.bnf +++ b/grammar/st.bnf @@ -1,16 +1,151 @@ -%extras line_comment, block_comment, /\s/ +%include "literals.bnf" +%include "pou.bnf" +%include "sfc.bnf" +%include "configuration.bnf" -line_comment -> /\/\/.*/ +# B.0 Programming Model + +library_element_declaration -> data_type_declaration + | function_declaration + | function_block_declaration + | program_declaration + | configuration_declaration + ; + +# B.3.1 Expressions + +expression -> xor_expression ('OR' xor_expression)* + ; + +xor_expression -> and_expression ('XOR' and_expression)* + ; + +and_expression -> comparison (('&'|'AND') comparison)* + ; + +comparison -> equ_expression (('='|'<>') equ_expression)* + ; + + +equ_expression -> add_expression (comparison_operator add_expression)* + ; + +comparison_operator -> '<' | '>' | '<=' | '>=' + ; + +add_expression -> term (add_operator term)* + ; + +add_operator -> '+' | '-' ; -# FIXME: Need to confirm greedyness -block_comment -> /\(\*/ characters* /\*\)/ - ; +term -> power_expression (multiply_operator power_expression)* + ; + +multiply_operator -> '*' | '/' | 'MOD' + ; + +power_expression -> unary_expression ('**' unary_expression)* + ; + +unary_expression -> unary_operator? primary_expression + ; + +unary_operator -> '-' | 'NOT' + ; + +primary_expression -> constant + | enumerated_value + | variable + | '(' expression ')' + | function_name '(' param_assignment (',' param_assignment)* ')' + ; + + +# B3.2 Statements + +statement_list -> (statement? ';')+ + ; + +statement -> assignment_statement + | subprogram_control_statement + | selection_statement + | iteration_statement + ; + +# B.3.2.1 Assignment Statements + +assignment_statement -> variable ':=' expression + ; + + +# B.3.2.2 Subprogram control statements + +subprogram_control_statement -> fb_invocation + | 'RETURN' + ; + +fb_invocation -> fb_name '(' (param_assignment (',' param_assignment)*)? ')' + ; + +param_assignment -> (variable_name ':=')? expression + | 'NOT'? variable_name '=>' variable + ; + +# B.3.2.3 Selection Statements + +selection_statement -> if_statement + | case_statement + ; + +if_statement -> 'IF' expression 'THEN' statement_list _elsif* _else? 'END_IF' + ; + + +_elsif -> 'ELSIF' expression 'THEN' statement_list + ; + +_else -> 'ELSE' statement_list + ; + +case_statement -> 'CASE' expression 'OF' case_element+ _else? 'END_CASE' + ; + +case_element -> case_list ':' statement_list + ; + + +case_list -> case_list_element (',' case_list_element)* + ; + +case_list_element -> subrange + | signed_integer + | enumerated_value + ; + +# B.3.2.4 Iteration Statements + +iteration_statement -> for_statement + | while_statement + | repeat_statement + | exit_statement + ; + +for_statement -> 'FOR' control_variable ':=' for_list 'DO' statement_list 'END_FOR' + ; + +control_variable -> identifier + ; + +for_list -> expression 'TO' expression ('BY' expression)? + ; + +while_statement -> 'WHILE' expression 'DO' statement_list 'END_WHILE' + ; + +repeat_statement -> 'REPEAT' statement_list 'UNTIL' expression 'END_REPEAT' + ; -%include 'control.bnf' -%include 'literals.bnf' -%include 'pou.bnf' -%include 'decl_blocks.bnf' -%include 'data_types.bnf' -%include 'stmt_section.bnf' +exit_statement -> 'EXIT' + ; diff --git a/grammar/stmt_section.bnf b/grammar/stmt_section.bnf deleted file mode 100644 index 8d3212f..0000000 --- a/grammar/stmt_section.bnf +++ /dev/null @@ -1,11 +0,0 @@ - -stmt_section -> ((identifier ':')? statement? ';')+ - ; - - -statement -> value_assignments - | subroutine_execution - | control_stmt - ; - - diff --git a/grammar/variables.bnf b/grammar/variables.bnf new file mode 100644 index 0000000..4dd3f10 --- /dev/null +++ b/grammar/variables.bnf @@ -0,0 +1,217 @@ + +%include 'literals.bnf' + +# B.1.4 Variables + +variable -> direct_variable + | symbolic_variable + ; + +symbolic_variable -> variable_name + | multi_element_variable + ; + +variable_name -> identifier + ; + + +# B.1.4.1 Directly represented variables + +direct_variable -> '%' location_prefix size_prefix? integer ('.' integer)* + ; + +location_prefix -> /[IQM]/ + ; + +# FIXME: Grammar defines 'NIL', which I suppose is empty... +size_prefix -> /[XBWDL]/ + ; + +# B.1.4.2 Multi-element variables + +multi_element_variable -> array_variable + | structured_variable + ; + +array_variable -> subscripted_variable subscript_list + ; + +subscripted_variable -> symbolic_variable + ; + +subscript_list -> '[' subscript (',' subscript)* ']' + ; + +subscript -> expression + ; + +structured_variable -> record_variable '.' field_selector + ; + +record_variable -> symbolic_variable + ; + +field_selector -> identifier + ; + +# B.1.4.3 Declaration and initialization + +input_declarations -> 'VAR_INPUT' /(NON_)?RETAIN/? (input_declaration ';')+ 'END_VAR' + ; + +input_declaration -> var_init_decl + | edge_declaration + ; + +edge_declaration -> var1_list ':' 'BOOL' /[RF]_EDGE/ + ; + +var_init_decl -> var1_init_decl + | array_var_init_decl + | structured_var_init_decl + | fb_name_decl + | string_var_declaration + ; + +var1_init_decl -> var1_list ':' _var1_init_decl + ; + +_var1_init_decl -> simple_spec_init + | subrange_spec_init + | enumerated_spec_init + ; + +var1_list -> variable_name (',' variable_name)* + ; + +array_var_init_decl -> var1_list ':' array_spec_init + ; + +structured_var_init_decl -> var1_list ':' initialized_structure + ; + +fb_name_decl -> fb_name_list ':' function_block_type_name (':=' structure_initialization)? + ; + +fb_name_list -> fb_name (',' fb_name)* + ; + +fb_name -> identifier + ; + +output_declarations -> 'VAR_OUTPUT' /(NON_)?RETAIN/? (var_init_decl ';')+ 'END_VAR' + ; + +input_output_declarations -> 'VAR_IN_OUT' (var_declaration ';')+ 'END_VAR' + ; + +var_declaration -> temp_var_decl + | fb_name_decl + ; + +temp_var_decl -> var1_declaration + | array_var_declaration + | structured_var_declaration + | string_var_declaration + ; + +var1_declaration -> var1_list ':' _var1_declaration + ; + +_var1_declaration -> simple_specification + | subrange_specification + | enumerated_specification + ; + +array_var_declaration -> var1_list ':' array_specification + ; + +structured_var_declaration -> var1_list ':' structure_type_name + ; + +var_declarations -> 'VAR' 'CONSTANT'? (var_init_decl ';')+ 'END_VAR' + ; + +retentive_var_declarations -> 'VAR' 'RETAIN' (var_init_decl ';')+ 'END_VAR' + ; + +located_var_declarations -> 'VAR' ('CONSTANT' | /(NON_)?RETAIN/) ? (located_var_decl ';')+ 'END_VAR' + ; + +located_var_decl -> variable_name? location ':' located_var_spec_init + ; + +external_var_declarations -> 'VAR_EXTERNAL' 'CONSTANT'? (external_declaration ';')+ 'END_VAR' + ; + +external_declaration -> global_var_name ':' _external_declaration + ; + +_external_declaration -> simple_specification + | subrange_specification + | enumerated_specification + | structured_variable + | function_block_type_name + ; + +global_var_name -> identifier + ; + +global_var_declarations -> 'VAR_GLOBAL' ('CONSTANT'|'RETAIN')? (global_var_decl ';')+ 'END_VAR' + ; + +global_var_decl -> global_var_spec ':' (located_var_spec_init | function_block_type_name)? + ; + +global_var_spec -> global_var_list + | global_var_name? location + ; + +located_var_spec_init -> simple_spec_init + | subrange_spec_init + | enumerated_spec_init + | array_spec_init + | initialized_structure + | single_byte_string_spec + | double_byte_string_spec + ; + +location -> 'AT' direct_variable + ; + +global_var_list -> global_var_name (',' global_var_name)* + ; + +string_var_declaration -> single_byte_string_var_declaration + | double_byte_string_var_declaration + ; + +single_byte_string_var_declaration -> var1_list ':' single_byte_string_spec + ; + +single_byte_string_spec -> 'STRING' ('[' integer ']')? (':=' single_byte_character_string)? + ; + + +double_byte_string_var_declaration -> var1_list ':' double_byte_string_spec + ; + +double_byte_string_spec -> 'WSTRING' ('[' integer ']')? (':=' double_byte_character_string)? + ; + +incompl_located_var_declarations -> 'VAR' /(NON_)?RETAIN/? (incompl_located_var_decl ';')+ 'END_VAR' + ; + +incompl_located_var_decl -> variable_name incompl_location ':' var_spec + ; + +incompl_location -> 'AT' '%' /[IQM]/ '*' + ; + +var_spec -> simple_specification + | subrange_specification + | enumerated_specification + | array_spec_init + | structure_type_name + | /W?STRING/ ('[' integer ']') + ; From 2bccc67b0a07bc55c09d5ad680a850159a6a96b1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Tue, 21 Jul 2026 12:37:12 +0100 Subject: [PATCH 5/7] Resolve unresolved tree-sitter grammar conflicts Several places where the IEC grammar reduces a bare identifier through multiple aliased or overlapping rules (type vs. function-block names, derived type-name variants, access-path segments) are genuinely ambiguous from syntax alone - which one applies is only decidable with a symbol table - so those spots are collapsed onto the existing shared `type_name`/`symbolic_variable` rules instead, deferring the distinction to a later semantic pass. Other conflicts are fixed more locally: - signed_integer/real_literal fold their optional sign into the token regex so tree-sitter's longest-match lexing picks them over a bare unary-minus operator. - bit_string_literal requires its type prefix, since an unprefixed integer was ambiguous with a plain integer literal. - character_string (used where no STRING/WSTRING keyword disambiguates) is now a single unaliased rule instead of a single-vs-double-byte choice that shared the same delimiter and representation set. - param_assignment's `NOT var_name => var` vs. unary_operator's `NOT expr` is a genuine local ambiguity (resolvable with more lookahead, not by restructuring), so it's handed to GLR via `%conflicts`. Also drops fb_name_decl, made dead by the type_name consolidation. --- grammar/configuration.bnf | 20 ++++++--- grammar/data_types.bnf | 91 +++++++++++++++++++++------------------ grammar/literals.bnf | 43 +++++++++++++++--- grammar/pou.bnf | 9 +++- grammar/sfc.bnf | 2 +- grammar/st.bnf | 15 ++++++- grammar/variables.bnf | 62 +++++++++++++++++--------- 7 files changed, 165 insertions(+), 77 deletions(-) diff --git a/grammar/configuration.bnf b/grammar/configuration.bnf index 2c6ad5a..f513533 100644 --- a/grammar/configuration.bnf +++ b/grammar/configuration.bnf @@ -31,8 +31,18 @@ access_declarations -> 'VAR_ACCESS' (access_declaration ';')+ 'ENV_VAR' access_declaration -> access_name ':' access_path ':' non_generic_type_name direction? ; -access_path -> (resource_name '.')? direct_variable - | (resource_name '.')? (program_name '.')? (fb_name '.')* symbolic_variable +# NOTE: deviates from the IEC grammar, which also builds an explicit +# `(resource_name '.')? (program_name '.')? ((identifier => fb_name) '.')*` +# prefix chain here before handing off to `symbolic_variable`. But +# `symbolic_variable` is already recursive through `structured_variable` / +# `record_variable` (variables.bnf), so `res.prog.fb1.var` already parses on +# its own as nested struct-field access - the explicit prefix chain was a +# second, redundant way to build the exact same dotted identifier sequence. +# Which segment is a resource, a program, an FB instance, or a struct field +# is a symbol-table question in any case, not something this CFG can answer, +# so we drop the redundant chain and rely on `symbolic_variable` alone. +access_path -> direct_variable + | symbolic_variable ; global_var_reference -> (resource_name '.')? global_var_name ('.' structure_element_name)? @@ -79,7 +89,7 @@ prog_conf_elements -> prog_conf_element (',' prog_conf_element)* prog_conf_element -> fb_task | prog_cnxn ; -fb_task -> fb_name 'WITH' task_name +fb_task -> (identifier => fb_name) 'WITH' task_name ; prog_cnxn -> symbolic_variable ':=' prog_data_source @@ -99,10 +109,10 @@ data_sink -> global_var_reference instance_specific_initializations -> 'VAR_CONFIG' (instance_specific_init ';')+ 'ENV_VAR' ; -instance_specific_init -> resource_name '.' program_name '.' (fb_name '.')* _instance_specific_init; +instance_specific_init -> resource_name '.' program_name '.' ((identifier => fb_name) '.')* _instance_specific_init; _instance_specific_init -> variable_name location? ':' located_var_spec_init - | fb_name ':' function_block_type_name ':=' structure_initialization + | (identifier => fb_name) ':' function_block_type_name ':=' structure_initialization ; # This syntax does not reflect the fact that location assignments are only allowed for references to variables which are marked by the asterisk notation at type declaration level. diff --git a/grammar/data_types.bnf b/grammar/data_types.bnf index 83d8d9c..5b6b5d8 100644 --- a/grammar/data_types.bnf +++ b/grammar/data_types.bnf @@ -61,33 +61,42 @@ bit_string_type_name -> 'BOOL' # B.1.3.3 Derived data types derived_type_name -> single_element_type_name - | array_type_name - | structure_type_name - | string_type_name + | (identifier => array_type_name) + | (identifier => structure_type_name) + | (identifier => string_type_name) ; -single_element_type_name -> simple_type_name - | subrange_type_name - | enumerated_type_name +single_element_type_name -> (identifier => simple_type_name) + | (identifier => subrange_type_name) + | (identifier => enumerated_type_name) ; -#FIXME: later we need to merge all these - -simple_type_name -> identifier - ; - -subrange_type_name -> identifier - ; - -enumerated_type_name -> identifier - ; - -array_type_name -> identifier - ; +# NOTE: simple_type_name, subrange_type_name, enumerated_type_name, array_type_name, +# structure_type_name and string_type_name are all syntactically bare identifiers. +# They are kept as distinct node labels via aliasing (rather than as separate rules) +# so the parser only ever has one production to reduce an identifier through here; +# which label actually applies is only resolved once enough of the surrounding +# declaration has been parsed to tell the kinds apart. + +# NOTE: a declaration that is just `identifier` on the right of ':' is an alias to +# some other, already-declared type. Its actual kind (simple/subrange/enumerated/ +# array/structure) is not decidable from syntax alone - that requires a symbol +# table. So this is kept as ONE shared rule and referenced wherever a bare alias +# would otherwise be reachable through more than one of simple_specification, +# subrange_specification, enumerated_specification, array_specification or a +# structure alias - keeping it in more than one of those would make the parser +# choose between indistinguishable alternatives. +type_name -> identifier + ; + +_type_name_reference_init -> constant + | enumerated_value + | array_initialization + | structure_initialization + ; -structure_type_name -> identifier - ; - +type_name_reference -> type_name (':=' _type_name_reference_init)? + ; data_type_declaration -> 'TYPE' (type_declaration ';')+ 'END_TYPE' ; @@ -96,6 +105,7 @@ type_declaration -> single_element_type_declaration | array_type_declaration | structure_type_declaration | string_type_declaration + | type_name_reference ; single_element_type_declaration -> simple_type_declaration @@ -103,50 +113,47 @@ single_element_type_declaration -> simple_type_declaration | enumerated_type_declaration ; -simple_type_declaration -> simple_type_name ':' simple_spec_init +simple_type_declaration -> (identifier => simple_type_name) ':' simple_spec_init ; simple_spec_init -> simple_specification (':=' constant)? ; simple_specification -> elementary_type_name - | simple_type_name ; -subrange_type_declaration -> subrange_type_name ':' subrange_spec_init +subrange_type_declaration -> (identifier => subrange_type_name) ':' subrange_spec_init ; - + subrange_spec_init -> subrange_specification (':=' signed_integer)? ; -subrange_specification -> integer_type_name '(' subrange ')' | subrange_type_name +subrange_specification -> integer_type_name '(' subrange ')' ; subrange -> signed_integer '..' signed_integer ; -enumerated_type_declaration -> enumerated_type_name ':' enumerated_spec_init +enumerated_type_declaration -> (identifier => enumerated_type_name) ':' enumerated_spec_init ; enumerated_spec_init -> enumerated_specification (':=' enumerated_value)? ; enumerated_specification -> '(' enumerated_value (',' enumerated_value)* ')' - | enumerated_type_name ; -enumerated_value -> (enumerated_type_name '#') identifier +enumerated_value -> ((identifier => enumerated_type_name) '#') identifier ; -array_type_declaration -> array_type_name ':' array_spec_init +array_type_declaration -> (identifier => array_type_name) ':' array_spec_init ; array_spec_init -> array_specification (':=' array_initialization)? ; -array_specification -> array_type_name - | 'ARRAY' '[' subrange (',' subrange)* ']' 'OF' non_generic_type_name +array_specification -> 'ARRAY' '[' subrange (',' subrange)* ']' 'OF' non_generic_type_name ; array_initialization -> '[' array_initial_elements (',' array_initial_elements)* ']' @@ -162,16 +169,12 @@ array_initial_element -> constant | array_initialization ; -structure_type_declaration -> structure_type_name ':' structure_specification +structure_type_declaration -> (identifier => structure_type_name) ':' structure_specification ; structure_specification -> structure_declaration - | initialized_structure ; -initialized_structure -> structure_type_name (':=' structure_initialization)? - ; - structure_declaration -> 'STRUCT' (structure_element_declaration ';')+ 'END_STRUCT' ; @@ -182,7 +185,7 @@ _structure_element_declaration -> simple_spec_init | subrange_spec_init | enumerated_spec_init | array_spec_init - | initialized_structure + | type_name_reference ; structure_element_name -> identifier @@ -200,9 +203,11 @@ _structure_element_initialization -> constant | structure_initialization ; -string_type_name -> identifier - ; - -string_type_declaration -> string_type_name ':' /W?STRING/ ('[' integer ']')? (':=' character_string)? +# NOTE: the '[' length ']' is mandatory here. A bare STRING/WSTRING with no length +# and no initializer is already covered by simple_type_declaration, via +# elementary_type_name -> /W?STRING/ (see simple_specification). Making the +# length optional here as well would let both rules derive the same input +# (e.g. `Foo : STRING;`), which is an unresolvable grammar conflict. +string_type_declaration -> (identifier => string_type_name) ':' /W?STRING/ '[' integer ']' (':=' character_string)? ; diff --git a/grammar/literals.bnf b/grammar/literals.bnf index 468f80a..187f202 100644 --- a/grammar/literals.bnf +++ b/grammar/literals.bnf @@ -31,7 +31,15 @@ _integer_literal -> signed_integer | hex_integer ; -signed_integer -> ('+'|'-')? integer +# NOTE: the sign is folded into the regex (rather than `('+'|'-')? integer`) +# so this is a single token. As two grammar symbols, the leading '-' would be +# the same terminal `unary_operator` uses, and at a position like `IF -5` +# the parser can't tell which production is consuming it - a genuine +# parser-level ambiguity. As one token, tree-sitter's lexer resolves it by +# longest match: `-5` (2 chars) wins over the lone `-` (1 char) wherever both +# are valid, so `unary_operator` only ever sees a bare sign when no digits +# immediately follow it. +signed_integer -> /[+-]?[0-9](_?[0-9])*/ ; integer -> /[0-9](_?[0-9])*/ @@ -47,10 +55,18 @@ hex_integer -> '16#' /[0-9A-F](_?[0-9A-F])*/ ; # FIXME: If spaces are possible around exponents, for instance, we need to check the regex -real_literal -> (real_type_name '#')? ('+'|'-')? /[0-9](_?[0-9])*\.[0-9](_?[0-9])*([Ee][+-]?[0-9](_?[0-9])*)?/ +# NOTE: sign folded into the regex, same reasoning as `signed_integer` above. +real_literal -> (real_type_name '#')? /[+-]?[0-9](_?[0-9])*\.[0-9](_?[0-9])*([Ee][+-]?[0-9](_?[0-9])*)?/ ; -bit_string_literal -> ( 'BYTE' | /[DL]?WORD/ )? _unsigned_integer_literal +# NOTE: deviates from the IEC grammar, where the type prefix here is optional. +# With it optional, an unprefixed integer is ambiguous between this rule and +# signed_integer (both reduce a bare `integer` with nothing else), which the +# IEC text only disambiguates via the semantics/context of the assignment +# target - something this CFG can't express. We require the prefix so the +# grammar stays unambiguous; recovering untagged bit-string literals, if +# needed, has to happen in a later semantic/type-checking pass. +bit_string_literal -> ('BYTE' | /[DL]?WORD/) '#' _unsigned_integer_literal ; _unsigned_integer_literal -> integer @@ -65,10 +81,27 @@ boolean_literal -> 'BOOL#'? /[10]/ | 'TRUE' | 'FALSE' # B.1.2.2 Character Strings -character_string -> single_byte_character_string - | double_byte_character_string +# NOTE: deviates from the IEC grammar, which offers `character_string` here as +# a choice between single_byte_character_string and double_byte_character_string +# - both delimited the same way, with mostly-overlapping representation sets. +# Even the escapes are ambiguous on their own: `$AAAA` is either one 4-digit +# escape, or a 2-digit escape `$AA` followed by two literal characters 'A','A' +# - so no amount of scanning tells single-byte from double-byte here. Which +# one it is only becomes knowable once the target's declared type (STRING vs +# WSTRING) is known - semantics/context, not this CFG. `single_byte_character_string` +# and `double_byte_character_string` stay directly usable where a preceding +# 'STRING'/'WSTRING' keyword already disambiguates (see variables.bnf); here, +# with no such keyword around, we use one shared, unaliased rule and defer the +# single/double-byte distinction to a later semantic pass. +character_string -> "'" character_representation* "'" ; +character_representation -> common_character_representation + | "$'" + | '"' + | /\$[0-9A-F][0-9A-F]([0-9A-F][0-9A-F])?/ + ; + single_byte_character_string -> "'" single_byte_character_representation* "'" ; diff --git a/grammar/pou.bnf b/grammar/pou.bnf index 679feca..35dfb41 100644 --- a/grammar/pou.bnf +++ b/grammar/pou.bnf @@ -19,8 +19,15 @@ derived_function_name -> identifier function_declaration -> 'FUNCTION' derived_function_name ':' _function_type_name? _function_decls function_body 'END_FUNCTION' ; +# NOTE: deviates from the IEC grammar, where the derived case here is +# `derived_type_name`. That rule aliases a bare identifier six different ways +# (simple/subrange/enumerated/array/structure/string type name) with nothing +# to tell them apart syntactically - which kind it is can only be known once +# it's looked up in a symbol table, i.e. semantics/context, not this CFG. We +# use the plain, unaliased `type_name` instead, same as `type_name_reference` +# does for the analogous "identifier refers to some other declared type" case. _function_type_name -> elementary_type_name - | derived_type_name + | type_name ; _function_decls -> io_var_declarations diff --git a/grammar/sfc.bnf b/grammar/sfc.bnf index 1ea9b19..e0d2ec5 100644 --- a/grammar/sfc.bnf +++ b/grammar/sfc.bnf @@ -24,7 +24,7 @@ action_association -> action_name '(' action_qualifier? (',' indicator_name)* ') action_name -> identifier ; -action_qualifier -> /{NRSP]/ +action_qualifier -> /[NRSP]/ | timed_qualifier ',' action_time ; diff --git a/grammar/st.bnf b/grammar/st.bnf index 8adb730..9473100 100644 --- a/grammar/st.bnf +++ b/grammar/st.bnf @@ -4,6 +4,19 @@ %include "sfc.bnf" %include "configuration.bnf" +# param_assignment's `'NOT'? variable_name '=>' variable` (negate an FB's +# output parameter on the way out) and unary_operator's `'NOT' primary_expression` +# (boolean negation in an expression) both start with 'NOT' identifier, and +# there's no precedence that makes one "win" over the other in general - +# whether a given `NOT x` belongs to the output form is only known once '=>' +# does or doesn't turn up right after. This is genuinely ambiguous locally, +# not something restructuring the grammar can remove, so we hand it to GLR. +# Tree-sitter's default conflict heuristic (prefer the longer/shift parse) +# gives the right answer: it tries the `'=>' variable` continuation first and +# falls back to a plain boolean-negation expression when '=>' isn't there. +# Verified against `NOT ENO => flag`, `x := NOT ENO`, and bare `NOT ENO`. +%conflicts [param_assignment, unary_operator] + # B.0 Programming Model library_element_declaration -> data_type_declaration @@ -86,7 +99,7 @@ subprogram_control_statement -> fb_invocation | 'RETURN' ; -fb_invocation -> fb_name '(' (param_assignment (',' param_assignment)*)? ')' +fb_invocation -> (identifier => fb_name) '(' (param_assignment (',' param_assignment)*)? ')' ; param_assignment -> (variable_name ':=')? expression diff --git a/grammar/variables.bnf b/grammar/variables.bnf index 4dd3f10..44dbcf9 100644 --- a/grammar/variables.bnf +++ b/grammar/variables.bnf @@ -66,10 +66,21 @@ input_declaration -> var_init_decl edge_declaration -> var1_list ':' 'BOOL' /[RF]_EDGE/ ; +# NOTE: deviates from the IEC grammar, which also offers `fb_name_decl` here +# (`(var1_list => fb_name_list) ':' function_block_type_name (':=' structure_initialization)?`). +# `function_block_type_name` is, in practice, just `derived_function_block_name +# -> identifier` (since `standard_function_block_name` is still a placeholder) +# - a bare identifier with nothing to distinguish it from `type_name` in +# `structured_var_init_decl` below. "This names a FUNCTION_BLOCK" vs "this +# names a TYPE" is only knowable from a symbol table, i.e. semantics/context, +# not this CFG. It's also redundant as written: `type_name_reference`'s +# initializer already includes `structure_initialization`, the only +# initializer `fb_name_decl` offered. Dropped in favor of +# `structured_var_init_decl` for both cases; `fb_name_decl` and its +# `fb_name_list` alias are removed entirely, since this was their only use. var_init_decl -> var1_init_decl | array_var_init_decl | structured_var_init_decl - | fb_name_decl | string_var_declaration ; @@ -87,26 +98,24 @@ var1_list -> variable_name (',' variable_name)* array_var_init_decl -> var1_list ':' array_spec_init ; -structured_var_init_decl -> var1_list ':' initialized_structure +structured_var_init_decl -> var1_list ':' type_name_reference ; -fb_name_decl -> fb_name_list ':' function_block_type_name (':=' structure_initialization)? - ; - -fb_name_list -> fb_name (',' fb_name)* - ; - -fb_name -> identifier - ; - output_declarations -> 'VAR_OUTPUT' /(NON_)?RETAIN/? (var_init_decl ';')+ 'END_VAR' ; input_output_declarations -> 'VAR_IN_OUT' (var_declaration ';')+ 'END_VAR' ; +# NOTE: deviates from the IEC grammar, which also offers `fb_name_decl` as an +# alternative here, same reasoning as `var_init_decl` above: whether a bare +# `foo : Bar;` names a FUNCTION_BLOCK instance or a struct-typed variable +# isn't decidable without a symbol table, and `fb_name_decl`'s only other +# content, `(':=' structure_initialization)?`, is optional and VAR_IN_OUT +# doesn't allow initializers anyway - so it adds nothing `structured_var_declaration` +# doesn't already accept. Dropped here (see `var_init_decl` above for where +# `fb_name_decl` itself was removed). var_declaration -> temp_var_decl - | fb_name_decl ; temp_var_decl -> var1_declaration @@ -126,7 +135,12 @@ _var1_declaration -> simple_specification array_var_declaration -> var1_list ':' array_specification ; -structured_var_declaration -> var1_list ':' structure_type_name +# NOTE: uses `type_name`, not `(identifier => structure_type_name)` as the +# IEC grammar does - since this now also covers the merged FUNCTION_BLOCK +# case (see `var_declaration` above), aliasing it specifically as +# "structure_type_name" would be misleading; which of the two it actually is +# isn't decidable here anyway. +structured_var_declaration -> var1_list ':' type_name ; var_declarations -> 'VAR' 'CONSTANT'? (var_init_decl ';')+ 'END_VAR' @@ -147,11 +161,15 @@ external_var_declarations -> 'VAR_EXTERNAL' 'CONSTANT'? (external_declaration '; external_declaration -> global_var_name ':' _external_declaration ; +# NOTE: deviates from the IEC grammar, which also offers `function_block_type_name` +# here alongside `type_name` - same ambiguity as above (both reduce to a bare +# identifier, and which kind it is isn't decidable without a symbol table). +# Dropped in favor of `type_name` alone. _external_declaration -> simple_specification | subrange_specification | enumerated_specification | structured_variable - | function_block_type_name + | type_name ; global_var_name -> identifier @@ -160,18 +178,20 @@ global_var_name -> identifier global_var_declarations -> 'VAR_GLOBAL' ('CONSTANT'|'RETAIN')? (global_var_decl ';')+ 'END_VAR' ; -global_var_decl -> global_var_spec ':' (located_var_spec_init | function_block_type_name)? +# NOTE: deviates from the IEC grammar, which also offers `function_block_type_name` +# here alongside `located_var_spec_init` - same ambiguity as `_external_declaration` +# above: `located_var_spec_init` already reaches `type_name` via +# `_structure_element_declaration -> ... | type_name_reference` +# (data_types.bnf), so `function_block_type_name` was a redundant, ambiguous +# alternative. Dropped. +global_var_decl -> global_var_spec ':' located_var_spec_init? ; global_var_spec -> global_var_list | global_var_name? location ; -located_var_spec_init -> simple_spec_init - | subrange_spec_init - | enumerated_spec_init - | array_spec_init - | initialized_structure +located_var_spec_init -> _structure_element_declaration | single_byte_string_spec | double_byte_string_spec ; @@ -212,6 +232,6 @@ var_spec -> simple_specification | subrange_specification | enumerated_specification | array_spec_init - | structure_type_name + | type_name | /W?STRING/ ('[' integer ']') ; From 2cb189e00ffcdb2e2158b8bf968bc62348f27903 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Wed, 22 Jul 2026 10:30:46 +0100 Subject: [PATCH 6/7] Fix remaining unresolved tree-sitter grammar conflicts Five GLR conflicts remained after the previous pass, all the same root cause: a bare identifier reachable through more than one differently labeled rule with nothing in the following tokens to tell them apart (program_access_decl/access_declaration's type name, array_specification's element type, data_source's resource/program/global-var prefix, global_var_reference's own prefix/suffix, and _instance_specific_init's fb_name vs variable_name). Each is resolved the same way the file already resolves this pattern elsewhere: drop the redundant labeled alternative and rely on the single already-reachable rule (type_name, symbolic_variable), letting a symbol table (not this CFG) resolve which kind it actually is. Dead rules left over from these fixes (non_generic_type_name and its six-way identifier aliasing, program_output_reference, function_block_type_name's derived-name half) are removed, and standard_function_block_name's FIXME placeholder is filled in with the fixed set of names from IEC 61131-3:2003 2.5.2.3 (SR/RS, R_TRIG/F_TRIG, CTU/CTD/CTUD variants, TP/TON/TOF), with a new call site for it. Also adds two pieces of parser infrastructure that were missing entirely: `%extras` for `(* ... *)` and `//` comments (real ST code is full of them), and a fix for the generated parser's start rule, which had no explicit axiom and so defaulted to whichever rule happened to land first after %include flattening - silently `elementary_type_name` rather than `library_element_declaration`. --- grammar/configuration.bnf | 41 +++++++++++++++++++++------ grammar/data_types.bnf | 59 +++++++++++++++++++++++++++++---------- grammar/pou.bnf | 32 +++++++++++++++++++-- grammar/st.bnf | 21 ++++++++++++++ grammar/variables.bnf | 30 ++++++++++++-------- 5 files changed, 146 insertions(+), 37 deletions(-) diff --git a/grammar/configuration.bnf b/grammar/configuration.bnf index f513533..6b44aef 100644 --- a/grammar/configuration.bnf +++ b/grammar/configuration.bnf @@ -28,7 +28,7 @@ access_declarations -> 'VAR_ACCESS' (access_declaration ';')+ 'ENV_VAR' ; -access_declaration -> access_name ':' access_path ':' non_generic_type_name direction? +access_declaration -> access_name ':' access_path ':' _type_name_or_elementary direction? ; # NOTE: deviates from the IEC grammar, which also builds an explicit @@ -45,16 +45,24 @@ access_path -> direct_variable | symbolic_variable ; -global_var_reference -> (resource_name '.')? global_var_name ('.' structure_element_name)? +# NOTE: deviates from the IEC grammar, which spells this out explicitly as +# `(resource_name '.')? global_var_name ('.' structure_element_name)?`. That +# shape is internally ambiguous on its own: given `a.b`, nothing tells the +# parser whether `a` is a `resource_name` (prefix present, no suffix) or `a` +# is the `global_var_name` itself (no prefix, `b` is the suffix +# `structure_element_name`) - both derive the same two-identifier chain. Same +# root cause as the `access_path` NOTE above (which segment is a resource, +# program, FB instance, or struct field is a symbol-table question, not a +# grammar one), so the fix is the same: drop the explicit prefix/suffix chain +# and rely on `symbolic_variable`'s own recursion through `structured_variable` +# to parse `a.b` unambiguously as nested struct-field access. +global_var_reference -> symbolic_variable ; access_name -> identifier ; -program_output_reference -> program_name '.' symbolic_variable - ; - program_name -> identifier ; @@ -70,10 +78,17 @@ task_name -> identifier task_initialization -> '(' ('SINGLE' ':=' data_source ',')? ('INTERVAL' ':=' data_source ',')? 'PRIORITY' ':=' integer ')' ; +# NOTE: deviates from the IEC grammar, which offers `global_var_reference` +# and `program_output_reference` here instead of `access_path`. Both build a +# `(resource_name|program_name) '.' ...` prefix in front of what is, once +# you look past the labels, the same dotted-identifier chain `symbolic_variable` +# already parses via `structured_variable`'s recursion - see the comment on +# `access_path` above. Kept apart, they're ambiguous: `a.b.c` matches both +# `resource_name '.' global_var_name '.' structure_element_name` and +# `program_name '.' symbolic_variable` with no way to tell which without a +# symbol table. `access_path` already covers direct and symbolic variables. data_source -> constant - | global_var_reference - | program_output_reference - | direct_variable + | access_path ; program_configuration -> 'PROGRAM' /(NON_)?RETAIN/? program_name ('WITH' task_name)? ':' program_type_name _prog_conf_elements? @@ -111,8 +126,16 @@ instance_specific_initializations -> 'VAR_CONFIG' (instance_specific_init ';')+ instance_specific_init -> resource_name '.' program_name '.' ((identifier => fb_name) '.')* _instance_specific_init; +# NOTE: deviates from the IEC grammar, which also offers +# `(identifier => fb_name) ':' function_block_type_name ':=' structure_initialization` +# here alongside `variable_name location? ':' located_var_spec_init` - same +# ambiguity as `external_declaration`/`global_var_decl` (variables.bnf): +# `located_var_spec_init` already reaches a bare type name via +# `_structure_element_declaration -> type_name_reference -> type_name`, and +# `type_name_reference`'s optional initializer already covers +# `structure_initialization`, so `fb_name : function_block_type_name := +# structure_initialization` is fully subsumed. Dropped. _instance_specific_init -> variable_name location? ':' located_var_spec_init - | (identifier => fb_name) ':' function_block_type_name ':=' structure_initialization ; # This syntax does not reflect the fact that location assignments are only allowed for references to variables which are marked by the asterisk notation at type declaration level. diff --git a/grammar/data_types.bnf b/grammar/data_types.bnf index 5b6b5d8..f359bdf 100644 --- a/grammar/data_types.bnf +++ b/grammar/data_types.bnf @@ -6,9 +6,17 @@ # | generic_type_name #; -non_generic_type_name -> elementary_type_name - | derived_type_name - ; +# NOTE: the IEC grammar's `non_generic_type_name -> elementary_type_name | +# derived_type_name` used to live here. `derived_type_name` aliased a bare +# identifier six different ways (simple/subrange/enumerated/array/structure/ +# string type name - see the old `single_element_type_name` further down) +# with nothing syntactically to tell them apart. Every call site +# (`program_access_decl`, `access_declaration`, `array_specification`'s +# element type) only has `:=`/`;`/`direction` following it - never enough +# context to resolve which alias applies, which tree-sitter reports as an +# unresolvable conflict. Each call site now uses `_type_name_or_elementary` +# (further down) instead, which relies on the plain `type_name` for the +# derived case - same fix already applied to `_function_type_name` (pou.bnf). # B.1.3.1 Elementary data types @@ -60,16 +68,27 @@ bit_string_type_name -> 'BOOL' # B.1.3.3 Derived data types -derived_type_name -> single_element_type_name - | (identifier => array_type_name) - | (identifier => structure_type_name) - | (identifier => string_type_name) - ; - -single_element_type_name -> (identifier => simple_type_name) - | (identifier => subrange_type_name) - | (identifier => enumerated_type_name) - ; +# NOTE: `derived_type_name` and `single_element_type_name` used to be defined +# here as: +# +# derived_type_name -> single_element_type_name +# | (identifier => array_type_name) +# | (identifier => structure_type_name) +# | (identifier => string_type_name) +# ; +# +# single_element_type_name -> (identifier => simple_type_name) +# | (identifier => subrange_type_name) +# | (identifier => enumerated_type_name) +# ; +# +# i.e. both were pure aliasing rules bundling the six bare-identifier type +# name aliases together with no other content. Once their only remaining +# caller, `non_generic_type_name`, was removed (see the NOTE above), they had +# no call sites left and were dropped rather than kept as unused grammar. The +# six aliases themselves (`simple_type_name` etc.) are still very much in +# use - directly, at each `type_declaration` alternative below - so only the +# two bundling rules went away, not the aliases. # NOTE: simple_type_name, subrange_type_name, enumerated_type_name, array_type_name, # structure_type_name and string_type_name are all syntactically bare identifiers. @@ -89,6 +108,18 @@ single_element_type_name -> (identifier => simple_type_name) type_name -> identifier ; +# Same rationale as `_function_type_name` (pou.bnf): the derived case here +# would normally be `derived_type_name`, but that rule aliases a bare +# identifier six different ways with nothing to tell them apart +# syntactically, so a type name reachable through it with no following +# context to disambiguate is an unresolvable conflict. Used wherever a type +# name is followed by nothing more specific than `:=`/`;`/a `direction` - +# never enough to tell which alias applies: `program_access_decl`, +# `access_declaration`, and `array_specification`'s element type. +_type_name_or_elementary -> elementary_type_name + | type_name + ; + _type_name_reference_init -> constant | enumerated_value | array_initialization @@ -153,7 +184,7 @@ array_type_declaration -> (identifier => array_type_name) ':' array_spec_init array_spec_init -> array_specification (':=' array_initialization)? ; -array_specification -> 'ARRAY' '[' subrange (',' subrange)* ']' 'OF' non_generic_type_name +array_specification -> 'ARRAY' '[' subrange (',' subrange)* ']' 'OF' _type_name_or_elementary ; array_initialization -> '[' array_initial_elements (',' array_initial_elements)* ']' diff --git a/grammar/pou.bnf b/grammar/pou.bnf index 35dfb41..1a84579 100644 --- a/grammar/pou.bnf +++ b/grammar/pou.bnf @@ -59,11 +59,37 @@ var2_init_decl -> var1_init_decl # B.1.5.2 Function blocks +# NOTE: unlike the IEC grammar, this does not also offer +# `derived_function_block_name` here. That case is a bare identifier, +# indistinguishable from `type_name` wherever both would be reachable at the +# same position (see the NOTEs around `var_init_decl` and +# `structured_var_declaration` in variables.bnf) - callers that need to +# accept a user-declared FB type already do so via `type_name`/ +# `type_name_reference`. This rule only needs to cover the standard names, +# which are fixed keyword literals and so don't collide with `type_name` the +# way a second bare-identifier rule would. function_block_type_name -> standard_function_block_name - | derived_function_block_name ; -standard_function_block_name -> 'FIXME' # as defined in 2.5.2.3 +# IEC 61131-3:2003 ยง2.5.2.3 (tables 34-38): the fixed set of standard +# function blocks common to every conformant implementation. +# - Bistable (2.5.2.3.1): SR (set-dominant), RS (reset-dominant) +# - Edge detection (2.5.2.3.2): R_TRIG, F_TRIG +# - Counters (2.5.2.3.3): CTU/CTD/CTUD, each with typed variants suffixed +# by their PV/CV type. CTU and CTD each have all five (INT is the +# unsuffixed base name, plus _DINT/_LINT/_UDINT/_ULINT); CTUD only has +# four - the standard's own table 36 has no CTUD_UDINT. +# - Timers (2.5.2.3.4): TP (pulse), TON (on-delay), TOF (off-delay). Table +# 37 also lists graphical-only shorthand forms, but "in textual +# languages, features 2b and 3b shall not be used" - not relevant here. +# Communication function blocks (2.5.2.3.5) are deferred entirely to +# IEC 61131-5, with no concrete names given in this part. +standard_function_block_name -> 'SR' | 'RS' + | 'R_TRIG' | 'F_TRIG' + | 'CTU' | 'CTU_DINT' | 'CTU_LINT' | 'CTU_UDINT' | 'CTU_ULINT' + | 'CTD' | 'CTD_DINT' | 'CTD_LINT' | 'CTD_UDINT' | 'CTD_ULINT' + | 'CTUD' | 'CTUD_DINT' | 'CTUD_LINT' | 'CTUD_ULINT' + | 'TP' | 'TON' | 'TOF' ; derived_function_block_name -> identifier @@ -106,6 +132,6 @@ _program_decls -> io_var_declarations program_access_decls -> 'VAR_ACCESS' program_access_decl ';' (program_access_decl ';')* 'END_VAR' ; -program_access_decl -> access_name ':' symbolic_variable ':' non_generic_type_name direction? +program_access_decl -> access_name ':' symbolic_variable ':' _type_name_or_elementary direction? ; diff --git a/grammar/st.bnf b/grammar/st.bnf index 9473100..fbf2b41 100644 --- a/grammar/st.bnf +++ b/grammar/st.bnf @@ -17,6 +17,11 @@ # Verified against `NOT ENO => flag`, `x := NOT ENO`, and bare `NOT ENO`. %conflicts [param_assignment, unary_operator] +# statement_list vs case_element: identifier after a statement could start +# a new assignment_statement or the next case_element's enumerated_value. +# Only disambiguated by the token after (# vs :=/(), so let GLR fork here. +%conflicts [statement_list] + # B.0 Programming Model library_element_declaration -> data_type_declaration @@ -26,6 +31,22 @@ library_element_declaration -> data_type_declaration | configuration_declaration ; +# IEC 61131-3 supports `(* ... *)` block comments (2003 edition, table B.1.7 +# lexical elements) and `//` line comments (a near-universal vendor extension, +# not in the 2003 text but present in virtually every real-world ST codebase - +# see tests/community_samples.txt, which uses both). Declared via `%extras` +# so a comment can appear anywhere whitespace can, without every rule needing +# to account for it explicitly. Kept below `library_element_declaration` +# deliberately: the generated parser's start rule defaults to the first rule +# written in this file, and `library_element_declaration` is the only rule +# that makes sense as "the thing this grammar parses" - putting `comment` +# ahead of it would silently hijack the start rule. +comment -> /\(\*[\s\S]*?\*\)/ + | /\/\/[^\n]*/ + ; + +%extras /\s/, comment + # B.3.1 Expressions expression -> xor_expression ('OR' xor_expression)* diff --git a/grammar/variables.bnf b/grammar/variables.bnf index 44dbcf9..9565ba3 100644 --- a/grammar/variables.bnf +++ b/grammar/variables.bnf @@ -66,24 +66,32 @@ input_declaration -> var_init_decl edge_declaration -> var1_list ':' 'BOOL' /[RF]_EDGE/ ; -# NOTE: deviates from the IEC grammar, which also offers `fb_name_decl` here -# (`(var1_list => fb_name_list) ':' function_block_type_name (':=' structure_initialization)?`). -# `function_block_type_name` is, in practice, just `derived_function_block_name -# -> identifier` (since `standard_function_block_name` is still a placeholder) -# - a bare identifier with nothing to distinguish it from `type_name` in -# `structured_var_init_decl` below. "This names a FUNCTION_BLOCK" vs "this +# NOTE: deviates from the IEC grammar's `fb_name_decl` +# (`(var1_list => fb_name_list) ':' function_block_type_name (':=' structure_initialization)?`) +# in one respect: the derived-FB-type half of `function_block_type_name` is +# dropped, since `derived_function_block_name -> identifier` is a bare +# identifier with nothing to distinguish it from `type_name` in +# `structured_var_init_decl` below - "this names a FUNCTION_BLOCK" vs "this # names a TYPE" is only knowable from a symbol table, i.e. semantics/context, -# not this CFG. It's also redundant as written: `type_name_reference`'s -# initializer already includes `structure_initialization`, the only -# initializer `fb_name_decl` offered. Dropped in favor of -# `structured_var_init_decl` for both cases; `fb_name_decl` and its -# `fb_name_list` alias are removed entirely, since this was their only use. +# not this CFG. A derived FB instance like `myTimer : MyCustomFb;` is still +# accepted, just folded into `structured_var_init_decl` like any other +# type reference. What's kept is the standard-FB-type half: since +# `function_block_type_name` is now restricted to `standard_function_block_name` +# (fixed keyword literals - see pou.bnf), it no longer collides with +# `type_name`, so `fb_instantiation` below reinstates it for names like +# `myTimer : TON := (PT := t#100ms);`. The `fb_name_list` alias from the IEC +# grammar isn't reinstated - `var1_list` alone is used, same as the other +# alternatives here. var_init_decl -> var1_init_decl | array_var_init_decl | structured_var_init_decl | string_var_declaration + | fb_instantiation ; +fb_instantiation -> var1_list ':' function_block_type_name (':=' structure_initialization)? + ; + var1_init_decl -> var1_list ':' _var1_init_decl ; From fe9505cea8af0edc95d34b52a19ee6caba10f9a2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Alberto=20Sim=C3=B5es?= Date: Wed, 22 Jul 2026 10:31:09 +0100 Subject: [PATCH 7/7] Add a real-world Structured Text test corpus Replaces tests/literals.txt, which had gone stale (written against an earlier structure of the grammar and no longer matching current rule names/shapes) and had never actually been exercised, since ts-bnf-tool generation was failing on the conflicts fixed in the previous commit. tests/community_samples.txt has six small, complete, hand-written ST declarations pulled from real MIT-licensed open-source projects (not AI-generated) - WengerAG/structured-text-utilities and tkucic/UniTest - covering a spread of rules (enum/struct TYPE, functions with nested calls, VAR_TEMP, IF/ELSE, a function block with FOR/array/struct-field access). tests/SOURCES.md has the license/attribution details. Only one of the six currently parses clean; the rest hit real grammar gaps the corpus surfaced (mandatory `;` after already-block-terminated constructs, enumerated_value's mandatory qualifier, STRING[n] as a general type name, FUNCTION's single non-repeatable, VAR_TEMP-less declarations slot). Documented in tests/FINDINGS.md with examples, root causes, and recommended fixes, held for review before changing the grammar further. --- tests/FINDINGS.md | 219 ++++++++++++++++++++++++++++++++++++ tests/SOURCES.md | 48 ++++++++ tests/community_samples.txt | 133 ++++++++++++++++++++++ tests/literals.txt | 161 -------------------------- 4 files changed, 400 insertions(+), 161 deletions(-) create mode 100644 tests/FINDINGS.md create mode 100644 tests/SOURCES.md create mode 100644 tests/community_samples.txt delete mode 100644 tests/literals.txt diff --git a/tests/FINDINGS.md b/tests/FINDINGS.md new file mode 100644 index 0000000..8f9fd9f --- /dev/null +++ b/tests/FINDINGS.md @@ -0,0 +1,219 @@ +# Findings from testing against real-world ST code + +`community_samples.txt` (see `SOURCES.md` for provenance) surfaced four +grammar gaps beyond the tree-sitter conflict fixes. This documents each one: +the failing example, the root cause, and a recommended fix. No grammar +changes have been made for these yet - this is a decision record to act on. + +## 1. Mandatory `;` after constructs that already end in a closing keyword + +**Fails today:** + +``` +TYPE + FLOAT_ARRAY_STATISTICS : STRUCT + count : UDINT; + END_STRUCT +END_TYPE +``` + +**Root cause:** `data_types.bnf:132` + +``` +data_type_declaration -> 'TYPE' (type_declaration ';')+ 'END_TYPE' +``` + +and `st.bnf:102` + +``` +statement_list -> (statement? ';')+ +``` + +Both require a trailing `;` after *every* element, with no exception for one +that already ends in its own closing keyword (`END_STRUCT`, `END_IF`, +`END_FOR`, `END_WHILE`, `END_CASE`, ...). Two of the six corpus samples hit +this: the struct `TYPE` (`END_STRUCT` directly followed by `END_TYPE`, no +`;`) and `utTestReporter` (`END_IF` directly followed by the next `IF`, no +`;`). Neither is a corpus mistake - both source files consistently omit the +`;` in this position. + +**Recommended fix:** make the separator optional specifically after +block-terminated alternatives, rather than loosening the rule everywhere +(which would also silently accept `42 ;;;; 43` as a valid decimal +literal - not the intent). Concretely, split `statement` into "the +alternatives that end in their own keyword" (`selection_statement`, +`iteration_statement`, `case_statement`, ...) and "the ones that don't" +(`assignment_statement`, `subprogram_control_statement`, ...), and only +require `;` after the latter: + +``` +statement_list -> (_terminated_statement | _self_terminated_statement ';'?)* +``` + +(exact rule split needs a look at every `statement` alternative; sketch only) +Same idea for `type_declaration ';'` in `data_type_declaration` - only +`single_element_type_declaration`/`type_name_reference` (bare `identifier +[:= ...]`, no closing keyword of their own) strictly need the `;`; +`array_type_declaration`/`structure_type_declaration`/`string_type_declaration` +already end in `END_STRUCT` or a length suffix and could make it optional the +same way. + +**Rationale:** this matches both real-world samples without becoming lenient +about the cases where `;` is the *only* thing separating two statements +(e.g. two bare assignments) - dropping it there would be a real ambiguity +risk, not just a style relaxation. + +## 2. `enumerated_value` requires a mandatory `TypeName#` qualifier + +**Fails today:** + +``` +TYPE + BINARY_ENCODING : (BASE64, BASE64_URL); +END_TYPE +``` + +**Root cause:** `data_types.bnf:178` + +``` +enumerated_value -> ((identifier => enumerated_type_name) '#') identifier +``` + +The `identifier '#'` qualifier prefix is not wrapped in `(...)?` - it's +mandatory. That's backwards for the most common case: when *declaring* an +enum's own value list (`enumerated_specification -> '(' enumerated_value +(',' enumerated_value)* ')'`, `data_types.bnf:170`), the values are bare +identifiers - `TypeName#Value` qualification is for *using* an enum value +elsewhere (disambiguating which enum a shared value name belongs to), not for +defining the list in the first place. + +**Recommended fix:** + +``` +enumerated_value -> ((identifier => enumerated_type_name) '#')? identifier + ; +``` + +**Rationale:** every real enum declaration in the corpus (and every IEC +example this reviewer is aware of) writes the value list as bare +identifiers. Making the qualifier optional doesn't lose the ability to parse +`TypeName#Value` where it's meaningful (`enumerated_value` is also reachable +from `_type_name_reference_init`/`array_initial_element` etc., where +qualification matters more), it just stops rejecting the declaration form. + +## 3. `STRING[n]` not accepted as a general type name + +**Fails today:** + +``` +FUNCTION CONCAT3 : STRING[254] + VAR_INPUT + in1 : STRING[254]; + END_VAR + CONCAT3 := in1; +END_FUNCTION +``` + +(fails on the `: STRING[254]` return type specifically; `in1 : STRING[254]` +inside `VAR_INPUT` is fine) + +**Root cause:** `data_types.bnf:23` / `variables.bnf:220` + +``` +elementary_type_name -> numeric_type_name | date_type_name + | bit_string_type_name | /W?STRING/ | 'TIME' + ; + +single_byte_string_spec -> 'STRING' ('[' integer ']')? (':=' single_byte_character_string)? + ; +``` + +The `[length]` suffix only exists on `single_byte_string_spec` / +`double_byte_string_spec`, which are reachable from a `VAR` declaration's +`_var1_init_decl`/`located_var_spec_init`, but *not* from +`_function_type_name` (`pou.bnf:29`, `elementary_type_name | type_name`) or +`simple_specification` (`data_types.bnf:153`, `-> elementary_type_name` +alone). Anywhere a `STRING`/`WSTRING` needs a length and isn't going through +one of the two `*_string_spec` rules, it can't have one. + +**Recommended fix:** thread the optional length onto `elementary_type_name`'s +`/W?STRING/` alternative directly (or a wrapper rule used in its place), +since the type name and its length aren't really separable concepts: + +``` +elementary_type_name -> numeric_type_name | date_type_name + | bit_string_type_name + | /W?STRING/ ('[' integer ']')? + | 'TIME' + ; +``` + +then drop the now-redundant `('[' integer ']')?` from +`single_byte_string_spec`/`double_byte_string_spec`, which would just be +`elementary_type_name (':=' ...)?` at that point - worth checking whether +those two rules collapse into `simple_spec_init` entirely once this lands, +rather than leaving a parallel path. + +**Rationale:** a return type, an input, and a plain variable are all "a type +name" in the same sense; there's no reason the length suffix should only be +reachable from one of the three paths that all eventually mean "this is a +`STRING`". + +## 4. `_function_decls` is single-occurrence and has no `VAR_TEMP` + +**Fails today:** + +``` +FUNCTION STRING_STARTSWITH : BOOL + VAR_INPUT + in1 : STRING[254]; + END_VAR + VAR_TEMP + in1_len : INT; + END_VAR + STRING_STARTSWITH := TRUE; +END_FUNCTION +``` + +**Root cause:** `pou.bnf:19,33` + +``` +function_declaration -> 'FUNCTION' derived_function_name ':' _function_type_name? _function_decls function_body 'END_FUNCTION' + ; + +_function_decls -> io_var_declarations + | function_var_decls + ; +``` + +Two separate problems: (a) `_function_decls` appears exactly once in +`function_declaration` - a function can have only *one* declarations block +total, when `VAR_INPUT`, `VAR_OUTPUT`, `VAR_IN_OUT`, and a temp/local block +are all ordinarily present together; (b) `function_var_decls -> 'VAR' +'CONSTANT'? ...` covers `VAR [CONSTANT] ... END_VAR`, but there's no +alternative covering `VAR_TEMP ... END_VAR` for functions - `temp_var_decls` +(`pou.bnf:109`) exists but is only wired into `other_var_declarations`, which +`_program_decls` (programs) uses, not `_function_decls` (functions). + +**Recommended fix:** + +``` +_function_decls -> (io_var_declarations | function_var_decls | temp_var_decls)* + ; +``` + +**Rationale:** matches how `_program_decls`/`other_var_declarations` +already handle repeated, mixed declaration blocks for `PROGRAM` and +`FUNCTION_BLOCK` - `FUNCTION` was the odd one out. Also brings the grammar in +line with the standard's own `+`/`*` cardinality on `io_var_declarations` in +the formal function-declaration production (a function needs at least one +input in practice, even though nothing here enforces that semantically - see +the existing `FIXME: NOTE 1` comment at `pou.bnf:57`). + +## Suggested order + +(2) and (4) are narrow, low-risk, single-rule changes. (3) touches a +lexical-ish rule used in a few places, worth a `make test` pass after. (1) is +the most involved - it changes cardinality/structure rather than adding an +alternative, and is worth doing last so the corpus can catch any fallout from +the other three first. diff --git a/tests/SOURCES.md b/tests/SOURCES.md new file mode 100644 index 0000000..2fe00fb --- /dev/null +++ b/tests/SOURCES.md @@ -0,0 +1,48 @@ +# Provenance of tests/community_samples.txt + +The test cases in `community_samples.txt` are small, complete excerpts of real, +human-written Structured Text taken from permissively-licensed open-source +repositories (not AI-generated), reformatted only to remove indentation that +came from a vendor-specific module wrapper the original files were nested in +(see below). Each is reproduced under the terms of the source project's MIT +license, which permits copying and redistribution provided the copyright +notice is retained - reproduced here: + +## WengerAG/structured-text-utilities + + - MIT License, +Copyright (c) 2019 Wenger Automation & Engineering AG, Winterthur, Switzerland. + +Used for: "Enumerated type", "Structure type", "Function with nested call +expression", "Function with VAR_TEMP and IF/ELSE", "Function referencing a +derived type and calling another function". + +Source files: `UTILITIES_BYTE.st`, `UTILITIES_MATH.st`, `UTILITIES_STRING.st`, +`UTILITIES_TIME.st`. In the original repository these declarations live inside +a B&R Automation Studio `UNIT ... INTERFACE ... IMPLEMENTATION ... END_UNIT` +module wrapper, which is a vendor-specific packaging construct, not part of +IEC 61131-3 itself; only the leading indentation from that wrapper was +stripped when extracting each declaration as a standalone top-level +`library_element_declaration`. + +## tkucic/UniTest + + - MIT License, Copyright (c) 2021 Toni Kucic. + +Used for: "Function block with FOR loop, array indexing and struct field +access" (`UniTest_br/UniTest/utTestReporter.st`). + +Note: two other files from this repository (`assertEqual_BOOL.st`, a +`FUNCTION`, and `Library_tests/main/main.st`, a `PROGRAM`) were considered but +dropped - in B&R Automation Studio projects the signature/`VAR` block of a POU +is stored in a separate companion file the IDE manages, so on their own these +particular files are implementation-only and don't parse as complete, +standalone declarations. They were not "fixed" with invented `VAR` blocks, +since that would stop being real source. + +## Coverage gap + +No standalone `PROGRAM` or `CONFIGURATION` declaration is included - these +tend to be project-specific and IDE-scaffolded rather than published in +shared, reusable libraries, so they were hard to find as complete, hand-written, +permissively-licensed single-file examples. diff --git a/tests/community_samples.txt b/tests/community_samples.txt new file mode 100644 index 0000000..f8292f8 --- /dev/null +++ b/tests/community_samples.txt @@ -0,0 +1,133 @@ +================== +Enumerated type (WengerAG structured-text-utilities) +================== + +TYPE + BINARY_ENCODING : ( + BASE64, // Base 64 Encoding as specified by RFC 4648 + BASE64_URL // Base 64 URL Encoding as specified by RFC 4648 + ); +END_TYPE + +--- + +================== +Structure type (WengerAG structured-text-utilities) +================== + +TYPE + // Contains statistics information of an array of floating values. + FLOAT_ARRAY_STATISTICS : STRUCT + count : UDINT; // count of values + sum : LREAL; // sum of all values + min_val : LREAL; // minimal value + max_val : LREAL; // maximal value + med_val : LREAL; // median value + avg_val : LREAL; // average value + sum_sqr : LREAL; // sum of squares + std_dev : LREAL; // standard deviation + END_STRUCT +END_TYPE + +--- + +================== +Function with nested call expression (WengerAG structured-text-utilities) +================== + +FUNCTION CONCAT3 : STRING[254] + VAR_INPUT + in1 : STRING[254]; + in2 : STRING[254]; + in3 : STRING[254]; + END_VAR + + CONCAT3 := CONCAT(CONCAT(in1, in2), in3); +END_FUNCTION + +--- + +================== +Function with VAR_TEMP and IF/ELSE (WengerAG structured-text-utilities) +================== + +FUNCTION STRING_STARTSWITH : BOOL + VAR_INPUT + in1 : STRING[254]; + in2 : STRING[254]; + END_VAR + + VAR_TEMP + in1_len : INT; + in2_len : INT; + END_VAR + + in1_len := LEN(in1); + in2_len := LEN(in2); + + IF in1_len < in2_len THEN + STRING_STARTSWITH := FALSE; + ELSE + STRING_STARTSWITH := LEFT(in1, in2_len) = in2; + END_IF; +END_FUNCTION + +--- + +================== +Function referencing a derived type and calling another function (WengerAG structured-text-utilities) +================== + +FUNCTION GET_DAY_OF_WEEK : USINT + VAR_INPUT + in : DATE; + kind : DAY_OF_WEEK_TYPE; // currently, only ISO 8601 is supported + END_VAR + + GET_DAY_OF_WEEK := GET_DAY_OF_WEEK_INTERNAL(DATE_TO_UDINT_INTERNAL(in)); +END_FUNCTION + +--- + +================== +Function block with FOR loop, array indexing and struct field access (tkucic/UniTest) +================== + +FUNCTION_BLOCK utTestReporter + (*Reset totals*) + NrPousUnderTest := 0; + NrOfTests := 0; + NrTestsPassed := 0; + NrTestsFailed := 0; + PassRate := 0; + TestsInProgress := FALSE; + Error := FALSE; + + (*Count totals*) + FOR i:=0 TO Size DO + IF Results[i].Id > 0 THEN + (*Count existing POUs under test*) + NrPousUnderTest := NrPousUnderTest + 1; + (*Count the pass rate. Needs to be divided in the end by the number of total tests*) + PassRate := PassRate + Results[i].PassRate; + (*Count the number of tests*) + NrOfTests := NrOfTests + Results[i].TotalTests; + (*Count the number of tests passed*) + NrTestsPassed := NrTestsPassed + Results[i].TestsPassed; + (*Count the number of tests failed*) + NrTestsFailed := NrTestsFailed + Results[i].TestsFailed; + + (*Indicators*) + IF Results[i].TestsRunning THEN + TestsInProgress := TRUE; + END_IF + IF Results[i].Error <> ut_NO_ERROR THEN + Error := TRUE; + END_IF + END_IF + END_FOR + PassRate := PassRate / NrPousUnderTest; +END_FUNCTION_BLOCK + +--- + diff --git a/tests/literals.txt b/tests/literals.txt deleted file mode 100644 index 0189a73..0000000 --- a/tests/literals.txt +++ /dev/null @@ -1,161 +0,0 @@ -================== -Decimal integer -================== - -42 - ---- - -(literal - (integer - (decimal_digit_string))) - -================== -Typed negative decimal integer -================== - -INT#-42 - ---- - -(literal - (integer - (integer_datatype) - (decimal_digit_string))) - -================== -Binary integer with separators -================== - -2#1010_0110 - ---- - -(literal - (integer - (binary_digit_string))) - -================== -Hexadecimal integer with separators -================== - -16#FF_00 - ---- - -(literal - (integer - (hex_digit_string))) - -================== -Floating point number with exponent -================== - -3.14e5 - ---- - -(literal - (floating_pointer_number - (decimal_digit_string) - (decimal_digit_string) - (decimal_digit_string))) - -================== -Typed negative floating point number -================== - -REAL#-1.0E-2 - ---- - -(literal - (floating_pointer_number - (float_datatype) - (decimal_digit_string) - (decimal_digit_string) - (decimal_digit_string))) - -================== -Duration in days -================== - -T#5d - ---- - -(literal - (time_literal - (duration - (sequence_representation - (decimal_digit_string))))) - -================== -Date -================== - -D#2024-01-15 - ---- - -(literal - (time_literal - (date - (date_information)))) - -================== -Time of day -================== - -TOD#12:30:00.5 - ---- - -(literal - (time_literal - (time_of_day - (time_of_day_information)))) - -================== -Date and time -================== - -DT#2024-01-15-12:30:00 - ---- - -(literal - (time_literal - (date_and_time - (date_information) - (time_of_day_information)))) - -================== -Character string with escape -================== - -'hello $N' - ---- - -(literal - (character_string - (characters) - (characters) - (characters) - (characters) - (characters) - (characters) - (characters))) - -================== -Typed character string -================== - -STRING#'a' - ---- - -(literal - (character_string - (characters)))