Compare commits

53 Commits
Author SHA1 Message Date
0x4261756D 820fc7a82e Remove some unnecessary nesting in the parser 2024-02-28 19:13:17 +01:00
0x4261756D 51390b24d3 More deduplication in the tokenizer 2024-02-28 18:55:39 +01:00
0x4261756D 0c93d45dbd deduplicate some tokenizer code 2024-02-28 18:22:24 +01:00
0x4261756D 28f110e2c4 Further cut down the amount of unnecessary coderegions 2024-02-28 17:41:36 +01:00
0x4261756D 40c744119c Make regions not part of superclass StatNode
Since the majority of subclasses already has a node with the same region this can be omitted
2024-02-28 17:02:23 +01:00
0x4261756D ef333f7d93 Throw on broken escape sequences 2024-02-21 17:43:53 +01:00
0x4261756D e3968034ce Add more tests 2024-02-21 17:38:55 +01:00
0x4261756D 769d18f2b0 Fix SingleQuoteBackslashZ expecting to close with " 2024-02-21 17:38:46 +01:00
0x4261756D 8cbfb8b941 Add unfinished implementation for \u escape sequences
Also add a note for \x escape sequences since they are also broken
2024-02-21 17:38:18 +01:00
0x4261756D ab4be05bf4 Ignore '!' in shebang comment 2024-02-21 17:36:59 +01:00
0x4261756D 637638d889 Implement \x escape sequences 2024-02-21 16:36:13 +01:00
0x4261756D ad3bb57dcc Include failed count in test output 2024-02-21 16:07:57 +01:00
0x4261756D 17a8beb2b7 Don't attempt to parse empty token streams 2024-02-21 16:04:58 +01:00
0x4261756D 3d40351771 Fix being unable to parse files with other line endings 2024-02-21 16:04:35 +01:00
0x4261756D 8b30b34bd1 Fix BigComments and StringWithLongBracket swallowing an ending ']' if ']' occurred somewhere within 2024-02-21 16:04:00 +01:00
0x4261756D 04f5804dff Fix trailing comma handling in ParseFieldlist
This is pretty ugly since it checks for every token type that could follow to make it a valid field but the previous check was flat out wrong
2024-02-21 16:01:58 +01:00
0x4261756D 83b0416c03 Make hex numeral 'x' case-insensitive 2024-02-21 16:00:41 +01:00
0x4261756D 6512439ce3 Fix empty strings not having string data 2024-02-21 15:59:31 +01:00
0x4261756D 43fea86d92 Fix missing increment after parsing dotted funcname 2024-02-21 15:11:45 +01:00
0x4261756D 2b75b7d79f Add missing increment if parlist starts with varargs 2024-02-21 15:11:17 +01:00
0x4261756D ae4a3e9993 Fix suffixexp.argsfirstarg expecting ')' instead of name after ':' 2024-02-21 15:10:37 +01:00
0x4261756D 3a6e024c9b Correctly remove last suffix when turning exp into {member, indexed} var 2024-02-21 15:09:55 +01:00
0x4261756D c7ac2cf091 Fix labels expecting a name instead of '::' to close 2024-02-21 15:08:50 +01:00
0x4261756D 6193441621 Fix if expecting a 'then' as else starter 2024-02-21 15:08:16 +01:00
0x4261756D fcaf1c1570 Fix parser crashing in repeat-until block 2024-02-21 15:07:37 +01:00
0x4261756D eaba371455 Fix '&' being parsed as '=' 2024-02-21 15:07:06 +01:00
0x4261756D 424f381755 Make INumeral json serializable 2024-02-21 15:06:51 +01:00
0x4261756D 25b3dd63c5 Improve Run output 2024-02-21 15:05:32 +01:00
0x4261756D d1b855144e Make parser nodes json serializable 2024-02-21 15:05:06 +01:00
0x4261756D 41ab249353 Rename tokenizerTests 2024-02-21 15:03:42 +01:00
0x4261756D d23a5cd70b Test parsing 2024-02-21 13:49:23 +01:00
0x4261756D 27f2917670 Implement missing parser methods 2024-02-21 13:49:04 +01:00
0x4261756D b7f725afee Fix crash if a block has no stats 2024-02-21 13:48:43 +01:00
0x4261756D dd813aa624 Explicitly pass values in RetstatNode constructor 2024-02-21 13:48:02 +01:00
0x4261756D 1679fe8e6e Fix Numeral and LiteralString ExpNodes requiring two regions 2024-02-21 13:47:38 +01:00
0x4261756D ed50d40c1c Fix SuffixexpNodes subclasses inheriting from the wrong base 2024-02-21 13:47:02 +01:00
0x4261756D bb954e99d5 Don't use implicit usings 2024-02-21 13:45:54 +01:00
0x4261756D 02aab2e590 Start working on the parser 2024-02-01 02:20:22 +01:00
0x4261756D 049a96191c Make testing recursive 2024-01-29 18:24:24 +01:00
0x4261756D 39521fbb19 Add float lexing 2024-01-29 18:24:08 +01:00
0x4261756D ec312b3132 Add shebang parsing (read: ignore shebang) 2024-01-29 18:23:20 +01:00
0x4261756D 1646a79055 better command line handling + big formatting stuff 2024-01-29 14:39:22 +01:00
0x4261756D 34cb88582d Port tokenizer to C#
Another language change, another unrefactored (but already better tested) tokenizer
2024-01-16 02:59:51 +01:00
0x4261756D 7889f4c27a Fix last token not getting tokenized 2024-01-15 21:28:40 +01:00
0x4261756D 23269baa0b Make numbers in basic test distinct 2023-11-28 04:06:51 +01:00
0x4261756D d0357f0a3a Fix numbers being tokenized as names 2023-11-28 04:04:57 +01:00
0x4261756D a824823786 Add more directories to .gitignore 2023-11-24 03:36:47 +01:00
0x4261756D c8cbf4659a Add test for table.get 2023-11-24 03:35:55 +01:00
0x4261756D 5dc1b9d50b Make tables work and add rawEquals methods
Since hashtables are a hassle with strings and complex datastructures
like the table they are an arraylist for now.

Also implement rawEquals for numerals and values to make table.get work
2023-11-24 03:22:15 +01:00
0x4261756D cfc288e279 More treewalker stuff 2023-11-17 01:37:29 +01:00
0x4261756D cdfa8d3f90 Implement ast dumping and start treewalker 2023-10-08 21:40:44 +02:00
0x4261756D b00a99ab6a Implement parsing and token locations 2023-09-21 18:30:50 +02:00
0x4261756D 721383a043 Add tokenizer 2023-09-15 11:07:50 +02:00
22 changed files with 4873 additions and 2966 deletions
+101
View File
@@ -0,0 +1,101 @@
root = true
[*]
indent_size = 4
indent_style = tab
trim_trailing_whitespace = true
end_of_line = lf
insert_final_newline = true
charset = utf-8
[*.cs]
csharp_space_after_keywords_in_control_flow_statements = false
csharp_indent_case_contents_when_block = false
csharp_style_unused_value_expression_preference = unused_local_variable
csharp_prefer_braces = true
csharp_prefer_static_local_function = true
dotnet_style_prefer_foreach_explicit_cast_in_source = always
dotnet_style_prefer_collection_expression = when_types_loosely_match
dotnet_diagnostic.IDE0001.severity = warning
dotnet_diagnostic.IDE0002.severity = warning
dotnet_diagnostic.IDE0004.severity = warning
dotnet_diagnostic.IDE0005.severity = warning
dotnet_diagnostic.IDE0011.severity = warning
dotnet_diagnostic.IDE0020.severity = warning
dotnet_diagnostic.IDE0028.severity = warning
dotnet_diagnostic.IDE0029.severity = warning
dotnet_diagnostic.IDE0030.severity = warning
dotnet_diagnostic.IDE0031.severity = warning
dotnet_diagnostic.IDE0035.severity = error
dotnet_diagnostic.IDE0038.severity = error
dotnet_diagnostic.IDE0041.severity = warning
dotnet_diagnostic.IDE0042.severity = warning
dotnet_diagnostic.IDE0051.severity = warning
dotnet_diagnostic.IDE0052.severity = warning
dotnet_diagnostic.IDE0054.severity = warning
dotnet_diagnostic.IDE0050.severity = warning
dotnet_diagnostic.IDE0056.severity = warning
dotnet_diagnostic.IDE0057.severity = warning
dotnet_diagnostic.IDE0058.severity = warning
dotnet_diagnostic.IDE0060.severity = warning
dotnet_diagnostic.IDE0062.severity = warning
dotnet_diagnostic.IDE0066.severity = warning
dotnet_diagnostic.IDE0071.severity = warning
dotnet_diagnostic.IDE0074.severity = warning
dotnet_diagnostic.IDE0075.severity = warning
dotnet_diagnostic.IDE0078.severity = warning
dotnet_diagnostic.IDE0080.severity = warning
dotnet_diagnostic.IDE0083.severity = warning
dotnet_diagnostic.IDE0090.severity = warning
dotnet_diagnostic.IDE0100.severity = warning
dotnet_diagnostic.IDE0150.severity = warning
dotnet_diagnostic.IDE0180.severity = warning
dotnet_diagnostic.IDE0200.severity = warning
dotnet_diagnostic.IDE0220.severity = warning
dotnet_diagnostic.IDE0260.severity = warning
dotnet_diagnostic.IDE0270.severity = warning
dotnet_diagnostic.IDE0300.severity = warning
dotnet_diagnostic.IDE0301.severity = warning
dotnet_diagnostic.IDE0302.severity = warning
dotnet_diagnostic.IDE0303.severity = warning
dotnet_diagnostic.IDE0304.severity = warning
dotnet_diagnostic.IDE0305.severity = warning
dotnet_diagnostic.CA1508.severity = warning
dotnet_diagnostic.CA1514.severity = warning
dotnet_diagnostic.CA1515.severity = warning
dotnet_diagnostic.CA1801.severity = warning
dotnet_diagnostic.CA1802.severity = warning
dotnet_diagnostic.CA1805.severity = warning
dotnet_diagnostic.CA1806.severity = warning
dotnet_diagnostic.CA1810.severity = warning
dotnet_diagnostic.CA1814.severity = warning
dotnet_diagnostic.CA1820.severity = warning
dotnet_diagnostic.CA1822.severity = warning
dotnet_diagnostic.CA1823.severity = warning
dotnet_diagnostic.CA1825.severity = warning
dotnet_diagnostic.CA1826.severity = error
dotnet_diagnostic.CA1827.severity = error
dotnet_diagnostic.CA1829.severity = error
dotnet_diagnostic.CA1830.severity = error
dotnet_diagnostic.CA1833.severity = warning
dotnet_diagnostic.CA1834.severity = warning
dotnet_diagnostic.CA1835.severity = warning
dotnet_diagnostic.CA1836.severity = warning
dotnet_diagnostic.CA1841.severity = error
dotnet_diagnostic.CA1845.severity = warning
dotnet_diagnostic.CA1846.severity = warning
dotnet_diagnostic.CA1847.severity = warning
dotnet_diagnostic.CA1849.severity = warning
dotnet_diagnostic.CA1850.severity = warning
dotnet_diagnostic.CA1851.severity = warning
dotnet_diagnostic.CA1853.severity = warning
dotnet_diagnostic.CA1859.severity = warning
dotnet_diagnostic.CA1860.severity = warning
dotnet_diagnostic.CA1861.severity = warning
dotnet_diagnostic.CA1864.severity = warning
dotnet_diagnostic.CA1865.severity = warning
dotnet_diagnostic.CA1866.severity = warning
dotnet_diagnostic.CA1867.severity = warning
dotnet_diagnostic.CA2007.severity = warning
dotnet_diagnostic.CA2011.severity = error
dotnet_diagnostic.CA2248.severity = error
+2 -17
View File
@@ -1,17 +1,2 @@
# ---> Rust
# Generated by Cargo
# will have compiled files and executables
debug/
target/
.vscode/
# Remove Cargo.lock from gitignore if creating an executable, leave it for libraries
# More information here https://doc.rust-lang.org/cargo/guide/cargo-toml-vs-cargo-lock.html
Cargo.lock
# These are backup files generated by rustfmt
**/*.rs.bk
# MSVC Windows builds of rustc generate these, which store debugging information
*.pdb
bin/
obj/
+26
View File
@@ -0,0 +1,26 @@
{
"version": "0.2.0",
"configurations": [
{
// Use IntelliSense to find out which attributes exist for C# debugging
// Use hover for the description of the existing attributes
// For further information visit https://github.com/dotnet/vscode-csharp/blob/main/debugger-launchjson.md
"name": ".NET Core Launch (console)",
"type": "coreclr",
"request": "launch",
"preLaunchTask": "build",
// If you have changed target frameworks, make sure to update the program path.
"program": "${workspaceFolder}/bin/Debug/net8.0/luaaaaah.dll",
"args": ["test/simpleString.lua"],
"cwd": "${workspaceFolder}",
// For more information about the 'console' field, see https://aka.ms/VSCode-CS-LaunchJson-Console
"console": "internalConsole",
"stopAtEntry": false
},
{
"name": ".NET Core Attach",
"type": "coreclr",
"request": "attach"
}
]
}
+41
View File
@@ -0,0 +1,41 @@
{
"version": "2.0.0",
"tasks": [
{
"label": "build",
"command": "dotnet",
"type": "process",
"args": [
"build",
"${workspaceFolder}/luaaaaah.csproj",
"/property:GenerateFullPaths=true",
"/consoleloggerparameters:NoSummary;ForceNoAlign"
],
"problemMatcher": "$msCompile"
},
{
"label": "publish",
"command": "dotnet",
"type": "process",
"args": [
"publish",
"${workspaceFolder}/luaaaaah.csproj",
"/property:GenerateFullPaths=true",
"/consoleloggerparameters:NoSummary;ForceNoAlign"
],
"problemMatcher": "$msCompile"
},
{
"label": "watch",
"command": "dotnet",
"type": "process",
"args": [
"watch",
"run",
"--project",
"${workspaceFolder}/luaaaaah.csproj"
],
"problemMatcher": "$msCompile"
}
]
}
Generated
-7
View File
@@ -1,7 +0,0 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 3
[[package]]
name = "luaaaaah"
version = "0.1.0"
-8
View File
@@ -1,8 +0,0 @@
[package]
name = "luaaaaah"
version = "0.1.0"
edition = "2021"
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
[dependencies]
-9
View File
@@ -1,9 +0,0 @@
MIT License
Copyright (c) 2023 0x4261756D
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+71
View File
@@ -0,0 +1,71 @@
using System.Text.Json.Serialization;
namespace luaaaaah;
[JsonDerivedType(typeof(Integer), typeDiscriminator: "int")]
[JsonDerivedType(typeof(Float), typeDiscriminator: "float")]
public interface INumeral
{
public class Integer(int value) : INumeral
{
public int value = value;
public bool RawEqual(INumeral other)
{
if(other is Integer integer)
{
return integer.value == value;
}
// TODO: Check if this is actually doing what is expected
return ((Float)other).value == value;
}
public override string ToString()
{
return $"Numeral Integer {value}";
}
}
public class Float(float value) : INumeral
{
public float value = value;
public bool RawEqual(INumeral other)
{
if(other is Float float_val)
{
return float_val.value == value;
}
// TODO: Check if this is actually doing what is expected
return ((Integer)other).value == value;
}
public override string ToString()
{
return $"Numeral Float {value}";
}
}
public bool RawEqual(INumeral other);
}
class CodeRegion(CodeLocation start, CodeLocation end)
{
public CodeLocation start = start;
public CodeLocation end = end;
public override string ToString()
{
return $"{start}-{end}";
}
}
class CodeLocation(int line, int col)
{
public int line = line;
public int col = col;
public CodeLocation(CodeLocation other) : this(line: other.line, col: other.col) { }
public override string ToString()
{
return $"{line + 1}:{col + 1}";
}
}
+1639
View File
@@ -0,0 +1,1639 @@
using System;
using System.Collections.Generic;
using System.Text.Json.Serialization;
namespace luaaaaah;
internal class Parser
{
public class ChunkNode(BlockNode block, CodeRegion startRegion, CodeRegion endRegion)
{
public BlockNode block = block;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class BlockNode(List<StatNode> stats, CodeRegion startRegion, CodeRegion endRegion)
{
public List<StatNode> stats = stats;
public RetstatNode? retstat;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
[JsonDerivedType(typeof(Semicolon), typeDiscriminator: "st Semicolon")]
[JsonDerivedType(typeof(Assignment), typeDiscriminator: "st Assignment")]
[JsonDerivedType(typeof(Functioncall), typeDiscriminator: "st Functioncall")]
[JsonDerivedType(typeof(Label), typeDiscriminator: "st Label")]
[JsonDerivedType(typeof(Break), typeDiscriminator: "st Break")]
[JsonDerivedType(typeof(Goto), typeDiscriminator: "st Goto")]
[JsonDerivedType(typeof(Do), typeDiscriminator: "st Do")]
[JsonDerivedType(typeof(While), typeDiscriminator: "st While")]
[JsonDerivedType(typeof(Repeat), typeDiscriminator: "st Repeat")]
[JsonDerivedType(typeof(If), typeDiscriminator: "st If")]
[JsonDerivedType(typeof(ForNumerical), typeDiscriminator: "st ForNum")]
[JsonDerivedType(typeof(ForGeneric), typeDiscriminator: "st ForGen")]
[JsonDerivedType(typeof(Function), typeDiscriminator: "st Function")]
[JsonDerivedType(typeof(LocalFunction), typeDiscriminator: "st LocalFunction")]
[JsonDerivedType(typeof(Local), typeDiscriminator: "st Local")]
public abstract class StatNode
{
public class Semicolon(CodeRegion region) : StatNode
{
public CodeRegion region = region;
}
public class Assignment(VarlistNode lhs, ExplistNode rhs, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public VarlistNode lhs = lhs;
public ExplistNode rhs = rhs;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class Functioncall(FunctioncallNode node) : StatNode
{
public FunctioncallNode node = node;
}
public class Label(string label, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public CodeRegion startRegion = startRegion;
public CodeRegion endRegion = endRegion;
public string label = label;
}
public class Break(CodeRegion region) : StatNode
{
public CodeRegion region = region;
}
public class Goto(string label, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public CodeRegion startRegion = startRegion;
public CodeRegion endRegion = endRegion;
public string label = label;
}
public class Do(BlockNode node, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public CodeRegion startRegion = startRegion;
public CodeRegion endRegion = endRegion;
public BlockNode node = node;
}
public class While(ExpNode condition, BlockNode body, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public ExpNode condition = condition;
public BlockNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class Repeat(ExpNode condition, BlockNode body, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public ExpNode condition = condition;
public BlockNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class If(ExpNode condition, BlockNode body, List<ElseifNode> elseifs, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public ExpNode condition = condition;
public BlockNode body = body;
public List<ElseifNode> elseifs = elseifs;
public BlockNode? else_;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class ForNumerical(string variable, ExpNode start, ExpNode end, ExpNode? change, BlockNode body, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public string variable = variable;
public ExpNode start = start;
public ExpNode end = end;
public ExpNode? change = change;
public BlockNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class ForGeneric(List<string> vars, ExplistNode exps, BlockNode body, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public List<string> vars = vars;
public ExplistNode exps = exps;
public BlockNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class Function(FunctionNode node) : StatNode
{
public FunctionNode node = node;
}
public class LocalFunction(string name, FuncbodyNode body, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public string name = name;
public FuncbodyNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class Local(AttnamelistNode attnames, ExplistNode? values, CodeRegion startRegion, CodeRegion endRegion) : StatNode
{
public AttnamelistNode attnames = attnames;
public ExplistNode? values = values;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
}
public class RetstatNode(ExplistNode? values, CodeRegion startRegion, CodeRegion endRegion)
{
public ExplistNode? values = values;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class FunctioncallNode(SuffixexpNode function, string? objectArg, ArgsNode args)
{
public SuffixexpNode function = function;
public string? objectArg = objectArg;
public ArgsNode args = args;
public CodeRegion startRegion = function.startRegion, endRegion = function.endRegion;
}
public class FunctionNode(FuncnameNode name, FuncbodyNode body, CodeRegion startRegion, CodeRegion endRegion)
{
public FuncnameNode name = name;
public FuncbodyNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class ExplistNode(List<ExpNode> exps, CodeRegion startRegion, CodeRegion endRegion)
{
public List<ExpNode> exps = exps;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class VarlistNode(List<VarNode> vars, CodeRegion startRegion, CodeRegion endRegion)
{
public List<VarNode> vars = vars;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
[JsonDerivedType(typeof(Normal), typeDiscriminator: "s Normal")]
[JsonDerivedType(typeof(Functioncall), typeDiscriminator: "s Functioncall")]
public abstract class SuffixexpNode(CodeRegion startRegion, CodeRegion endRegion)
{
public CodeRegion startRegion = startRegion, endRegion = endRegion;
public class Normal(SuffixexpFirstPart firstPart, List<SuffixexpSuffix> suffixes, CodeRegion startRegion, CodeRegion endRegion) : SuffixexpNode(startRegion, endRegion)
{
public SuffixexpFirstPart firstPart = firstPart;
public List<SuffixexpSuffix> suffixes = suffixes;
}
public class Functioncall(FunctioncallNode node) : SuffixexpNode(node.startRegion, node.endRegion)
{
public FunctioncallNode node = node;
}
}
[JsonDerivedType(typeof(Bracketed), typeDiscriminator: "a Bracketed")]
[JsonDerivedType(typeof(Tableconstructor), typeDiscriminator: "a Tableconstructor")]
[JsonDerivedType(typeof(Literal), typeDiscriminator: "a Literal")]
public abstract class ArgsNode(CodeRegion startRegion, CodeRegion endRegion)
{
public CodeRegion startRegion = startRegion, endRegion = endRegion;
public class Bracketed(ExplistNode? node, CodeRegion startRegion, CodeRegion endRegion) : ArgsNode(startRegion, endRegion)
{
public ExplistNode? node = node;
}
public class Tableconstructor(TableconstructorNode node, CodeRegion startRegion, CodeRegion endRegion) : ArgsNode(startRegion, endRegion)
{
public TableconstructorNode node = node;
}
public class Literal(string name, CodeRegion startRegion, CodeRegion endRegion) : ArgsNode(startRegion, endRegion)
{
public string name = name;
}
}
[JsonDerivedType(typeof(Nil), typeDiscriminator: "e Nil")]
[JsonDerivedType(typeof(False), typeDiscriminator: "e True")]
[JsonDerivedType(typeof(True), typeDiscriminator: "e False")]
[JsonDerivedType(typeof(Numeral), typeDiscriminator: "e Numeral")]
[JsonDerivedType(typeof(LiteralString), typeDiscriminator: "e Literal")]
[JsonDerivedType(typeof(Varargs), typeDiscriminator: "e Varargs")]
[JsonDerivedType(typeof(Functiondef), typeDiscriminator: "e Functiondef")]
[JsonDerivedType(typeof(Suffixexp), typeDiscriminator: "e Suffixexp")]
[JsonDerivedType(typeof(Tableconstructor), typeDiscriminator: "e Tableconstructor")]
[JsonDerivedType(typeof(Unop), typeDiscriminator: "e Unop")]
[JsonDerivedType(typeof(Binop), typeDiscriminator: "e Binop")]
public abstract class ExpNode
{
public class Nil(CodeRegion region) : ExpNode
{
public CodeRegion region = region;
}
public class False(CodeRegion region) : ExpNode
{
public CodeRegion region = region;
}
public class True(CodeRegion region) : ExpNode
{
public CodeRegion region = region;
}
public class Numeral(INumeral value, CodeRegion region) : ExpNode
{
public CodeRegion region = region;
public INumeral value = value;
}
public class LiteralString(string value, CodeRegion region) : ExpNode
{
public CodeRegion region = region;
public string value = value;
}
public class Varargs(CodeRegion region) : ExpNode
{
public CodeRegion region = region;
}
public class Functiondef(FuncbodyNode node, CodeRegion startRegion, CodeRegion endRegion) : ExpNode
{
public CodeRegion startRegion = startRegion;
public CodeRegion endRegion = endRegion;
public FuncbodyNode node = node;
}
public class Suffixexp(SuffixexpNode node) : ExpNode
{
public SuffixexpNode node = node;
}
public class Tableconstructor(TableconstructorNode node) : ExpNode
{
public TableconstructorNode node = node;
}
public class Unop(UnopNode node) : ExpNode
{
public UnopNode node = node;
}
public class Binop(BinopNode node) : ExpNode
{
public BinopNode node = node;
}
}
public class ElseifNode(ExpNode condition, BlockNode body, CodeRegion startRegion, CodeRegion endRegion)
{
public ExpNode condition = condition;
public BlockNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class FuncnameNode(string name, List<string> dottedNames, string? firstArg, CodeRegion startRegion, CodeRegion endRegion)
{
public string name = name;
public List<string> dottedNames = dottedNames;
public string? firstArg = firstArg;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class FuncbodyNode(ParlistNode? pars, BlockNode body, CodeRegion startRegion, CodeRegion endRegion)
{
public ParlistNode? pars = pars;
public BlockNode body = body;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class AttnamelistNode(List<AttnameNode> attnames, CodeRegion startRegion, CodeRegion endRegion)
{
public List<AttnameNode> attnames = attnames;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
[JsonDerivedType(typeof(Name), typeDiscriminator: "v Name")]
[JsonDerivedType(typeof(Indexed), typeDiscriminator: "v Indexed")]
[JsonDerivedType(typeof(Member), typeDiscriminator: "v Member")]
public abstract class VarNode(CodeRegion startRegion, CodeRegion endRegion)
{
public CodeRegion startRegion = startRegion, endRegion = endRegion;
public class Name(string name, CodeRegion startRegion, CodeRegion endRegion) : VarNode(startRegion, endRegion)
{
public string name = name;
}
public class Indexed(IndexedVarNode node, CodeRegion startRegion, CodeRegion endRegion) : VarNode(startRegion, endRegion)
{
public IndexedVarNode node = node;
}
public class Member(MemberVarNode node, CodeRegion startRegion, CodeRegion endRegion) : VarNode(startRegion, endRegion)
{
public MemberVarNode node = node;
}
}
public class TableconstructorNode(FieldlistNode? exps, CodeRegion startRegion, CodeRegion endRegion)
{
public FieldlistNode? exps = exps;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class UnopNode(UnopType type, ExpNode exp, CodeRegion startRegion, CodeRegion endRegion)
{
public UnopType type = type;
public ExpNode exp = exp;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public enum UnopType
{
Minus, LogicalNot, Length, BinaryNot,
}
public class BinopNode(ExpNode lhs, BinopType type, ExpNode rhs, CodeRegion startRegion, CodeRegion endRegion)
{
public ExpNode lhs = lhs;
public BinopType type = type;
public ExpNode rhs = rhs;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public enum BinopType
{
LogicalOr,
LogicalAnd,
Lt, Gt, LtEquals, GtEquals, NotEquals, Equals,
BinaryOr,
BinaryNot,
BinaryAnd,
Shl, Shr,
Concat,
Add, Sub,
Mul, Div, IntDiv, Mod,
Exp,
}
public class ParlistNode(List<string> names, bool hasVarargs, CodeRegion startRegion, CodeRegion endRegion)
{
public List<string> names = names;
public bool hasVarargs = hasVarargs;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class AttnameNode(string name, string? attribute, CodeRegion startRegion, CodeRegion endRegion)
{
public string name = name;
public string? attribute = attribute;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class IndexedVarNode(SuffixexpNode value, ExpNode index, CodeRegion startRegion, CodeRegion endRegion)
{
public SuffixexpNode value = value;
public ExpNode index = index;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class MemberVarNode(SuffixexpNode value, string name, CodeRegion startRegion, CodeRegion endRegion)
{
public SuffixexpNode value = value;
public string name = name;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
[JsonDerivedType(typeof(Name), typeDiscriminator: "sfp Name")]
[JsonDerivedType(typeof(BracketedExp), typeDiscriminator: "sfp BracketedExp")]
public abstract class SuffixexpFirstPart(CodeRegion startRegion, CodeRegion endRegion)
{
public CodeRegion startRegion = startRegion, endRegion = endRegion;
public class Name(string name, CodeRegion startRegion, CodeRegion endRegion) : SuffixexpFirstPart(startRegion, endRegion)
{
public string name = name;
}
public class BracketedExp(ExpNode node, CodeRegion startRegion, CodeRegion endRegion) : SuffixexpFirstPart(startRegion, endRegion)
{
public ExpNode node = node;
}
}
[JsonDerivedType(typeof(Dot))]
[JsonDerivedType(typeof(Indexed))]
[JsonDerivedType(typeof(Args))]
[JsonDerivedType(typeof(ArgsFirstArg))]
public abstract class SuffixexpSuffix(CodeRegion startRegion, CodeRegion endRegion)
{
public CodeRegion startRegion = startRegion, endRegion = endRegion;
public class Dot(string name, CodeRegion startRegion, CodeRegion endRegion) : SuffixexpSuffix(startRegion, endRegion)
{
public string name = name;
}
public class Indexed(ExpNode node, CodeRegion startRegion, CodeRegion endRegion) : SuffixexpSuffix(startRegion, endRegion)
{
public ExpNode node = node;
}
public class Args(ArgsNode node, CodeRegion startRegion, CodeRegion endRegion) : SuffixexpSuffix(startRegion, endRegion)
{
public ArgsNode node = node;
}
public class ArgsFirstArg(ArgsFirstArgNode node) : SuffixexpSuffix(node.startRegion, node.endRegion)
{
public ArgsFirstArgNode node = node;
}
}
public class FieldlistNode(List<FieldNode> exps, CodeRegion startRegion, CodeRegion endRegion)
{
public List<FieldNode> exps = exps;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class ArgsFirstArgNode(string name, ArgsNode rest, CodeRegion startRegion, CodeRegion endRegion)
{
public string name = name;
public ArgsNode rest = rest;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
[JsonDerivedType(typeof(IndexedAssignment))]
[JsonDerivedType(typeof(Assignment))]
[JsonDerivedType(typeof(Exp))]
public abstract class FieldNode(CodeRegion startRegion, CodeRegion endRegion)
{
public CodeRegion startRegion = startRegion, endRegion = endRegion;
public class IndexedAssignment(IndexedAssignmentNode node) : FieldNode(node.startRegion, node.endRegion)
{
public IndexedAssignmentNode node = node;
}
public class Assignment(FieldAssignmentNode node) : FieldNode(node.startRegion, node.endRegion)
{
public FieldAssignmentNode node = node;
}
public class Exp(ExpNode node, CodeRegion startRegion, CodeRegion endRegion) : FieldNode(startRegion, endRegion)
{
public ExpNode node = node;
}
}
public class IndexedAssignmentNode(ExpNode index, ExpNode rhs, CodeRegion startRegion, CodeRegion endRegion)
{
public ExpNode index = index;
public ExpNode rhs = rhs;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public class FieldAssignmentNode(string lhs, ExpNode rhs, CodeRegion startRegion, CodeRegion endRegion)
{
public string lhs = lhs;
public ExpNode rhs = rhs;
public CodeRegion startRegion = startRegion, endRegion = endRegion;
}
public int index;
public ChunkNode Parse(Token[] tokens)
{
return ParseChunk(tokens);
}
public ChunkNode ParseChunk(Token[] tokens)
{
BlockNode body = ParseBlock(tokens);
return new ChunkNode(block: body, startRegion: body.startRegion, endRegion: body.endRegion);
}
public BlockNode ParseBlock(Token[] tokens)
{
CodeRegion startRegion = tokens[index].region;
List<StatNode> stats = [];
while(index < tokens.Length &&
tokens[index].type != TokenType.Return &&
tokens[index].type != TokenType.End &&
tokens[index].type != TokenType.Elseif &&
tokens[index].type != TokenType.Else &&
tokens[index].type != TokenType.Until)
{
stats.Add(ParseStat(tokens));
}
BlockNode ret = new(stats: stats, startRegion: startRegion, endRegion: (stats.Count == 0 && index > 0) ? startRegion : tokens[index - 1].region);
if(index < tokens.Length && tokens[index].type == TokenType.Return)
{
ret.retstat = ParseRetstat(tokens);
}
return ret;
}
public StatNode ParseStat(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}");
}
CodeRegion startRegion = tokens[index].region;
switch(tokens[index].type)
{
case TokenType.Semicolon:
{
index += 1;
return new StatNode.Semicolon(startRegion);
}
case TokenType.Break:
{
index += 1;
return new StatNode.Break(startRegion);
}
case TokenType.Goto:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name for goto at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name for goto, got {tokens[index].type}");
}
StatNode.Goto ret = new(label: ((Token.StringData)tokens[index].data!).data, startRegion: startRegion, endRegion: tokens[index].region);
index += 1;
return ret;
}
case TokenType.Do:
{
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected end for `do` at {startRegion}");
}
if(tokens[index].type != TokenType.End)
{
throw new Exception($"{tokens[index].region}: Expected `end` to close `do` at {startRegion}, got {tokens[index].type}");
}
StatNode.Do ret = new(node: body, startRegion: startRegion, endRegion: tokens[index].region);
index += 1;
return ret;
}
case TokenType.While:
{
index += 1;
ExpNode condition = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `do` after condition of while loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Do)
{
throw new Exception($"{tokens[index].region}: Expected `do` after condition of while starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` after body of while loop starting at {startRegion}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new StatNode.While(condition: condition, body: body, startRegion: startRegion, endRegion: endRegion);
}
case TokenType.Repeat:
{
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `until` after body of until loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Until)
{
throw new Exception($"{tokens[index].region}: Expected `until` after block of until loop starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
ExpNode conditon = ParseExp(tokens);
return new StatNode.Repeat(condition: conditon, body: body, startRegion: startRegion, endRegion: tokens[index - 1].region);
}
case TokenType.If:
{
index += 1;
ExpNode condition = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `then` after condition of if starting at {startRegion}");
}
if(tokens[index].type != TokenType.Then)
{
throw new Exception($"{tokens[index].region}: Expected `then` after condition of if starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` after body of if starting at {startRegion}");
}
List<ElseifNode> elseifs = [];
while(tokens[index].type == TokenType.Elseif)
{
CodeRegion elseifStartRegion = tokens[index].region;
index += 1;
ExpNode elseifCondition = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `then` after condition of elseif starting at {elseifStartRegion}");
}
if(tokens[index].type != TokenType.Then)
{
throw new Exception($"{tokens[index].region}: Expected `then` after condition of elseif starting at {elseifStartRegion}, got {tokens[index].type}");
}
index += 1;
BlockNode elseifBody = ParseBlock(tokens);
elseifs.Add(new(condition: elseifCondition, body: elseifBody, startRegion: elseifStartRegion, endRegion: elseifBody.endRegion));
}
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` after else-ifs of if starting at {startRegion}");
}
StatNode.If ret = new(condition: condition, body: body, elseifs: elseifs, startRegion: startRegion, endRegion: tokens[index - 1].region);
if(tokens[index].type == TokenType.Else)
{
index += 1;
ret.else_ = ParseBlock(tokens);
}
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` to close if starting at {startRegion}");
}
if(tokens[index].type != TokenType.End)
{
throw new Exception($"{tokens[index].region}: Expected `end` to close if starting at {startRegion}, got {tokens[index].type}");
}
ret.endRegion = tokens[index].region;
index += 1;
return ret;
}
case TokenType.For:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name after for at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name after for at {startRegion}, got {tokens[index].type}");
}
string variable = ((Token.StringData)tokens[index].data!).data;
index += 1;
switch(tokens[index].type)
{
case TokenType.Equals:
{
index += 1;
ExpNode start = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `,` after start value of numerical for loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Comma)
{
throw new Exception($"{tokens[index].region}: Expected `,` after start value of for loop starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
ExpNode end = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `do` or `,` after end value of numerical for loop starting at {startRegion}");
}
ExpNode? change = null;
if(tokens[index].type == TokenType.Comma)
{
index += 1;
change = ParseExp(tokens);
}
if(index >= tokens.Length)
{
string t = (change == null) ? "`do` or `,` after end value" : "`do` after change value";
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected {t} of numerical for loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Do)
{
string t = (change == null) ? "`do` or `,` after end value" : "`do` after change value";
throw new Exception($"{tokens[index].region}: Expected {t} of numerical for loop starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` to close numerical for loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.End)
{
throw new Exception($"{tokens[index].region}: Expected `end` to close numerical for loop starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new StatNode.ForNumerical(variable: variable, start: start, end: end, change: change, body: body, startRegion: startRegion, endRegion: endRegion);
}
case TokenType.Comma:
{
List<string> names = [variable];
while(tokens[index].type == TokenType.Comma)
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected another name in name list of for-in loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected another name in name list of for-in loop starting at {startRegion}, got {tokens[index].type}");
}
names.Add(((Token.StringData)tokens[index].data!).data);
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `,` or `in` in for-in loop starting at {startRegion}");
}
}
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `in` after name list of for-in loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.In)
{
throw new Exception($"{tokens[index].region}: Expected `in` after name list of for-in loop starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
ExplistNode exps = ParseExplist(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `do` after exp list of for-in loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Do)
{
throw new Exception($"{tokens[index].region}: Expected `do` after exp list of for-in loop starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` to close for-in loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.End)
{
throw new Exception($"{tokens[index].region}: Expected `end` to close for-in loop starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new StatNode.ForGeneric(vars: names, exps: exps, body: body, startRegion: startRegion, endRegion: endRegion);
}
case TokenType.In:
{
index += 1;
ExplistNode exps = ParseExplist(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `do` after exp list of for-in loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.Do)
{
throw new Exception($"{tokens[index].region}: Expected `do` after exp list of for-in loop starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` to close for-in loop starting at {startRegion}");
}
if(tokens[index].type != TokenType.End)
{
throw new Exception($"{tokens[index].region}: Expected `end` to close for-in loop starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new StatNode.ForGeneric(vars: [variable], exps: exps, body: body, startRegion: startRegion, endRegion: endRegion);
}
default:
{
throw new Exception($"{tokens[index].type}: Expected either `=`, `,` or `in` after first name of for loop {startRegion}, got {tokens[index].type}");
}
}
}
case TokenType.Function:
{
index += 1;
FuncnameNode name = ParseFuncname(tokens);
FuncbodyNode body = ParseFuncbody(tokens);
return new StatNode.Function(new(name: name, body: body, startRegion: startRegion, endRegion: body.endRegion));
}
case TokenType.Local:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length} after `local` at {startRegion}");
}
if(tokens[index].type == TokenType.Function)
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name of local function starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name of local function starting at {startRegion}, got {tokens[index].type}");
}
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
FuncbodyNode body = ParseFuncbody(tokens);
return new StatNode.LocalFunction(name: name, body: body, startRegion: startRegion, endRegion: body.endRegion);
}
else
{
AttnamelistNode attnames = ParseAttnamelist(tokens);
StatNode.Local ret = new(attnames: attnames, values: null, startRegion: startRegion, endRegion: attnames.endRegion);
if(index < tokens.Length && tokens[index].type == TokenType.Equals)
{
index += 1;
ret.values = ParseExplist(tokens);
ret.endRegion = ret.values.endRegion;
}
return ret;
}
}
case TokenType.ColonColon:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name of label starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name of label starting at {startRegion}, got {tokens[index].type}");
}
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `::` after label name starting at {startRegion}");
}
if(tokens[index].type != TokenType.ColonColon)
{
throw new Exception($"{tokens[index].region}: Expected `::` after label name starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new StatNode.Label(label: name, startRegion: startRegion, endRegion: endRegion);
}
case TokenType.Name:
case TokenType.RoundOpen:
{
SuffixexpNode suffixExp = ParseSuffixExp(tokens);
if(index >= tokens.Length)
{
if(suffixExp is SuffixexpNode.Normal)
{
throw new Exception($"{startRegion}: Expected function call, got normal suffix expression");
}
if(suffixExp is SuffixexpNode.Functioncall functioncall)
{
return new StatNode.Functioncall(node: functioncall.node);
}
}
else
{
switch(tokens[index].type)
{
case TokenType.Equals:
{
index += 1;
List<VarNode> lhs = [SuffixExpToVar(suffixExp)];
ExplistNode rhs = ParseExplist(tokens);
return new StatNode.Assignment(lhs: new(vars: lhs, startRegion: startRegion, endRegion: suffixExp.endRegion), rhs: rhs, startRegion: startRegion, endRegion: rhs.endRegion);
}
case TokenType.Comma:
{
List<VarNode> vars = [SuffixExpToVar(suffixExp)];
while(index < tokens.Length && tokens[index].type == TokenType.Comma)
{
index += 1;
vars.Add(ParseVar(tokens));
}
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `=` for assignment starting at {startRegion}");
}
if(tokens[index].type != TokenType.Equals)
{
throw new Exception($"{tokens[index].region}: Expected `=` for assignment starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
VarlistNode varlistNode = new(vars: vars, startRegion: startRegion, endRegion: vars[^1].endRegion);
ExplistNode rhs = ParseExplist(tokens);
return new StatNode.Assignment(lhs: varlistNode, rhs: rhs, startRegion: startRegion, endRegion: rhs.endRegion);
}
}
if(suffixExp is SuffixexpNode.Normal)
{
throw new Exception($"{startRegion}: Expected function call, got normal suffix expression");
}
if(suffixExp is SuffixexpNode.Functioncall functioncall)
{
return new StatNode.Functioncall(node: functioncall.node);
}
}
}
break;
default:
{
throw new Exception($"Unexpected token {tokens[index]} at {startRegion}");
}
}
throw new NotImplementedException();
}
private VarNode ParseVar(Token[] tokens)
{
return SuffixExpToVar(ParseSuffixExp(tokens));
}
private static VarNode SuffixExpToVar(SuffixexpNode suffixExp)
{
if(suffixExp is not SuffixexpNode.Normal normal)
{
throw new Exception($"Expected a normal suffix expression to convert to var at {suffixExp.startRegion}-{suffixExp.endRegion}");
}
if(normal.suffixes.Count == 0)
{
if(normal.firstPart is not SuffixexpFirstPart.Name name)
{
throw new Exception($"Expected a name as first part of suffix expression to convert to var at {normal.firstPart.startRegion}-{normal.firstPart.endRegion}");
}
return new VarNode.Name(name: name.name, startRegion: suffixExp.startRegion, endRegion: suffixExp.endRegion);
}
SuffixexpSuffix last = normal.suffixes[^1];
_ = normal.suffixes.Remove(last);
return last switch
{
SuffixexpSuffix.Dot dot => new VarNode.Member(node: new(name: dot.name, value: normal, startRegion: suffixExp.startRegion, endRegion: suffixExp.endRegion), startRegion: suffixExp.startRegion, endRegion: dot.endRegion),
SuffixexpSuffix.Indexed indexed => new VarNode.Indexed(node: new(index: indexed.node, value: normal, startRegion: suffixExp.startRegion, endRegion: suffixExp.endRegion), startRegion: suffixExp.startRegion, endRegion: indexed.endRegion),
_ => throw new Exception($"Expected dot or indexed suffix expression to convert to var at {last.startRegion}-{last.endRegion}")
};
}
private SuffixexpNode ParseSuffixExp(Token[] tokens)
{
// primaryexp { '.' 'Name' | '[' exp']' | ':' 'Name' args | args }
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}");
}
CodeRegion startRegion = tokens[index].region;
SuffixexpFirstPart firstPart;
switch(tokens[index].type)
{
case TokenType.Name:
{
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
firstPart = new SuffixexpFirstPart.Name(name, startRegion, startRegion);
}
break;
case TokenType.RoundOpen:
{
index += 1;
ExpNode inner = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `)` to close bracketed expression starting at {startRegion}");
}
if(tokens[index].type != TokenType.RoundClosed)
{
throw new Exception($"{tokens[index].region}: Expected `)` to close bracketed expression at {startRegion}, got {tokens[index].type}");
}
firstPart = new SuffixexpFirstPart.BracketedExp(node: inner, startRegion: startRegion, endRegion: tokens[index].region);
index += 1;
}
break;
default:
throw new Exception($"{startRegion}: Expected either `)` or name as first part of suffix-expression, got {tokens[index].type}");
}
List<SuffixexpSuffix> suffixes = [];
bool shouldContinue = true;
while(shouldContinue && index < tokens.Length)
{
CodeRegion suffixStartRegion = tokens[index].region;
switch(tokens[index].type)
{
case TokenType.Dot:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name in dotted suffix of suffix expression starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name in dotted suffix of suffix expression at {startRegion}, got {tokens[index].type}");
}
CodeRegion suffixEndRegion = tokens[index].region;
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
suffixes.Add(new SuffixexpSuffix.Dot(name, startRegion: suffixStartRegion, endRegion: suffixEndRegion));
}
break;
case TokenType.SquareOpen:
{
index += 1;
ExpNode inner = ParseExp(tokens);
suffixes.Add(new SuffixexpSuffix.Indexed(node: inner, startRegion: suffixStartRegion, endRegion: tokens[index - 1].region));
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `]` to close indexed suffix of suffix-expression starting at {suffixStartRegion}");
}
if(tokens[index].type != TokenType.SquareClosed)
{
throw new Exception($"{tokens[index].region}: Expected `]` to close indexed suffix of suffix-expression at {suffixStartRegion}, got {tokens[index].type}");
}
index += 1;
}
break;
case TokenType.Colon:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name as first arg after `:` in args suffix in suffix-expression starting at {suffixStartRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name as first arg after `:` in args suffix in suffix-expression at {suffixStartRegion}, got {tokens[index].type}");
}
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
ArgsNode args = ParseArgs(tokens);
suffixes.Add(new SuffixexpSuffix.ArgsFirstArg(new(name, rest: args, startRegion: suffixStartRegion, endRegion: args.endRegion)));
}
break;
case TokenType.RoundOpen:
case TokenType.CurlyOpen:
case TokenType.StringLiteral:
{
ArgsNode args = ParseArgs(tokens);
suffixes.Add(new SuffixexpSuffix.Args(node: args, startRegion: suffixStartRegion, endRegion: args.endRegion));
}
break;
default:
{
shouldContinue = false;
}
break;
}
}
CodeRegion endRegion;
if(suffixes.Count > 0)
{
endRegion = suffixes[^1].endRegion;
SuffixexpNode? ret = suffixes[^1] switch
{
SuffixexpSuffix.Args args => new SuffixexpNode.Functioncall(
node: new(
function: new SuffixexpNode.Normal(firstPart, suffixes[..^1], startRegion, args.endRegion),
args: args.node,
objectArg: null
)
),
SuffixexpSuffix.ArgsFirstArg node => new SuffixexpNode.Functioncall(
node: new(
function: new SuffixexpNode.Normal(firstPart: firstPart, suffixes: suffixes[..^1], startRegion, node.endRegion),
objectArg: node.node.name,
args: node.node.rest
)
),
_ => null,
};
if(ret is not null)
{
return ret;
}
}
else
{
endRegion = firstPart.endRegion;
}
return new SuffixexpNode.Normal(firstPart: firstPart, suffixes: suffixes, startRegion: startRegion, endRegion: endRegion);
}
private ArgsNode ParseArgs(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `(`, `{{` or string to start args");
}
CodeRegion startRegion = tokens[index].region;
switch(tokens[index].type)
{
case TokenType.RoundOpen:
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected explist or `)` to continue args starting at {startRegion}");
}
if(tokens[index].type == TokenType.RoundClosed)
{
CodeRegion endRegion = tokens[index].region;
index += 1;
return new ArgsNode.Bracketed(null, startRegion: startRegion, endRegion: endRegion);
}
ExplistNode exps = ParseExplist(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `)` to close args starting at {startRegion}");
}
if(tokens[index].type != TokenType.RoundClosed)
{
throw new Exception($"{tokens[index].region}: Expected `)` to close args starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
return new ArgsNode.Bracketed(node: exps, startRegion: startRegion, endRegion: exps.endRegion);
}
case TokenType.CurlyOpen:
{
TableconstructorNode node = ParseTableconstructor(tokens);
return new ArgsNode.Tableconstructor(node: node, startRegion: startRegion, endRegion: node.endRegion);
}
case TokenType.StringLiteral:
{
string value = ((Token.StringData)tokens[index].data!).data;
index += 1;
return new ArgsNode.Literal(name: value, startRegion: startRegion, endRegion: startRegion);
}
default:
throw new Exception($"{tokens[index].region}: Expected explist or `)` to continue args starting at {startRegion}");
}
}
private TableconstructorNode ParseTableconstructor(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `{{` to start tableconstructor");
}
CodeRegion startRegion = tokens[index].region;
if(tokens[index].type != TokenType.CurlyOpen)
{
throw new Exception($"{startRegion}: Expected `{{` to start tableconstructor, got {tokens[index].type}");
}
index += 1;
if(index < tokens.Length && tokens[index].type == TokenType.CurlyClosed)
{
CodeRegion emptyEndRegion = tokens[index].region;
index += 1;
return new TableconstructorNode(exps: null, startRegion: startRegion, endRegion: emptyEndRegion);
}
FieldlistNode fields = ParseFieldlist(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `}}` to close tableconstructor starting at {startRegion}");
}
if(tokens[index].type != TokenType.CurlyClosed)
{
throw new Exception($"{tokens[index].region}: Expected `}}` to close tableconstructor starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new TableconstructorNode(exps: fields, startRegion: startRegion, endRegion: endRegion);
}
private FieldlistNode ParseFieldlist(Token[] tokens)
{
List<FieldNode> fields = [ParseField(tokens)];
while(index < tokens.Length && IsFieldsep(tokens[index]))
{
index += 1;
if(index < tokens.Length && tokens[index].type is TokenType.SquareOpen or
TokenType.Name or TokenType.Nil or TokenType.True or TokenType.False or TokenType.Numeral or TokenType.StringLiteral or
TokenType.DotDotDot or TokenType.CurlyOpen or TokenType.Function or TokenType.Minus or TokenType.Hash or TokenType.Not or TokenType.Nil or TokenType.RoundOpen)
{
fields.Add(ParseField(tokens));
}
}
// NOTE: Since at least 1 field is parsed the list accesses are safe
return new FieldlistNode(exps: fields, startRegion: fields[0].startRegion, endRegion: fields[^1].endRegion);
}
private static bool IsFieldsep(Token token) => token.type is TokenType.Comma or TokenType.Semicolon;
private FieldNode ParseField(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `[` or name to start field");
}
CodeRegion startRegion = tokens[index].region;
switch(tokens[index].type)
{
case TokenType.SquareOpen:
{
index += 1;
ExpNode indexNode = ParseExp(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `]` to close indexed field in indexed field assignment starting at {startRegion}");
}
if(tokens[index].type != TokenType.SquareClosed)
{
throw new Exception($"{tokens[index].region}: Expected `]` to close indexed field starting in indexed field assignment at {startRegion}, got {tokens[index].type}");
}
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `=` to continue indexed field assignment starting at {startRegion}");
}
if(tokens[index].type != TokenType.Equals)
{
throw new Exception($"{tokens[index].region}: Expected `=` to continue indexed field assignment starting at {startRegion}, got {tokens[index].type}");
}
index += 1;
ExpNode rhs = ParseExp(tokens);
return new FieldNode.IndexedAssignment(node: new(index: indexNode, rhs: rhs, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
case TokenType.Name:
{
if(index + 1 < tokens.Length && tokens[index + 1].type == TokenType.Equals)
{
string name = ((Token.StringData)tokens[index].data!).data;
index += 2;
ExpNode rhs = ParseExp(tokens);
return new FieldNode.Assignment(node: new(lhs: name, rhs: rhs, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
ExpNode exp = ParseExp(tokens);
return new FieldNode.Exp(node: exp, startRegion: startRegion, endRegion: tokens[index - 1].region);
}
default:
{
ExpNode exp = ParseExp(tokens);
return new FieldNode.Exp(node: exp, startRegion: startRegion, endRegion: tokens[index - 1].region);
}
}
}
private AttnamelistNode ParseAttnamelist(Token[] tokens)
{
List<AttnameNode> attnames = [ParseAttname(tokens)];
while(index < tokens.Length && tokens[index].type == TokenType.Comma)
{
index += 1;
attnames.Add(ParseAttname(tokens));
}
// NOTE: Since at least 1 attname is parsed the list accesses are safe
return new(attnames: attnames, startRegion: attnames[0].startRegion, endRegion: attnames[^1].endRegion);
}
private AttnameNode ParseAttname(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name to start attname");
}
CodeRegion startRegion = tokens[index].region;
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name to start attname at {startRegion}, got {tokens[index].type}");
}
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
if(index < tokens.Length && tokens[index].type == TokenType.Lt)
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected attribute name of attname starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected attribute name of attname at {startRegion}, got {tokens[index].type}");
}
string attribute = ((Token.StringData)tokens[index].data!).data;
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `>` to close attribute of attname starting at {startRegion}");
}
if(tokens[index].type != TokenType.Gt)
{
throw new Exception($"{tokens[index].region}: Expected `>` to close attribute of attname starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new AttnameNode(name: name, attribute: attribute, startRegion: startRegion, endRegion: endRegion);
}
return new AttnameNode(name: name, attribute: null, startRegion: startRegion, endRegion: startRegion);
}
private FuncbodyNode ParseFuncbody(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `(` to start funcbody");
}
CodeRegion startRegion = tokens[index].region;
if(tokens[index].type != TokenType.RoundOpen)
{
throw new Exception($"{tokens[index].region}: Expected `(` to start funcbody at {startRegion}, got {tokens[index].type}");
}
index += 1;
ParlistNode? pars;
if(index < tokens.Length && tokens[index].type == TokenType.RoundClosed)
{
index += 1;
pars = null;
}
else
{
pars = ParseParlist(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `)` to close parlist of funcbody starting at {startRegion}");
}
if(tokens[index].type != TokenType.RoundClosed)
{
throw new Exception($"{tokens[index].region}: Expected `)` to close parlist of funcbody at {startRegion}, got {tokens[index].type}");
}
index += 1;
}
BlockNode body = ParseBlock(tokens);
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `end` to close funcbody starting at {startRegion}");
}
if(tokens[index].type != TokenType.End)
{
throw new Exception($"{tokens[index].region}: Expected `end` to close funcbody starting at {startRegion}, got {tokens[index].type}");
}
CodeRegion endRegion = tokens[index].region;
index += 1;
return new FuncbodyNode(pars: pars, body: body, startRegion: startRegion, endRegion: endRegion);
}
private ParlistNode ParseParlist(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `...` or name to start parlist");
}
CodeRegion startRegion = tokens[index].region;
if(tokens[index].type == TokenType.DotDotDot)
{
index += 1;
return new ParlistNode(names: [], hasVarargs: true, startRegion: startRegion, endRegion: startRegion);
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{startRegion}: Expected `...` or name to start parlist, got {tokens[index].type}");
}
List<string> names = [((Token.StringData)tokens[index].data!).data];
index += 1;
while(index < tokens.Length && tokens[index].type == TokenType.Comma)
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `...` or name to continue parlist starting at {startRegion}");
}
switch(tokens[index].type)
{
case TokenType.Name:
{
names.Add(((Token.StringData)tokens[index].data!).data);
index += 1;
}
break;
case TokenType.DotDotDot:
{
CodeRegion endRegion = tokens[index].region;
index += 1;
return new ParlistNode(names: names, hasVarargs: true, startRegion: startRegion, endRegion: endRegion);
};
default:
{
throw new Exception($"{tokens[index].region}: Expected `...` or name to continue parlist starting at {startRegion}, got {tokens[index].type}");
}
}
}
return new ParlistNode(names: names, hasVarargs: false, startRegion: startRegion, endRegion: tokens[index - 1].region);
}
private FuncnameNode ParseFuncname(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name to start funcname");
}
CodeRegion startRegion = tokens[index].region;
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{startRegion}: Expected name to start funcname, got {tokens[index].type}");
}
string name = ((Token.StringData)tokens[index].data!).data;
index += 1;
List<string> dottedNames = [];
while(index < tokens.Length && tokens[index].type == TokenType.Dot)
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name in dotted funcname starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name in dotted funcname starting at {startRegion}, got {tokens[index].type}");
}
dottedNames.Add(((Token.StringData)tokens[index].data!).data);
index += 1;
}
if(index < tokens.Length && tokens[index].type == TokenType.Colon)
{
index += 1;
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected name as first arg name after `:` in funcname starting at {startRegion}");
}
if(tokens[index].type != TokenType.Name)
{
throw new Exception($"{tokens[index].region}: Expected name as first arg name after `:` in funcname starting at {startRegion}, got {tokens[index].type}");
}
string firstArg = ((Token.StringData)tokens[index].data!).data;
CodeRegion endRegion = tokens[index].region;
index += 1;
return new FuncnameNode(name: name, dottedNames: dottedNames, firstArg: firstArg, startRegion: startRegion, endRegion: endRegion);
}
return new FuncnameNode(name: name, dottedNames: dottedNames, firstArg: null, startRegion: startRegion, endRegion: tokens[index - 1].region);
}
private ExplistNode ParseExplist(Token[] tokens)
{
CodeRegion startRegion = tokens[index].region;
List<ExpNode> exps = [ParseExp(tokens)];
while(index < tokens.Length && tokens[index].type == TokenType.Comma)
{
index += 1;
exps.Add(ParseExp(tokens));
}
return new ExplistNode(exps: exps, startRegion: startRegion, endRegion: tokens[index - 1].region);
}
private ExpNode ParseExp(Token[] tokens)
{
ExpNode lhs = ParseExpPrimary(tokens);
return ParseExpPrecedence(tokens, lhs, 0);
}
private ExpNode ParseExpPrecedence(Token[] tokens, ExpNode lhs, int minPrecedence)
{
ExpNode currentLhs = lhs;
while(index < tokens.Length && IsBinop(tokens[index]))
{
CodeRegion startRegion = tokens[index].region;
int precedence = GetPrecedence(tokens[index]);
if(precedence < minPrecedence)
{
break;
}
BinopType op = GetBinopType(tokens[index]);
index += 1;
ExpNode rhs = ParseExpPrimary(tokens);
while(index < tokens.Length && IsBinop(tokens[index]) && (GetPrecedence(tokens[index]) > precedence || (GetPrecedence(tokens[index]) == precedence && IsRightAssociative(tokens[index]))))
{
int associativityBoost = (GetPrecedence(tokens[index]) == precedence) ? 0 : 1;
rhs = ParseExpPrecedence(tokens, lhs: rhs, minPrecedence: precedence + associativityBoost);
}
currentLhs = new ExpNode.Binop(node: new(lhs: currentLhs, type: op, rhs: rhs, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
return currentLhs;
}
private static bool IsRightAssociative(Token token) => token.type is TokenType.DotDot or TokenType.Caret;
private static BinopType GetBinopType(Token token) => token.type switch
{
TokenType.Or => BinopType.LogicalOr,
TokenType.And => BinopType.LogicalAnd,
TokenType.Lt => BinopType.Lt,
TokenType.Gt => BinopType.Gt,
TokenType.LtEquals => BinopType.LtEquals,
TokenType.GtEquals => BinopType.GtEquals,
TokenType.LtLt => BinopType.Shl,
TokenType.GtGt => BinopType.Shr,
TokenType.TildeEquals => BinopType.NotEquals,
TokenType.EqualsEquals => BinopType.Equals,
TokenType.Pipe => BinopType.BinaryOr,
TokenType.Tilde => BinopType.BinaryNot,
TokenType.Ampersand => BinopType.BinaryAnd,
TokenType.DotDot => BinopType.Concat,
TokenType.Plus => BinopType.Add,
TokenType.Minus => BinopType.Sub,
TokenType.Star => BinopType.Mul,
TokenType.Slash => BinopType.Div,
TokenType.SlashSlash => BinopType.IntDiv,
TokenType.Percent => BinopType.Mod,
TokenType.Caret => BinopType.Exp,
_ => throw new Exception($"{token.region}: Expected binary operator with precedence, got {token.type}"),
};
private static int GetPrecedence(Token token) => token.type switch
{
TokenType.Or => 2,
TokenType.And => 4,
TokenType.Lt or TokenType.Gt or TokenType.LtEquals or TokenType.GtEquals or TokenType.TildeEquals or TokenType.EqualsEquals => 6,
TokenType.Pipe => 8,
TokenType.Tilde => 10,
TokenType.Ampersand => 12,
TokenType.LtLt or TokenType.GtGt => 14,
TokenType.DotDot => 16,
TokenType.Plus or TokenType.Minus => 18,
TokenType.Star or TokenType.Slash or TokenType.SlashSlash or TokenType.Percent => 20,
TokenType.Caret => 22,
_ => throw new Exception($"{token.region}: Expected binary operator with precedence, got {token.type}"),
};
private static bool IsBinop(Token token) => token.type switch
{
TokenType.Or or TokenType.And or TokenType.Lt or TokenType.Gt or TokenType.LtEquals or TokenType.GtEquals or TokenType.TildeEquals or TokenType.EqualsEquals or
TokenType.Pipe or TokenType.Tilde or TokenType.Ampersand or TokenType.LtLt or TokenType.GtGt or TokenType.DotDot or TokenType.Plus or TokenType.Minus or
TokenType.Star or TokenType.Slash or TokenType.SlashSlash or TokenType.Percent or TokenType.Caret => true,
_ => false
};
private ExpNode ParseExpPrimary(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected primary expression (`nil`, `true`, `false`, numeral, string, `...`, `function`, `{{`, `#`, `not`, `~`)");
}
CodeRegion startRegion = tokens[index].region;
switch(tokens[index].type)
{
case TokenType.Nil:
{
index += 1;
return new ExpNode.Nil(region: startRegion);
}
case TokenType.True:
{
index += 1;
return new ExpNode.True(region: startRegion);
}
case TokenType.False:
{
index += 1;
return new ExpNode.False(region: startRegion);
}
case TokenType.Numeral:
{
INumeral numeral = ((Token.NumeralData)tokens[index].data!).numeral;
index += 1;
return new ExpNode.Numeral(value: numeral, region: startRegion);
}
case TokenType.StringLiteral:
{
string value = ((Token.StringData)tokens[index].data!).data;
index += 1;
return new ExpNode.LiteralString(value: value, region: startRegion);
}
case TokenType.DotDotDot:
{
index += 1;
return new ExpNode.Varargs(region: startRegion);
}
case TokenType.CurlyOpen:
{
TableconstructorNode inner = ParseTableconstructor(tokens);
return new ExpNode.Tableconstructor(node: inner);
}
case TokenType.Function:
{
index += 1;
FuncbodyNode body = ParseFuncbody(tokens);
return new ExpNode.Functiondef(node: body, startRegion: startRegion, endRegion: body.endRegion);
}
case TokenType.Minus:
{
index += 1;
ExpNode unop = ParseExp(tokens);
return new ExpNode.Unop(node: new(type: UnopType.Minus, exp: unop, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
case TokenType.Hash:
{
index += 1;
ExpNode unop = ParseExp(tokens);
return new ExpNode.Unop(node: new(type: UnopType.Length, exp: unop, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
case TokenType.Not:
{
index += 1;
ExpNode unop = ParseExp(tokens);
return new ExpNode.Unop(node: new(type: UnopType.LogicalNot, exp: unop, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
case TokenType.Tilde:
{
index += 1;
ExpNode unop = ParseExp(tokens);
return new ExpNode.Unop(node: new(type: UnopType.BinaryNot, exp: unop, startRegion: startRegion, endRegion: tokens[index - 1].region));
}
default:
{
SuffixexpNode suffixexp = ParseSuffixExp(tokens);
return new ExpNode.Suffixexp(node: suffixexp);
}
}
}
private RetstatNode ParseRetstat(Token[] tokens)
{
if(index >= tokens.Length)
{
throw new Exception($"Index {index} out of bounds of {tokens.Length}, expected `return` to start retstat");
}
CodeRegion startRegion = tokens[index].region;
if(tokens[index].type != TokenType.Return)
{
throw new Exception($"{startRegion}: Expected `return` to start retstat, got {tokens[index].type}");
}
index += 1;
if(index >= tokens.Length)
{
return new RetstatNode(values: null, startRegion: startRegion, endRegion: startRegion);
}
if(tokens[index].type is TokenType.Semicolon or TokenType.Else or TokenType.Elseif or TokenType.End)
{
CodeRegion emptyEndRegion;
if(tokens[index].type == TokenType.Semicolon)
{
emptyEndRegion = tokens[index].region;
index += 1;
}
else
{
emptyEndRegion = startRegion;
}
return new RetstatNode(values: null, startRegion: startRegion, endRegion: emptyEndRegion);
}
ExplistNode values = ParseExplist(tokens);
CodeRegion endRegion;
if(index < tokens.Length && tokens[index].type == TokenType.Semicolon)
{
endRegion = tokens[index].region;
index += 1;
}
else
{
endRegion = values.endRegion;
}
return new RetstatNode(values: values, startRegion: startRegion, endRegion: endRegion);
}
}
+94
View File
@@ -0,0 +1,94 @@
using System;
using System.Collections.Generic;
using System.IO;
using System.Text.Json;
namespace luaaaaah;
public class Program
{
internal static JsonSerializerOptions options = new()
{
IncludeFields = true,
WriteIndented = true,
};
public static void Main(string[] args)
{
switch(args[0])
{
case "test":
{
Test(args[1]);
}
break;
case "run":
{
Run(args[1], true);
}
break;
}
}
public static void Run(string file, bool debug)
{
string content = File.ReadAllText(file).ReplaceLineEndings();
Token[] tokens = new Tokenizer().Tokenize(content);
if(debug)
{
foreach(Token token in tokens)
{
Console.WriteLine($"{token.region}: {token.type} {{{token.data}}}");
}
}
if(tokens.Length == 0)
{
return;
}
if(Path.GetFileName(file).StartsWith("tokenizer"))
{
Console.WriteLine($"Skipping parsing of `{file}`");
}
else
{
Parser.ChunkNode root = new Parser().Parse(tokens);
if(debug)
{
Console.WriteLine("Parsed tree:");
Console.WriteLine(JsonSerializer.Serialize(root, options: options));
}
}
}
static readonly Dictionary<string, string> failedFiles = [];
public static void Test(string directory)
{
TestRecursive(directory);
Console.WriteLine("===FAILED===");
foreach(KeyValuePair<string, string> entry in failedFiles)
{
Console.WriteLine($"{entry.Key}: {entry.Value}");
}
Console.WriteLine($"==={failedFiles.Count}===");
}
public static void TestRecursive(string directory)
{
foreach(string file in Directory.EnumerateFiles(directory))
{
if(file.EndsWith(".lua"))
{
try
{
Run(file, false);
}
catch(Exception e)
{
Console.WriteLine($"{file}: {e}");
failedFiles.Add(file, e.ToString());
}
}
}
foreach(string dir in Directory.EnumerateDirectories(directory))
{
TestRecursive(dir);
}
}
}
-3
View File
@@ -1,3 +0,0 @@
# luaaaaah
lua interpreter
+2836
View File
@@ -0,0 +1,2836 @@
using System;
using System.Collections.Generic;
using System.Text;
namespace luaaaaah;
class Tokenizer
{
private readonly List<Token> tokens = [];
private State state = State.Start;
int? lastIndex;
int index;
int openingLongBracketLevel;
int closingLongBracketLevel;
Token? currentToken;
CodeLocation currentLocation = new(line: 0, col: 0);
long escapeSequenceNumber;
public Token[] Tokenize(string content)
{
if(content.StartsWith('#'))
{
content = content[content.IndexOf('\n')..];
}
while(index < content.Length)
{
TokenizeChar(content[index]);
if(content[index] == '\n')
{
currentLocation.line += 1;
currentLocation.col = 0;
}
else
{
currentLocation.col += 1;
}
index += 1;
}
TokenizeChar('\n');
return [.. tokens];
}
private void AppendDataChar(char ch)
{
if((Token.StringData?)currentToken!.data == null)
{
currentToken!.data = new Token.StringData($"{ch}");
}
else
{
((Token.StringData?)currentToken!.data!).data += ch;
}
currentToken.region.end = new(currentLocation);
}
private void AppendDataInt(char ch)
{
if((Token.NumeralData?)currentToken!.data == null)
{
currentToken!.data = new Token.NumeralData(new INumeral.Integer(ch - '0'));
}
else
{
((INumeral.Integer)((Token.NumeralData?)currentToken!.data!).numeral).value *= 10;
((INumeral.Integer)((Token.NumeralData?)currentToken!.data!).numeral).value += ch - '0';
}
currentToken.region.end = new(currentLocation);
}
private void AppendDataIntHex(char ch)
{
int v = char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a';
if((Token.NumeralData?)currentToken!.data == null)
{
currentToken!.data = new Token.NumeralData(new INumeral.Integer(v));
}
else
{
((INumeral.Integer)((Token.NumeralData?)currentToken!.data!).numeral).value *= 16;
((INumeral.Integer)((Token.NumeralData?)currentToken!.data!).numeral).value += v;
}
currentToken.region.end = new(currentLocation);
}
private void TokenizeTerminal(State newState, TokenType type)
{
lastIndex = index;
state = newState;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: type);
}
private void TokenizeTerminalName(State newState, char ch)
{
lastIndex = index;
state = newState;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.Name, data: new Token.StringData($"{ch}"));
}
private void Backtrack(TokenType newType)
{
if(currentToken == null || currentToken.type == null)
{
throw new Exception($"Lexer error at {currentLocation}");
}
currentToken.type = newType;
currentToken.data = null;
currentLocation = new(currentToken.region.end);
tokens.Add(currentToken);
currentToken = null;
index = lastIndex!.Value;
lastIndex = null;
state = State.Start;
}
private void BacktrackNoClear(TokenType newType)
{
if(currentToken == null || currentToken.type == null)
{
throw new Exception($"Lexer error at {currentLocation}");
}
currentToken.type = newType;
currentLocation = new(currentToken.region.end);
tokens.Add(currentToken);
currentToken = null;
index = lastIndex!.Value;
lastIndex = null;
state = State.Start;
}
private void BacktrackNoTypeChange()
{
if(currentToken == null || currentToken.type == null)
{
throw new Exception($"Lexer error at {currentLocation}");
}
currentLocation = new(currentToken.region.end);
tokens.Add(currentToken);
currentToken = null;
index = lastIndex!.Value;
lastIndex = null;
state = State.Start;
}
private void TokenizeChar(char ch)
{
switch(state)
{
case State.Start:
{
switch(ch)
{
case '-':
TokenizeTerminal(State.Minus, TokenType.Minus);
break;
case ',':
TokenizeTerminal(State.Comma, TokenType.Comma);
break;
case '=':
TokenizeTerminal(State.Equals, TokenType.Equals);
break;
case '(':
TokenizeTerminal(State.RoundOpen, TokenType.RoundOpen);
break;
case ')':
TokenizeTerminal(State.RoundClosed, TokenType.RoundClosed);
break;
case '.':
TokenizeTerminal(State.Dot, TokenType.Dot);
break;
case ':':
TokenizeTerminal(State.Colon, TokenType.Colon);
break;
case '{':
TokenizeTerminal(State.CurlyOpen, TokenType.CurlyOpen);
break;
case '}':
TokenizeTerminal(State.CurlyClosed, TokenType.CurlyClosed);
break;
case '[':
TokenizeTerminal(State.SquareOpen, TokenType.SquareOpen);
break;
case ']':
TokenizeTerminal(State.SquareClosed, TokenType.SquareClosed);
break;
case '+':
TokenizeTerminal(State.Plus, TokenType.Plus);
break;
case '~':
TokenizeTerminal(State.Tilde, TokenType.Tilde);
break;
case '>':
TokenizeTerminal(State.Gt, TokenType.Gt);
break;
case '<':
TokenizeTerminal(State.Lt, TokenType.Lt);
break;
case '#':
TokenizeTerminal(State.Hash, TokenType.Hash);
break;
case '|':
TokenizeTerminal(State.Pipe, TokenType.Pipe);
break;
case '&':
TokenizeTerminal(State.Ampersand, TokenType.Ampersand);
break;
case '%':
TokenizeTerminal(State.Percent, TokenType.Percent);
break;
case '*':
TokenizeTerminal(State.Star, TokenType.Star);
break;
case '/':
TokenizeTerminal(State.Slash, TokenType.Slash);
break;
case ';':
TokenizeTerminal(State.Semicolon, TokenType.Semicolon);
break;
case '^':
TokenizeTerminal(State.Caret, TokenType.Caret);
break;
case 'a':
TokenizeTerminalName(State.A, ch);
break;
case 'b':
TokenizeTerminalName(State.B, ch);
break;
case 'd':
TokenizeTerminalName(State.D, ch);
break;
case 'e':
TokenizeTerminalName(State.E, ch);
break;
case 'f':
TokenizeTerminalName(State.F, ch);
break;
case 'i':
TokenizeTerminalName(State.I, ch);
break;
case 'g':
TokenizeTerminalName(State.G, ch);
break;
case 'l':
TokenizeTerminalName(State.L, ch);
break;
case 'n':
TokenizeTerminalName(State.N, ch);
break;
case 'o':
TokenizeTerminalName(State.O, ch);
break;
case 'r':
TokenizeTerminalName(State.R, ch);
break;
case 't':
TokenizeTerminalName(State.T, ch);
break;
case 'u':
TokenizeTerminalName(State.U, ch);
break;
case 'w':
TokenizeTerminalName(State.W, ch);
break;
case '0':
{
lastIndex = index;
state = State.Zero;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.Numeral, data: new Token.NumeralData(new INumeral.Integer(0)));
} /* tokenizeTerminalIntNum(TokenType.Numeral, TokenizerState.Zero, tokenNumeral, ch); */
break;
case '"':
{
currentToken = null;
state = State.Quote;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
}
break;
case '\'':
{
currentToken = null;
state = State.SingleQuote;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
}
break;
default:
{
if(char.IsWhiteSpace(ch)) { }
else if(char.IsAsciiLetter(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.Name, data: new Token.StringData($"{ch}"));
}
else if(char.IsDigit(ch))
{
lastIndex = index;
state = State.Integer;
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.Numeral, data: new Token.NumeralData(new INumeral.Integer(ch - '0')));
}
else
{
throw new NotImplementedException($"{ch} at {currentLocation}");
}
}
break;
}
}
break;
case State.Quote:
{
if(ch == '\\')
{
state = State.QuoteBackslash;
}
else if(ch == '"')
{
lastIndex = index;
state = State.String;
if(currentToken == null || currentToken.type == null)
{
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
}
else
{
currentToken.type = TokenType.StringLiteral;
currentToken.region.end = new(currentLocation);
}
}
else
{
AppendDataChar(ch);
}
}
break;
case State.QuoteBackslash:
{
switch(ch)
{
case 'a':
{
AppendDataChar('\u0007');
state = State.Quote;
}
break;
case 'b':
{
AppendDataChar('\u0008');
state = State.Quote;
}
break;
case 't':
{
AppendDataChar('\t');
state = State.Quote;
}
break;
case 'n':
case '\n':
{
AppendDataChar('\n');
state = State.Quote;
}
break;
case 'v':
{
AppendDataChar('\u000b');
state = State.Quote;
}
break;
case 'f':
{
AppendDataChar('\u000c');
state = State.Quote;
}
break;
case 'r':
{
AppendDataChar('\r');
state = State.Quote;
}
break;
case '\\':
{
AppendDataChar('\\');
state = State.Quote;
}
break;
case '"':
{
AppendDataChar('"');
state = State.Quote;
}
break;
case '\'':
{
AppendDataChar('\'');
state = State.Quote;
}
break;
case 'z':
{
state = State.QuoteBackslashZ;
}
break;
case 'x':
{
state = State.QuoteBackslashX;
throw new NotImplementedException($"\\u escape sequences are broken right now");
}
case 'u':
{
state = State.QuoteBackslashU;
throw new NotImplementedException($"\\u escape sequences are broken right now");
}
default: throw new Exception($"Unknown escape sequence: \\{ch} at {currentLocation}");
}
}
break;
case State.QuoteBackslashU:
{
if(ch == '{')
{
state = State.QuoteBackslashUBracket;
}
else
{
throw new Exception($"Expected `{{` to continue \\u escape sequence at {currentLocation}, got {ch}");
}
}
break;
case State.QuoteBackslashUBracket:
{
if(char.IsAsciiHexDigit(ch))
{
state = State.QuoteBackslashUBracketHex;
escapeSequenceNumber = char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a';
}
else
{
throw new Exception($"Expected hex digit to continue \\u escape sequence at {currentLocation}, got {ch}");
}
}
break;
case State.QuoteBackslashUBracketHex:
{
if(char.IsAsciiHexDigit(ch))
{
escapeSequenceNumber = (escapeSequenceNumber * 16) + (char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a');
if(escapeSequenceNumber > uint.MaxValue)
{
throw new Exception($"{currentLocation}: \\u escape sequence has a value > 2^31 which is not permitted");
}
}
else if(ch == '}')
{
state = State.Quote;
// TODO: THIS IS WRONG, there is zero padding due to the fixed size array
char[] chars = Encoding.UTF8.GetChars(BitConverter.GetBytes((uint)escapeSequenceNumber));
for(int i = 0; i < chars.Length; i++)
{
AppendDataChar(chars[i]);
}
escapeSequenceNumber = 0;
}
else
{
throw new Exception($"Expected second hex digit to continue \\u escape sequence at {currentLocation}, got {ch}");
}
}
break;
case State.QuoteBackslashZ:
{
if(ch == '\\')
{
state = State.QuoteBackslash;
}
else if(ch == '"')
{
lastIndex = index;
state = State.String;
if(currentToken == null || currentToken.type == null)
{
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
}
else
{
currentToken.type = TokenType.StringLiteral;
currentToken.region.end = new(currentLocation);
currentToken.data = new Token.StringData("");
}
}
else if(!char.IsWhiteSpace(ch))
{
AppendDataChar(ch);
state = State.Quote;
}
else
{
// Noop, https://www.lua.org/manual/5.4/manual.html#3.1:
// "The escape sequence '\z' skips the following span of whitespace characters, including line breaks;"
}
}
break;
case State.SingleQuote:
{
if(ch == '\\')
{
state = State.SingleQuoteBackslash;
}
else if(ch == '\'')
{
lastIndex = index;
state = State.String;
if(currentToken == null || currentToken.type == null)
{
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
}
else
{
currentToken.type = TokenType.StringLiteral;
currentToken.region.end = new(currentLocation);
currentToken.data = new Token.StringData("");
}
}
else
{
AppendDataChar(ch);
}
}
break;
case State.SingleQuoteBackslash:
{
switch(ch)
{
case 'a':
{
AppendDataChar('\u0007');
state = State.SingleQuote;
}
break;
case 'b':
{
AppendDataChar('\u0008');
state = State.SingleQuote;
}
break;
case 't':
{
AppendDataChar('\t');
state = State.SingleQuote;
}
break;
case 'n':
case '\n':
{
AppendDataChar('\n');
state = State.SingleQuote;
}
break;
case 'v':
{
AppendDataChar('\u000b');
state = State.SingleQuote;
}
break;
case 'f':
{
AppendDataChar('\u000c');
state = State.SingleQuote;
}
break;
case 'r':
{
AppendDataChar('\r');
state = State.SingleQuote;
}
break;
case '\\':
{
AppendDataChar('\\');
state = State.SingleQuote;
}
break;
case '"':
{
AppendDataChar('"');
state = State.SingleQuote;
}
break;
case '\'':
{
AppendDataChar('\'');
state = State.SingleQuote;
}
break;
case 'z':
state = State.SingleQuoteBackslashZ;
break;
case 'x':
state = State.SingleQuoteBackslashX;
break;
case 'u':
state = State.SingleQuoteBackslashU;
break;
default: throw new Exception($"Unknown escape sequence: \\{ch}");
}
}
break;
case State.SingleQuoteBackslashU:
state = ch == '{'
? State.SingleQuoteBackslashUBracket
: throw new Exception($"Expected `{{` to continue \\u escape sequence at {currentLocation}, got {ch}");
break;
case State.SingleQuoteBackslashUBracket:
{
if(char.IsAsciiHexDigit(ch))
{
state = State.SingleQuoteBackslashUBracketHex;
escapeSequenceNumber = char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a';
}
else
{
throw new Exception($"Expected hex digit to continue \\u escape sequence at {currentLocation}, got {ch}");
}
}
break;
case State.SingleQuoteBackslashUBracketHex:
{
if(char.IsAsciiHexDigit(ch))
{
escapeSequenceNumber = (escapeSequenceNumber * 16) + (char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a');
if(escapeSequenceNumber > uint.MaxValue)
{
throw new Exception($"{currentLocation}: \\u escape sequence has a value > 2^31 which is not permitted");
}
}
else if(ch == '}')
{
state = State.SingleQuote;
// TODO: THIS IS WRONG, there is zero padding due to the fixed size array
char[] chars = Encoding.UTF8.GetChars(BitConverter.GetBytes((uint)escapeSequenceNumber));
for(int i = 0; i < chars.Length; i++)
{
AppendDataChar(chars[i]);
}
escapeSequenceNumber = 0;
}
else
{
throw new Exception($"Expected second hex digit to continue \\u escape sequence at {currentLocation}, got {ch}");
}
}
break;
case State.SingleQuoteBackslashZ:
{
if(ch == '\\')
{
state = State.SingleQuoteBackslash;
}
else if(ch == '\'')
{
lastIndex = index;
state = State.String;
if(currentToken == null || currentToken.type == null)
{
currentToken = new(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
}
else
{
currentToken.type = TokenType.StringLiteral;
currentToken.region.end = new(currentLocation);
}
}
else if(!char.IsWhiteSpace(ch))
{
AppendDataChar(ch);
state = State.SingleQuote;
}
else
{
// Noop, https://www.lua.org/manual/5.4/manual.html#3.1:
// "The escape sequence '\z' skips the following span of whitespace characters, including line breaks;"
}
}
break;
case State.SingleQuoteBackslashX:
{
if(char.IsAsciiHexDigit(ch))
{
state = State.SingleQuoteBackslashXHex;
escapeSequenceNumber = char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a';
}
else
{
throw new Exception($"{currentLocation}: Expected hex digit in \\x escape sequence, got {ch}");
}
}
break;
case State.SingleQuoteBackslashXHex:
{
if(char.IsAsciiHexDigit(ch))
{
state = State.SingleQuote;
escapeSequenceNumber = (escapeSequenceNumber * 16) + (char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a');
// TODO: THIS IS WRONG, there is zero padding due to the fixed size array
foreach(char c in Encoding.UTF8.GetChars(BitConverter.GetBytes(escapeSequenceNumber)))
{
AppendDataChar(c);
}
escapeSequenceNumber = 0;
}
else
{
throw new Exception($"{currentLocation}: Expected second hex digit in \\x escape sequence, got {ch}");
}
}
break;
case State.QuoteBackslashX:
{
if(char.IsAsciiHexDigit(ch))
{
state = State.QuoteBackslashXHex;
escapeSequenceNumber = char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a';
}
else
{
throw new Exception($"{currentLocation}: Expected hex digit in \\x escape sequence, got {ch}");
}
}
break;
case State.QuoteBackslashXHex:
{
if(char.IsAsciiHexDigit(ch))
{
state = State.Quote;
escapeSequenceNumber = (escapeSequenceNumber * 16) + (char.IsAsciiDigit(ch) ? ch - '0' : 10 + char.ToLower(ch) - 'a');
// TODO: THIS IS WRONG, there is zero padding due to the fixed size array
foreach(char c in Encoding.UTF8.GetChars(BitConverter.GetBytes(escapeSequenceNumber)))
{
AppendDataChar(c);
}
escapeSequenceNumber = 0;
}
else
{
throw new Exception($"{currentLocation}: Expected second hex digit in \\x escape sequence, got {ch}");
}
}
break;
case State.String:
{
BacktrackNoClear(TokenType.StringLiteral);
}
break;
case State.Name:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Zero:
{
if(ch is 'x' or 'X')
{
currentToken!.type = null;
state = State.HexNumberX;
}
else if(ch == '.')
{
state = State.Float;
currentToken!.type = null;
currentToken!.data = null;
AppendDataChar('0');
AppendDataChar('.');
}
else if(char.IsAsciiDigit(ch))
{
lastIndex = index;
AppendDataInt(ch);
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Float:
{
if(char.IsAsciiDigit(ch))
{
lastIndex = index;
AppendDataChar(ch);
}
else
{
if(currentToken == null)
{
throw new Exception($"Lexer error at {currentLocation}");
}
currentLocation = new(currentToken.region.end);
currentToken.type = TokenType.Numeral;
currentToken.data = new Token.NumeralData(new INumeral.Float(float.Parse(((Token.StringData)currentToken.data!).data)));
tokens.Add(currentToken);
currentToken = null;
index = lastIndex!.Value;
lastIndex = null;
state = State.Start;
}
}
break;
case State.HexNumberX:
{
if(char.IsAsciiHexDigit(ch))
{
lastIndex = index;
currentToken!.type = TokenType.Numeral;
AppendDataIntHex(ch);
state = State.HexNumber;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.HexNumber:
{
if(ch == 'p')
{
currentToken!.type = null;
state = State.HexExpNumber;
}
else if(char.IsAsciiHexDigit(ch))
{
lastIndex = index;
currentToken!.type = TokenType.Numeral;
AppendDataIntHex(ch);
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Integer:
{
if(ch == 'e')
{
currentToken!.type = null;
state = State.ExpNumber;
}
else if(ch == '.')
{
currentToken!.type = null;
currentToken.data = new Token.StringData($"{((INumeral.Integer)((Token.NumeralData)currentToken!.data!).numeral).value}.");
state = State.Float;
}
else if(char.IsAsciiDigit(ch))
{
lastIndex = index;
currentToken!.type = TokenType.Numeral;
AppendDataInt(ch);
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.SquareOpen:
{
if(ch == '[')
{
currentToken = new Token(region: new(start: new(currentLocation), end: new(currentLocation)), type: TokenType.StringLiteral);
state = State.StringWithLongBracket;
}
else if(ch == '=')
{
openingLongBracketLevel = 1;
state = State.StringStartLongBracket;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Comma:
case State.RoundOpen:
case State.RoundClosed:
case State.CurlyOpen:
case State.CurlyClosed:
case State.Plus:
case State.TildeEquals:
case State.EqualsEquals:
case State.Hash:
case State.GtEquals:
case State.LtEquals:
case State.SquareClosed:
case State.Pipe:
case State.Ampersand:
case State.Percent:
case State.Star:
case State.Semicolon:
case State.Caret:
case State.DotDotDot:
case State.GtGt:
case State.LtLt:
case State.ColonColon:
case State.SlashSlash:
{
BacktrackNoTypeChange();
}
break;
case State.Tilde:
{
if(ch == '=')
{
lastIndex = index;
state = State.TildeEquals;
currentToken!.type = TokenType.TildeEquals;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Gt:
{
if(ch == '=')
{
lastIndex = index;
state = State.GtEquals;
currentToken!.type = TokenType.GtEquals;
}
else if(ch == '>')
{
lastIndex = index;
state = State.GtGt;
currentToken!.type = TokenType.GtGt;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Lt:
{
if(ch == '=')
{
lastIndex = index;
state = State.LtEquals;
currentToken!.type = TokenType.LtEquals;
}
else if(ch == '<')
{
lastIndex = index;
state = State.LtLt;
currentToken!.type = TokenType.LtLt;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Slash:
{
if(ch == '/')
{
lastIndex = index;
state = State.SlashSlash;
currentToken!.type = TokenType.SlashSlash;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Dot:
{
if(ch == '.')
{
lastIndex = index;
state = State.DotDot;
currentToken!.type = TokenType.DotDot;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.DotDot:
{
if(ch == '.')
{
lastIndex = index;
state = State.DotDotDot;
currentToken!.type = TokenType.DotDotDot;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Colon:
{
if(ch == ':')
{
lastIndex = index;
state = State.ColonColon;
currentToken!.type = TokenType.ColonColon;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Equals:
{
if(ch == '=')
{
lastIndex = index;
state = State.EqualsEquals;
currentToken!.type = TokenType.EqualsEquals;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.Minus:
{
if(ch == '-')
{
lastIndex = index;
state = State.SmallCommentStart;
currentToken = null;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.SmallCommentStart:
{
if(ch == '[')
{
state = State.BigCommentStartLongBracket;
}
else if(ch == '\n')
{
state = State.Start;
lastIndex = null;
}
else
{
state = State.SmallComment;
}
}
break;
case State.SmallComment:
{
if(ch == '\n')
{
state = State.Start;
lastIndex = null;
}
}
break;
case State.BigCommentStartLongBracket:
{
if(ch == '=')
{
openingLongBracketLevel += 1;
}
else if(ch == '[')
{
state = State.BigComment;
}
else if(ch == '\n')
{
state = State.Start;
}
else
{
state = State.SmallComment;
}
}
break;
case State.BigComment:
{
if(ch == ']')
{
state = State.BigCommentEndLongBracket;
closingLongBracketLevel = 0;
}
}
break;
case State.BigCommentEndLongBracket:
{
if(ch == '=')
{
closingLongBracketLevel += 1;
if(openingLongBracketLevel < closingLongBracketLevel)
{
state = State.BigComment;
}
}
else if(ch == ']' && openingLongBracketLevel == closingLongBracketLevel)
{
state = State.Start;
openingLongBracketLevel = 0;
closingLongBracketLevel = 0;
}
else
{
closingLongBracketLevel = 0;
state = State.BigComment;
}
}
break;
case State.StringStartLongBracket:
{
if(ch == '=')
{
openingLongBracketLevel += 1;
}
else if(ch == '[')
{
state = State.StringWithLongBracket;
}
else
{
BacktrackNoTypeChange();
}
}
break;
case State.StringWithLongBracket:
{
if(ch == ']')
{
state = State.StringEndLongBracket;
closingLongBracketLevel = 0;
}
else
{
AppendDataChar(ch);
}
}
break;
case State.StringEndLongBracket:
{
if(ch == '=')
{
closingLongBracketLevel += 1;
if(openingLongBracketLevel < closingLongBracketLevel)
{
state = State.StringWithLongBracket;
}
AppendDataChar(ch);
}
else if(ch == ']' && openingLongBracketLevel == closingLongBracketLevel)
{
if(currentToken == null || currentToken.type == null)
{
throw new Exception($"Lexer error at {currentLocation}");
}
if((Token.StringData?)currentToken.data == null)
{
currentToken.data = new Token.StringData("");
}
currentToken.type = TokenType.StringLiteral;
((Token.StringData)currentToken.data).data = ((Token.StringData)currentToken.data).data.Remove(((Token.StringData)currentToken.data).data.Length - closingLongBracketLevel);
currentLocation = new(currentToken.region.end);
tokens.Add(currentToken);
currentToken = null;
lastIndex = null;
state = State.Start;
openingLongBracketLevel = 0;
closingLongBracketLevel = 0;
}
else
{
closingLongBracketLevel = 0;
AppendDataChar(ch);
state = State.StringWithLongBracket;
}
}
break;
case State.A:
{
if(ch == 'n')
{
lastIndex = index;
state = State.An;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.An:
{
if(ch == 'd')
{
lastIndex = index;
state = State.And;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.And:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.And);
}
}
break;
case State.W:
{
if(ch == 'h')
{
lastIndex = index;
state = State.Wh;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Wh:
{
if(ch == 'i')
{
lastIndex = index;
state = State.Whi;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Whi:
{
if(ch == 'l')
{
lastIndex = index;
state = State.Whil;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Whil:
{
if(ch == 'e')
{
lastIndex = index;
state = State.While;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.While:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.While);
}
}
break;
case State.B:
{
if(ch == 'r')
{
lastIndex = index;
state = State.Br;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Br:
{
if(ch == 'e')
{
lastIndex = index;
state = State.Bre;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Bre:
{
if(ch == 'a')
{
lastIndex = index;
state = State.Brea;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Brea:
{
if(ch == 'k')
{
lastIndex = index;
state = State.Break;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Break:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Break);
}
}
break;
case State.G:
{
if(ch == 'o')
{
lastIndex = index;
state = State.Go;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Go:
{
if(ch == 't')
{
lastIndex = index;
state = State.Got;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Got:
{
if(ch == 'o')
{
lastIndex = index;
state = State.Goto;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Goto:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Goto);
}
}
break;
case State.R:
{
if(ch == 'e')
{
lastIndex = index;
state = State.Re;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Re:
{
if(ch == 't')
{
lastIndex = index;
state = State.Ret;
AppendDataChar(ch);
}
else if(ch == 'p')
{
lastIndex = index;
state = State.Rep;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Ret:
{
if(ch == 'u')
{
lastIndex = index;
state = State.Retu;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Retu:
{
if(ch == 'r')
{
lastIndex = index;
state = State.Retur;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Retur:
{
if(ch == 'n')
{
lastIndex = index;
state = State.Return;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Return:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Return);
}
}
break;
case State.Rep:
{
if(ch == 'e')
{
lastIndex = index;
state = State.Repe;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Repe:
{
if(ch == 'a')
{
lastIndex = index;
state = State.Repea;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Repea:
{
if(ch == 't')
{
lastIndex = index;
state = State.Repeat;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Repeat:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Repeat);
}
}
break;
case State.N:
{
if(ch == 'i')
{
lastIndex = index;
state = State.Ni;
AppendDataChar(ch);
}
else if(ch == 'o')
{
lastIndex = index;
state = State.No;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Ni:
{
if(ch == 'l')
{
lastIndex = index;
state = State.Nil;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Nil:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Nil);
}
}
break;
case State.No:
{
if(ch == 't')
{
lastIndex = index;
state = State.Not;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Not:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Not);
}
}
break;
case State.T:
{
if(ch == 'h')
{
lastIndex = index;
state = State.Th;
AppendDataChar(ch);
}
else if(ch == 'r')
{
lastIndex = index;
state = State.Tr;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Th:
{
if(ch == 'e')
{
lastIndex = index;
state = State.The;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.The:
{
if(ch == 'n')
{
lastIndex = index;
state = State.Then;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Then:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Then);
}
}
break;
case State.Tr:
{
if(ch == 'u')
{
lastIndex = index;
state = State.Tru;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Tru:
{
if(ch == 'e')
{
lastIndex = index;
state = State.True;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.True:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.True);
}
}
break;
case State.E:
{
if(ch == 'l')
{
lastIndex = index;
state = State.El;
AppendDataChar(ch);
}
else if(ch == 'n')
{
lastIndex = index;
state = State.En;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.El:
{
if(ch == 's')
{
lastIndex = index;
state = State.Els;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Els:
{
if(ch == 'e')
{
lastIndex = index;
state = State.Else;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Else:
{
if(ch == 'i')
{
lastIndex = index;
state = State.Elsei;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Else);
}
}
break;
case State.Elsei:
{
if(ch == 'f')
{
lastIndex = index;
state = State.Elseif;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Elseif:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Elseif);
}
}
break;
case State.En:
{
if(ch == 'd')
{
lastIndex = index;
state = State.End;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.End:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.End);
}
}
break;
case State.O:
{
if(ch == 'r')
{
lastIndex = index;
state = State.Or;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Or:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Or);
}
}
break;
case State.D:
{
if(ch == 'o')
{
lastIndex = index;
state = State.Do;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Do:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Do);
}
}
break;
case State.I:
{
if(ch == 'f')
{
lastIndex = index;
state = State.If;
AppendDataChar(ch);
}
else if(ch == 'n')
{
lastIndex = index;
state = State.In;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.In:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.In);
}
}
break;
case State.If:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.If);
}
}
break;
case State.F:
{
if(ch == 'u')
{
lastIndex = index;
state = State.Fu;
AppendDataChar(ch);
}
else if(ch == 'a')
{
lastIndex = index;
state = State.Fa;
AppendDataChar(ch);
}
else if(ch == 'o')
{
lastIndex = index;
state = State.Fo;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Fu:
{
if(ch == 'n')
{
lastIndex = index;
state = State.Fun;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Fun:
{
if(ch == 'c')
{
lastIndex = index;
state = State.Func;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Func:
{
if(ch == 't')
{
lastIndex = index;
state = State.Funct;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Funct:
{
if(ch == 'i')
{
lastIndex = index;
state = State.Functi;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Functi:
{
if(ch == 'o')
{
lastIndex = index;
state = State.Functio;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Functio:
{
if(ch == 'n')
{
lastIndex = index;
state = State.Function;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Function:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Function);
}
}
break;
case State.Fa:
{
if(ch == 'l')
{
lastIndex = index;
state = State.Fal;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Fal:
{
if(ch == 's')
{
lastIndex = index;
state = State.Fals;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Fals:
{
if(ch == 'e')
{
lastIndex = index;
state = State.False;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.False:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.False);
}
}
break;
case State.Fo:
{
if(ch == 'r')
{
lastIndex = index;
state = State.For;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.For:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.For);
}
}
break;
case State.L:
{
if(ch == 'o')
{
lastIndex = index;
state = State.Lo;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Lo:
{
if(ch == 'c')
{
lastIndex = index;
state = State.Loc;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Loc:
{
if(ch == 'a')
{
lastIndex = index;
state = State.Loca;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Loca:
{
if(ch == 'l')
{
lastIndex = index;
state = State.Local;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Local:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Local);
}
}
break;
case State.U:
{
if(ch == 'n')
{
lastIndex = index;
state = State.Un;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Un:
{
if(ch == 't')
{
lastIndex = index;
state = State.Unt;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Unt:
{
if(ch == 'i')
{
lastIndex = index;
state = State.Unti;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Unti:
{
if(ch == 'l')
{
lastIndex = index;
state = State.Until;
AppendDataChar(ch);
}
else if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
BacktrackNoClear(TokenType.Name);
}
}
break;
case State.Until:
{
if(char.IsAsciiLetterOrDigit(ch) || ch == '_')
{
lastIndex = index;
state = State.Name;
currentToken!.type = TokenType.Name;
AppendDataChar(ch);
}
else
{
Backtrack(TokenType.Until);
}
}
break;
default:
throw new NotImplementedException(state.ToString());
}
}
private enum State
{
Start,
Quote, SingleQuote, Name, Integer, Float, Zero,
A, B, D, E, F, G, I, L, N, O, R, T, U, W,
Plus, Minus, Star, Slash, Percent, Caret, Hash,
Ampersand, Tilde, Pipe, Lt, Gt, Equals, RoundOpen, RoundClosed, CurlyOpen, CurlyClosed, SquareOpen, SquareClosed, StringStartLongBracket, StringWithLongBracket, StringEndLongBracket,
Colon, Semicolon, Comma, Dot,
An, Br, Do, El, En, Fa, Fo, Fu, Go, If, In, Lo, Ni, No, Or, Re, Th, Tr, Un, Wh,
LtLt, GtGt, SlashSlash, EqualsEquals, TildeEquals, LtEquals, GtEquals, ColonColon, DotDot,
SmallCommentStart, QuoteBackslash, SingleQuoteBackslash, String, HexNumberX, ExpNumber,
And, Bre, Els, End, Fal, For, Fun, Got, Loc, Nil, Not, Rep, Ret, The, Tru, Unt, Whi,
DotDotDot, HexNumber, QuoteBackslashZ, SingleQuoteBackslashZ, QuoteBackslashX, SingleQuoteBackslashX, QuoteBackslashXHex, SingleQuoteBackslashXHex,
SingleQuoteBackslashU, SingleQuoteBackslashUBracket, SingleQuoteBackslashUBracketHex,
QuoteBackslashU, QuoteBackslashUBracket, QuoteBackslashUBracketHex,
SmallComment, BigComment, BigCommentStartLongBracket, BigCommentEndLongBracket,
Brea, Else, Fals, Func, Goto, Loca, Repe, Retu, Then, True, Unti, Whil, HexExpNumber,
Break, Elsei, False, Funct, Local, Repea, Retur, Until, While,
Elseif, Functi, Repeat, Return,
Functio,
Function,
}
}
internal class Token(CodeRegion region, TokenType? type = null, Token.IData? data = null)
{
public CodeRegion region = region;
public IData? data = data;
public TokenType? type = type;
public interface IData { }
public class NumeralData(INumeral numeral) : IData
{
public INumeral numeral = numeral;
public override string ToString()
{
return $"NumeralData {numeral}";
}
}
public class StringData(string data) : IData
{
public string data = data;
public override string ToString()
{
return $"StringData \"{data}\"";
}
}
}
public enum TokenType
{
Name,
And, Break, Do, Else, Elseif, End,
False, For, Function, Goto, If, In,
Local, Nil, Not, Or, Repeat, Return,
Then, True, Until, While,
Plus, Minus, Star, Slash, Percent, Caret, Hash,
Ampersand, Tilde, Pipe, LtLt, GtGt, SlashSlash,
EqualsEquals, TildeEquals, LtEquals, GtEquals, Lt, Gt, Equals,
RoundOpen, RoundClosed, CurlyOpen, CurlyClosed, SquareOpen, SquareClosed, ColonColon,
Semicolon, Colon, Comma, Dot, DotDot, DotDotDot,
Numeral,
StringLiteral,
}
+9
View File
@@ -0,0 +1,9 @@
<Project
Sdk="Microsoft.NET.Sdk">
<PropertyGroup>
<OutputType>Exe</OutputType>
<TargetFramework>net8.0</TargetFramework>
<ImplicitUsings>disable</ImplicitUsings>
<Nullable>enable</Nullable>
</PropertyGroup>
</Project>
-480
View File
@@ -1,480 +0,0 @@
use crate::tokenizer::Token;
pub enum Rule
{
Terminal(u8, Token),
NonTerminal(u8, u8, u8)
}
pub const NONTERMINAL_NAMES: [&str; 115] =
[
"stat__15",
"funcbody__50",
"fieldlist",
"fieldlist__1",
"namelist",
"parlist",
">_non",
"chunk",
",_non",
"field__29",
"stat__11",
"stat__45",
"stat__33",
"funcname",
"prefixexp__25",
";_?",
"stat__5",
"stat__38",
"else_non",
"stat__16",
"<_non",
"explist",
"stat_*",
"{_non",
"funcnamedotexpansion_*",
"stat__6",
"funcnamedotexpansion",
"moreattribs_*",
"field__28",
"retstat_?",
"stat__9",
"stat__37",
"field",
"local_non",
")_non",
"morevars_*",
"moreattribs__19",
"morevars",
"forthirdarg_?",
"::_non",
"stat",
"morefields_*",
"]_non",
"[_non",
"assign",
"field__31",
"stat__4",
"do_non",
"unop",
"elseifblocks__18",
"elseblock_?",
"label__21",
"funcbody__51",
"in_non",
"stat__34",
"exp",
"funcnamecolonexpansion_?",
"(_non",
"if_non",
"stat__7",
"attrib__20",
"elseifblocks__17",
"fieldsep_?",
"while_non",
"elseifblocks_*",
"binop",
"..._non",
"then_non",
"return_non",
"._non",
"stat__35",
"prefixexp",
"attnamelist",
"functioncall__26",
"elseifblocks",
"function_non",
"}_non",
"var__22",
"tableconstructor__53",
"varlist",
"S_0",
"Name_non",
"goto_non",
"parlistvarargs_?",
"elseif_non",
"for_non",
"stat__8",
"args",
"stat__41",
"stat__14",
"until_non",
"morenames_*",
"stat__39",
"stat__32",
"end_non",
"attnamelist__46",
"moreexps_*",
"stat__10",
"=_non",
":_non",
"stat__36",
"functioncall__27",
"funcname__48",
"attrib",
"repeat_non",
"args__49",
"var",
"var__23",
"stat__40",
"exp__0",
"stat__42",
"morefields",
"moreattribs",
"funcbody",
"retstat__47",
];
pub const TERMINAL_RULES: [(u8, Token); 125] =
[
(57, Token::RoundOpen),
(34, Token::RoundClosed),
(8, Token::Comma),
(66, Token::DotDotDot),
(69, Token::Dot),
(39, Token::ColonColon),
(99, Token::Colon),
(15, Token::Semicolon),
(20, Token::Lt),
(98, Token::Equals),
(6, Token::Gt),
(81, Token::Name(String::new())),
(80, Token::Return),
(80, Token::Semicolon),
(80, Token::Break),
(43, Token::SquareOpen),
(42, Token::SquareClosed),
(87, Token::StringLiteral(String::new())),
(105, Token::RoundClosed),
(72, Token::Name(String::new())),
(65, Token::Plus),
(65, Token::Minus),
(65, Token::Star),
(65, Token::Slash),
(65, Token::SlashSlash),
(65, Token::Caret),
(65, Token::Percent),
(65, Token::Ampersand),
(65, Token::Pipe),
(65, Token::GtGt),
(65, Token::LtLt),
(65, Token::DotDot),
(65, Token::Lt),
(65, Token::LtEquals),
(65, Token::Gt),
(65, Token::GtEquals),
(65, Token::EqualsEquals),
(65, Token::TildeEquals),
(65, Token::And),
(65, Token::Or),
(7, Token::Return),
(7, Token::Semicolon),
(7, Token::Break),
(47, Token::Do),
(18, Token::Else),
(50, Token::Else),
(84, Token::Elseif),
(49, Token::Then),
(94, Token::End),
(55, Token::Nil),
(55, Token::False),
(55, Token::True),
(55, Token::Numeral(String::new())),
(55, Token::StringLiteral(String::new())),
(55, Token::DotDotDot),
(55, Token::Name(String::new())),
(21, Token::Nil),
(21, Token::False),
(21, Token::True),
(21, Token::Numeral(String::new())),
(21, Token::StringLiteral(String::new())),
(21, Token::DotDotDot),
(21, Token::Name(String::new())),
(32, Token::Nil),
(32, Token::False),
(32, Token::True),
(32, Token::Numeral(String::new())),
(32, Token::StringLiteral(String::new())),
(32, Token::DotDotDot),
(32, Token::Name(String::new())),
(2, Token::Nil),
(2, Token::False),
(2, Token::True),
(2, Token::Numeral(String::new())),
(2, Token::StringLiteral(String::new())),
(2, Token::DotDotDot),
(2, Token::Name(String::new())),
(3, Token::Comma),
(3, Token::Semicolon),
(62, Token::Comma),
(62, Token::Semicolon),
(85, Token::For),
(13, Token::Name(String::new())),
(75, Token::Function),
(82, Token::Goto),
(58, Token::If),
(53, Token::In),
(33, Token::Local),
(36, Token::Name(String::new())),
(4, Token::Name(String::new())),
(5, Token::DotDotDot),
(5, Token::Name(String::new())),
(71, Token::Name(String::new())),
(104, Token::Repeat),
(29, Token::Return),
(114, Token::Semicolon),
(114, Token::Nil),
(114, Token::False),
(114, Token::True),
(114, Token::Numeral(String::new())),
(114, Token::StringLiteral(String::new())),
(114, Token::DotDotDot),
(114, Token::Name(String::new())),
(68, Token::Return),
(40, Token::Semicolon),
(40, Token::Break),
(22, Token::Semicolon),
(22, Token::Break),
(54, Token::End),
(70, Token::End),
(100, Token::End),
(11, Token::Name(String::new())),
(25, Token::End),
(78, Token::CurlyClosed),
(67, Token::Then),
(48, Token::Minus),
(48, Token::Not),
(48, Token::Hash),
(48, Token::Tilde),
(90, Token::Until),
(106, Token::Name(String::new())),
(79, Token::Name(String::new())),
(63, Token::While),
(23, Token::CurlyOpen),
(76, Token::CurlyClosed),
];
pub const NONTERMINAL_RULES: [(u8, u8, u8); 219] =
[
(80, 22, 29),
(80, 68, 114),
(80, 40, 22),
(80, 82, 81),
(80, 79, 44),
(80, 47, 25),
(80, 63, 46),
(80, 104, 59),
(80, 85, 30),
(80, 75, 89),
(80, 33, 0),
(80, 58, 93),
(80, 85, 31),
(80, 33, 11),
(80, 71, 87),
(80, 71, 73),
(80, 39, 51),
(87, 57, 105),
(87, 23, 78),
(105, 21, 34),
(44, 98, 21),
(72, 81, 95),
(95, 103, 27),
(95, 20, 60),
(95, 112, 27),
(95, 8, 36),
(103, 20, 60),
(60, 81, 6),
(7, 22, 29),
(7, 68, 114),
(7, 40, 22),
(7, 82, 81),
(7, 79, 44),
(7, 47, 25),
(7, 63, 46),
(7, 104, 59),
(7, 85, 30),
(7, 75, 89),
(7, 33, 0),
(7, 58, 93),
(7, 85, 31),
(7, 33, 11),
(7, 71, 87),
(7, 71, 73),
(7, 39, 51),
(50, 18, 7),
(74, 84, 61),
(64, 74, 64),
(64, 84, 61),
(61, 55, 49),
(49, 67, 7),
(55, 48, 55),
(55, 55, 109),
(55, 75, 113),
(55, 23, 78),
(55, 57, 14),
(55, 71, 77),
(55, 71, 26),
(55, 71, 87),
(55, 71, 73),
(109, 65, 55),
(21, 55, 96),
(21, 48, 55),
(21, 55, 109),
(21, 75, 113),
(21, 23, 78),
(21, 57, 14),
(21, 71, 77),
(21, 71, 26),
(21, 71, 87),
(21, 71, 73),
(32, 43, 28),
(32, 81, 45),
(32, 48, 55),
(32, 55, 109),
(32, 75, 113),
(32, 23, 78),
(32, 57, 14),
(32, 71, 77),
(32, 71, 26),
(32, 71, 87),
(32, 71, 73),
(28, 55, 9),
(9, 42, 45),
(45, 98, 55),
(2, 32, 3),
(2, 43, 28),
(2, 81, 45),
(2, 48, 55),
(2, 55, 109),
(2, 75, 113),
(2, 23, 78),
(2, 57, 14),
(2, 71, 77),
(2, 71, 26),
(2, 71, 87),
(2, 71, 73),
(3, 41, 62),
(3, 111, 41),
(3, 62, 32),
(38, 8, 55),
(113, 57, 1),
(1, 5, 52),
(1, 34, 25),
(52, 34, 25),
(13, 81, 102),
(102, 24, 56),
(102, 26, 24),
(102, 69, 81),
(102, 99, 81),
(56, 99, 81),
(26, 69, 81),
(24, 26, 24),
(24, 69, 81),
(73, 99, 101),
(101, 81, 87),
(51, 81, 39),
(112, 8, 36),
(27, 112, 27),
(27, 8, 36),
(36, 81, 103),
(96, 38, 96),
(96, 8, 55),
(111, 62, 32),
(41, 111, 41),
(41, 62, 32),
(91, 37, 91),
(91, 8, 81),
(37, 8, 81),
(35, 37, 35),
(35, 8, 81),
(4, 81, 91),
(5, 4, 83),
(5, 81, 91),
(83, 8, 66),
(71, 57, 14),
(71, 71, 77),
(71, 71, 26),
(71, 71, 87),
(71, 71, 73),
(14, 55, 34),
(29, 68, 114),
(114, 21, 15),
(114, 55, 96),
(114, 48, 55),
(114, 55, 109),
(114, 75, 113),
(114, 23, 78),
(114, 57, 14),
(114, 71, 77),
(114, 71, 26),
(114, 71, 87),
(114, 71, 73),
(40, 82, 81),
(40, 79, 44),
(40, 47, 25),
(40, 63, 46),
(40, 104, 59),
(40, 85, 30),
(40, 75, 89),
(40, 33, 0),
(40, 58, 93),
(40, 85, 31),
(40, 33, 11),
(40, 71, 87),
(40, 71, 73),
(40, 39, 51),
(22, 40, 22),
(22, 82, 81),
(22, 79, 44),
(22, 47, 25),
(22, 63, 46),
(22, 104, 59),
(22, 85, 30),
(22, 75, 89),
(22, 33, 0),
(22, 58, 93),
(22, 85, 31),
(22, 33, 11),
(22, 71, 87),
(22, 71, 73),
(22, 39, 51),
(97, 53, 10),
(10, 21, 16),
(89, 13, 113),
(0, 75, 19),
(19, 81, 113),
(93, 55, 12),
(12, 67, 54),
(54, 7, 70),
(54, 64, 100),
(54, 50, 94),
(70, 64, 100),
(70, 50, 94),
(100, 50, 94),
(31, 81, 17),
(17, 98, 92),
(92, 55, 108),
(46, 55, 16),
(108, 8, 88),
(88, 55, 110),
(110, 38, 16),
(110, 47, 25),
(11, 72, 44),
(11, 81, 95),
(16, 47, 25),
(25, 7, 94),
(59, 7, 86),
(59, 90, 55),
(86, 90, 55),
(30, 4, 97),
(78, 2, 76),
(106, 71, 77),
(106, 71, 26),
(77, 43, 107),
(107, 55, 42),
(79, 106, 35),
(79, 71, 77),
(79, 71, 26),
];
-31
View File
@@ -1,31 +0,0 @@
pub mod tokenizer;
pub mod parser;
pub mod grammar;
use std::{env, fs};
use crate::{tokenizer::{Token, tokenize}, parser::parse};
fn main()
{
let args: Vec<String> = env::args().collect();
let file_content = fs::read_to_string(&args[1]).expect("Could not read source file");
match compile(&file_content)
{
Ok(()) =>
{
println!("Done compiling");
}
Err(msg) => println!("ERROR: {}", msg)
}
}
fn compile(file_content: &String) -> Result<(), &'static str>
{
let tokens: Vec<Token> = tokenize(&file_content)?;
println!("{:?}", tokens);
let node = parse(tokens)?;
println!("{:?}", node);
return Ok(());
}
-1201
View File
@@ -1,1201 +0,0 @@
use crate::tokenizer::Token;
use crate::grammar::{NONTERMINAL_NAMES, NONTERMINAL_RULES, TERMINAL_RULES};
pub fn parse(tokens: Vec<Token>) -> Result<ChunkNode, &'static str>
{
return own(tokens);
}
fn own(tokens: Vec<Token>) -> Result<ChunkNode, &'static str>
{
return parse_chunk(&tokens, &mut 0);
}
#[derive(Debug)]
pub struct ChunkNode
{
block: BlockNode
}
#[derive(Debug)]
pub struct BlockNode
{
stats: Vec<StatNode>,
retstat: Option<RetstatNode>
}
#[derive(Debug)]
pub enum StatNode
{
Semicolon,
Assignment { lhs: VarlistNode, rhs: ExplistNode },
Functioncall(FunctioncallNode),
Label(String),
Break,
Goto(String),
Do(BlockNode),
While { condition: ExpNode, body: BlockNode },
Repeat { condition: ExpNode, body: BlockNode },
If { condition: ExpNode, body: BlockNode, elseifs: Vec<ElseifNode>, else_: Option<BlockNode> },
ForEq { var: String, start: ExpNode, end: ExpNode, change: Option<ExpNode>, body: BlockNode },
ForIn { vars: Vec<String>, exps: ExplistNode, body: BlockNode },
Function { name: FuncnameNode, body: FuncbodyNode },
LocalFunction { name: String, body: FuncbodyNode },
Local { attnames: AttnamelistNode, values: Option<ExplistNode> }
}
#[derive(Debug)]
pub struct RetstatNode
{
values: Option<ExplistNode>
}
#[derive(Debug)]
pub enum ExpNode
{
Nil,
False,
True,
Numeral(f64),
LiteralString(String),
Varargs,
Functiondef(FuncbodyNode),
Suffixexp(Box<SuffixexpNode>),
Tableconstructor(TableconstructorNode),
Unop(UnopType, Box<ExpNode>),
Binop { lhs: Box<ExpNode>, op: BinopType, rhs: Box<ExpNode> }
}
#[derive(Debug)]
pub enum UnopType
{
Minus, LogicalNot, Length, BinaryNot,
}
#[derive(Debug)]
pub enum BinopType
{
LogicalOr,
LocicalAnd,
Lt, Gt, LtEquals, GtEquals, NotEquals, Equals,
BinaryOr,
BinaryNot,
BinaryAnd,
Shl, Shr,
Concat,
Add, Sub,
Mul, Div, IntDiv, Mod,
Exp,
}
#[derive(Debug)]
pub struct ExplistNode
{
exps: Vec<ExpNode>
}
#[derive(Debug)]
pub struct TableconstructorNode
{
exps: Option<FieldlistNode>
}
#[derive(Debug)]
pub struct FieldlistNode
{
exps: Vec<FieldNode>
}
#[derive(Debug)]
pub enum FieldNode
{
IndexedAssignment { index: ExpNode, rhs: ExpNode },
Assignment { lhs: String, rhs: ExpNode },
Exp(ExpNode),
}
#[derive(Debug)]
pub struct VarlistNode
{
vars: Vec<VarNode>
}
#[derive(Debug)]
pub struct FunctioncallNode
{
function: SuffixexpNode,
object_arg: Option<String>,
args: ArgsNode,
}
#[derive(Debug)]
pub enum ArgsNode
{
Bracketed(Option<ExplistNode>),
Tableconstructor(TableconstructorNode),
Literal(String),
}
#[derive(Debug)]
pub struct ElseifNode
{
condition: ExpNode,
body: BlockNode,
}
#[derive(Debug)]
pub struct FuncnameNode
{
name: String,
dotted_names: Vec<String>,
first_arg: Option<String>,
}
#[derive(Debug)]
pub struct ParlistNode
{
names: Vec<String>,
has_varargs: bool,
}
#[derive(Debug)]
pub struct FuncbodyNode
{
pars: Option<ParlistNode>,
body: BlockNode,
}
#[derive(Debug)]
pub struct AttnamelistNode
{
attnames: Vec<AttnameNode>
}
#[derive(Debug)]
pub struct AttnameNode
{
name: String,
attribute: Option<String>,
}
#[derive(Debug)]
pub enum VarNode
{
Name(String),
Indexed { value: SuffixexpNode, index: ExpNode },
Member { value: SuffixexpNode, name: String }
}
#[derive(Debug)]
pub struct SuffixexpNode
{
first_part: SuffixexpFirstPart,
suffixes: Vec<SuffixexpSuffix>,
}
#[derive(Debug)]
pub enum SuffixexpFirstPart // a:b:test() => a:b.test(b) => a.b.test(a, b)
{
Name(String),
BracketedExpr(ExpNode),
}
#[derive(Debug)]
pub enum SuffixexpSuffix
{
Dot(String),
Indexed(ExpNode),
Args(ArgsNode),
ArgsFirstArg(String, ArgsNode),
}
fn parse_chunk(tokens: &Vec<Token>, i: &mut usize) -> Result<ChunkNode, &'static str>
{
return Ok(ChunkNode { block: parse_block(tokens, i)? });
}
fn parse_block(tokens: &Vec<Token>, i: &mut usize) -> Result<BlockNode, &'static str>
{
let mut stats: Vec<StatNode> = Vec::new();
while *i < tokens.len() && tokens[*i] != Token::Return && tokens[*i] != Token::End && tokens[*i] != Token::Elseif &&
tokens[*i] != Token::Else
{
stats.push(parse_stat(tokens, i)?);
}
let retstat =
if *i < tokens.len() && tokens[*i] == Token::Return { Some(parse_retstat(tokens, i)?) }
else { None };
return Ok(BlockNode { stats, retstat });
}
fn parse_stat(tokens: &Vec<Token>, i: &mut usize) -> Result<StatNode, &'static str>
{
if *i >= tokens.len()
{
return Err("Reached end of file while parsing stat");
}
match tokens[*i]
{
Token::Semicolon =>
{
*i += 1;
Ok(StatNode::Semicolon)
}
Token::Break =>
{
*i += 1;
Ok(StatNode::Break)
}
Token::Goto =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of stream but expected name for goto");
}
return if let Token::Name(name) = &tokens[*i]
{
*i += 1;
Ok(StatNode::Goto(name.clone()))
}
else
{
Err("Expecting name for goto")
};
}
Token::Do =>
{
*i += 1;
let body = parse_block(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::End
{
return Err("Missing 'end' for do block");
}
*i += 1;
return Ok(StatNode::Do(body));
}
Token::While =>
{
*i += 1;
let condition = parse_exp(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::Do
{
return Err("Expected 'do' after while condition")
}
*i += 1;
let body = parse_block(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::End
{
return Err("Missing 'end' for do block");
}
*i += 1;
return Ok(StatNode::While { condition, body });
}
Token::Repeat =>
{
*i += 1;
let body = parse_block(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::Until
{
return Err("Expected 'until' after repeat body");
}
*i += 1;
return Ok(StatNode::Repeat { condition: parse_exp(tokens, i)?, body });
}
Token::If =>
{
*i += 1;
let condition = parse_exp(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::Then
{
return Err("Expected 'then' after if condition");
}
*i += 1;
let body = parse_block(tokens, i)?;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing if");
}
let mut elseifs: Vec<ElseifNode> = Vec::new();
while tokens[*i] == Token::Elseif
{
*i += 1;
let elseif_condition = parse_exp(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::Then
{
return Err("Expected 'then' after elseif condition");
}
*i += 1;
elseifs.push(ElseifNode { condition: elseif_condition, body: parse_block(tokens, i)? });
}
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing if");
}
let else_ = if tokens[*i] == Token::Else
{
*i += 1;
Some(parse_block(tokens, i)?)
}
else
{
None
};
if *i >= tokens.len() || tokens[*i] != Token::End
{
return Err("Expected 'end' to close if");
}
*i += 1;
return Ok(StatNode::If { condition, body, elseifs, else_ });
}
Token::For =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing for");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing for after first name");
}
match tokens[*i]
{
Token::Equals =>
{
*i += 1;
let start = parse_exp(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::Comma
{
return Err("Expected ',' after 'for eq' start value");
}
*i += 1;
let end = parse_exp(tokens, i)?;
if *i >= tokens.len()
{
return Err("Reached end of tokens after end value in 'for eq'");
}
let change = if tokens[*i] == Token::Comma
{
*i += 1;
Some(parse_exp(tokens, i)?)
}
else
{
None
};
if *i >= tokens.len() || tokens[*i] != Token::Do
{
return Err("Expected 'do' after 'for eq' head");
}
*i += 1;
let body = parse_block(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::End
{
return Err("Expected 'end' to close 'for eq'");
}
return Ok(StatNode::ForEq { var: name.clone(), start, end, change, body });
}
Token::Comma =>
{
let mut names = Vec::from([name.clone()]);
while tokens[*i] == Token::Comma
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing 'for in' namelist");
}
if let Token::Name(next_name) = &tokens[*i]
{
names.push(next_name.clone());
}
else
{
return Err("Expected another name in 'for in' namelist");
}
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing 'for in' namelist after name");
}
}
if tokens[*i] != Token::In
{
return Err("Expected 'in' after 'for in' namelist");
}
*i += 1;
let exps = parse_explist(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::Do
{
return Err("Expected 'do' after 'for in' explist");
}
*i += 1;
let body = parse_block(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::End
{
return Err("Expected 'end' after 'for in' body");
}
*i += 1;
return Ok(StatNode::ForIn { vars: names, exps, body });
}
_ => Err("Unexpected token after first name in for")
}
}
else
{
return Err("Expected name after 'for'");
}
}
Token::Function =>
{
*i += 1;
let funcname = parse_funcname(tokens, i)?;
return Ok(StatNode::Function { name: funcname, body: parse_funcbody(tokens, i)? });
}
Token::Local =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing local");
}
if tokens[*i] == Token::Function
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing local function");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
return Ok(StatNode::LocalFunction { name: name.clone(), body: parse_funcbody(tokens, i)? });
}
else
{
return Err("Expected local function name");
}
}
let attnames = parse_attnamelist(tokens, i)?;
let initials = if *i < tokens.len() && tokens[*i] == Token::Equals
{
*i += 1;
Some(parse_explist(tokens, i)?)
}
else
{
None
};
return Ok(StatNode::Local { attnames, values: initials });
}
Token::ColonColon =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing label");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
if *i >= tokens.len() || tokens[*i] != Token::ColonColon
{
return Err("Expected '::' after name in label declaration");
}
*i += 1;
return Ok(StatNode::Label(name.clone()));
}
else
{
return Err("Expected a name after '::' in label declaration")
}
}
Token::Name(_) | Token::RoundOpen =>
{
// assignment or functioncall
let suffix_expression = parse_suffixexp(tokens, i)?;
match tokens[*i]
{
Token::Equals =>
{
*i += 1;
return Ok(StatNode::Assignment { lhs: VarlistNode { vars: Vec::from([suffix_expression_to_var(suffix_expression)?]) }, rhs: parse_explist(tokens, i)? });
}
Token::Comma =>
{
let mut vars = Vec::from([suffix_expression_to_var(suffix_expression)?]);
while tokens[*i] == Token::Comma
{
*i += 1;
vars.push(parse_var(tokens, i)?);
}
if *i >= tokens.len() || tokens[*i] != Token::Equals
{
return Err("Expected '=' for assignment");
}
*i += 1;
return Ok(StatNode::Assignment { lhs: VarlistNode { vars }, rhs: parse_explist(tokens, i)? });
}
_ =>
{
if suffix_expression.suffixes.is_empty()
{
println!("{:?} {} {:?}", tokens[*i], i, suffix_expression);
return Err("Expected function call but suffix is empty");
}
if let Some(SuffixexpSuffix::Args(_)) = suffix_expression.suffixes.last()
{
return Ok(StatNode::Functioncall(suffix_expression_to_functioncall(suffix_expression)?));
}
if let Some(SuffixexpSuffix::ArgsFirstArg(_, _)) = suffix_expression.suffixes.last()
{
return Ok(StatNode::Functioncall(suffix_expression_to_functioncall(suffix_expression)?));
}
else
{
println!("{:?} {} {:?}", tokens[*i], i, suffix_expression.suffixes.last());
return Err("Expected function call");
}
}
}
}
_ =>
{
println!("{:?} {:?} {:?}", tokens[*i - 2], tokens[*i - 1], tokens[*i]);
Err("Unexpected token while parsing stat")
}
}
}
fn suffix_expression_to_functioncall(suffixexp: SuffixexpNode) -> Result<FunctioncallNode, &'static str>
{
let mut new_suffixexp = suffixexp;
let last = new_suffixexp.suffixes.pop();
if let Some(SuffixexpSuffix::Args(args)) = last
{
return Ok(FunctioncallNode { function: new_suffixexp, object_arg: None, args });
}
if let Some(SuffixexpSuffix::ArgsFirstArg(first_arg, args)) = last
{
return Ok(FunctioncallNode { function: new_suffixexp, object_arg: Some(first_arg.clone()), args });
}
return Err("Cannot convert suffixexp to functioncall");
}
fn suffix_expression_to_var(suffixexp: SuffixexpNode) -> Result<VarNode, &'static str>
{
if suffixexp.suffixes.is_empty()
{
return if let SuffixexpFirstPart::Name(name) = suffixexp.first_part
{
Ok(VarNode::Name(name.clone()))
}
else
{
Err("Can only convert suffix exp without suffix to var if its first part is a name")
};
}
let mut new_suffixexp = suffixexp;
let last = new_suffixexp.suffixes.pop();
if let Some(SuffixexpSuffix::Dot(name)) = last
{
return Ok(VarNode::Member { value: new_suffixexp, name: name.clone() });
}
if let Some(SuffixexpSuffix::Indexed(index)) = last
{
return Ok(VarNode::Indexed { value: new_suffixexp, index: index });
}
return Err("Cannot convert suffixexp to var");
}
fn parse_var(tokens: &Vec<Token>, i: &mut usize) -> Result<VarNode, &'static str>
{
todo!()
}
fn parse_args(tokens: &Vec<Token>, i: &mut usize) -> Result<ArgsNode, &'static str>
{
if *i > tokens.len()
{
return Err("Reached end of tokens while parsing args");
}
match &tokens[*i]
{
Token::RoundOpen =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while paring bracketed args");
}
if tokens[*i] == Token::RoundClosed
{
*i += 1;
return Ok(ArgsNode::Bracketed(None));
}
let exps = parse_explist(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::RoundClosed
{
println!("|{:?}|{}|{:?}|", tokens[*i], i, exps);
return Err("Expected ')' to close bracketed args");
}
*i += 1;
return Ok(ArgsNode::Bracketed(Some(exps)));
}
Token::CurlyOpen =>
{
return Ok(ArgsNode::Tableconstructor(parse_tableconstructor(tokens, i)?));
}
Token::StringLiteral(name) =>
{
*i += 1;
return Ok(ArgsNode::Literal(name.clone()));
}
_ => return Err("Unexpected token while parsing args")
}
}
fn parse_suffixexp(tokens: &Vec<Token>, i: &mut usize) -> Result<SuffixexpNode, &'static str>
{
// primaryexp { '.' 'Name' | '[' exp']' | ':' 'Name' args | args }
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing suffixexp");
}
let first_part = match &tokens[*i]
{
Token::Name(name) =>
{
*i += 1;
SuffixexpFirstPart::Name(name.clone())
},
Token::RoundOpen =>
{
*i += 1;
let ret = SuffixexpFirstPart::BracketedExpr(parse_exp(tokens, i)?);
if *i >= tokens.len() || tokens[*i] != Token::RoundClosed
{
return Err("Expected ')' to close bracketed primary expression");
}
*i += 1;
ret
}
_ => return Err("Unexpected token as first part of suffixexp")
};
let mut suffixes = Vec::new();
while *i < tokens.len()
{
match tokens[*i]
{
Token::Dot =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens but expected name for dotted suffix expression");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
suffixes.push(SuffixexpSuffix::Dot(name.clone()));
}
else
{
return Err("Expected name for dotted suffix expression");
}
}
Token::SquareOpen =>
{
*i += 1;
suffixes.push(SuffixexpSuffix::Indexed(parse_exp(tokens, i)?));
if *i >= tokens.len() || tokens[*i] != Token::SquareClosed
{
return Err("Expected ']' to close indexed suffix expression");
}
*i += 1;
}
Token::Colon =>
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens but expected name for dotted suffix expression");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
suffixes.push(SuffixexpSuffix::ArgsFirstArg(name.clone(), parse_args(tokens, i)?));
}
else
{
return Err("Expected name for dotted suffix expression");
}
}
Token::RoundOpen | Token::CurlyOpen | Token::StringLiteral(_) =>
{
suffixes.push(SuffixexpSuffix::Args(parse_args(tokens, i)?));
}
_ => break,
}
}
return Ok(SuffixexpNode { first_part, suffixes });
}
fn parse_retstat(tokens: &Vec<Token>, i: &mut usize) -> Result<RetstatNode, &'static str>
{
if *i >= tokens.len() || tokens[*i] != Token::Return
{
return Err("Expected 'return' to start retstat");
}
*i += 1;
if *i >= tokens.len() || tokens[*i] == Token::Semicolon || tokens[*i] == Token::Else || tokens[*i] == Token::Elseif ||
tokens[*i] == Token::End
{
if *i < tokens.len() && tokens[*i] == Token::Semicolon
{
*i += 1;
}
return Ok(RetstatNode { values: None });
}
let values = parse_explist(tokens, i)?;
if *i < tokens.len() && tokens[*i] == Token::Semicolon
{
*i += 1;
}
return Ok(RetstatNode { values: Some(values) });
}
fn parse_exp(tokens: &Vec<Token>, i: &mut usize) -> Result<ExpNode, &'static str>
{
let lhs = parse_exp_primary(tokens, i)?;
return parse_exp_precedence(tokens, i, lhs, 0);
}
fn get_precedence(token: &Token) -> Result<u8, &'static str>
{
match token
{
Token::Or => Ok(2),
Token::And => Ok(4),
Token::Lt | Token::Gt | Token::LtEquals | Token::GtEquals | Token::TildeEquals | Token::EqualsEquals => Ok(6),
Token::Pipe => Ok(8),
Token::Tilde => Ok(10),
Token::Ampersand => Ok(12),
Token::LtLt | Token::GtGt => Ok(14),
Token::DotDot => Ok(16),
Token::Plus | Token::Minus => Ok(18),
Token::Star | Token::Slash | Token::SlashSlash | Token::Percent => Ok(20),
Token::Caret => Ok(22),
_ => Err("Tried to get precedence for unknown operator"),
}
}
fn get_binop(token: &Token) -> Result<BinopType, &'static str>
{
match token
{
Token::Or => Ok(BinopType::LogicalOr),
Token::And => Ok(BinopType::LocicalAnd),
Token::Lt => Ok(BinopType::Lt),
Token::Gt => Ok(BinopType::Lt),
Token::LtEquals => Ok(BinopType::LtEquals),
Token::GtEquals => Ok(BinopType::GtEquals),
Token::TildeEquals => Ok(BinopType::NotEquals),
Token::EqualsEquals => Ok(BinopType::Equals),
Token::Pipe => Ok(BinopType::BinaryOr),
Token::Tilde => Ok(BinopType::BinaryNot),
Token::Ampersand => Ok(BinopType::BinaryAnd),
Token::DotDot => Ok(BinopType::Concat),
Token::Plus => Ok(BinopType::Add),
Token::Minus => Ok(BinopType::Sub),
Token::Star => Ok(BinopType::Mul),
Token::Slash => Ok(BinopType::Div),
Token::SlashSlash => Ok(BinopType::IntDiv),
Token::Percent => Ok(BinopType::Mod),
Token::Caret => Ok(BinopType::Exp),
_ =>
{
println!("{:?}", token);
Err("Tried to get binop type for unknown operator")
}
}
}
fn is_binop(token: &Token) -> bool
{
match token
{
Token::Or | Token::And | Token::Lt | Token::Gt | Token::LtEquals | Token::GtEquals | Token::TildeEquals | Token::EqualsEquals |
Token::Pipe | Token::Tilde | Token::Ampersand | Token::LtLt | Token::GtGt | Token::DotDot | Token::Plus | Token::Minus |
Token::Star | Token::Slash | Token::SlashSlash | Token::Percent | Token::Caret =>
{
true
}
_ => false
}
}
fn is_right_associative(token: &Token) -> bool
{
return token == &Token::DotDot || token == &Token::Caret;
}
fn parse_exp_precedence(tokens: &Vec<Token>, i: &mut usize, lhs: ExpNode, min_precedence: u8) -> Result<ExpNode, &'static str>
{
let mut lhs = lhs;
while *i < tokens.len() && is_binop(&tokens[*i])
{
let precedence = get_precedence(&tokens[*i])?;
if precedence < min_precedence
{
break;
}
let op = get_binop(&tokens[*i])?;
*i += 1;
let mut rhs = parse_exp_primary(tokens, i)?;
while *i < tokens.len() && is_binop(&tokens[*i]) && (get_precedence(&tokens[*i])? > precedence ||
(get_precedence(&tokens[*i])? == precedence && is_right_associative(&tokens[*i])))
{
rhs = parse_exp_precedence(tokens, i, rhs, precedence + if precedence == get_precedence(&tokens[*i])? {0} else {1})?;
}
lhs = ExpNode::Binop { lhs: Box::new(lhs), op, rhs: Box::new(rhs) };
}
return Ok(lhs);
}
fn parse_exp_primary(tokens: &Vec<Token>, i: &mut usize) -> Result<ExpNode, &'static str>
{
if *i >= tokens.len()
{
return Err("Reached end of tokens but expected primary expression");
}
match &tokens[*i]
{
Token::Nil =>
{
*i += 1;
Ok(ExpNode::Nil)
},
Token::True =>
{
*i += 1;
Ok(ExpNode::True)
},
Token::False =>
{
*i += 1;
Ok(ExpNode::False)
},
Token::Numeral(number_str) =>
{
*i += 1;
Ok(ExpNode::Numeral(number_str.parse::<f64>().map_err(|_| "Could not parse number")?))
},
Token::StringLiteral(string) =>
{
*i += 1;
Ok(ExpNode::LiteralString(string.clone()))
},
Token::DotDotDot =>
{
*i += 1;
Ok(ExpNode::Varargs)
},
Token::Function =>
{
*i += 1;
Ok(ExpNode::Functiondef(parse_funcbody(tokens, i)?))
}
Token::CurlyOpen => Ok(ExpNode::Tableconstructor(parse_tableconstructor(tokens, i)?)),
Token::Minus =>
{
Ok(ExpNode::Unop(UnopType::Minus, Box::new(parse_exp(tokens, i)?)))
}
Token::Hash =>
{
Ok(ExpNode::Unop(UnopType::Length, Box::new(parse_exp(tokens, i)?)))
}
Token::Not =>
{
Ok(ExpNode::Unop(UnopType::LogicalNot, Box::new(parse_exp(tokens, i)?)))
}
Token::Tilde =>
{
Ok(ExpNode::Unop(UnopType::BinaryNot, Box::new(parse_exp(tokens, i)?)))
}
_ => Ok(ExpNode::Suffixexp(Box::new(parse_suffixexp(tokens, i)?))),
}
}
fn parse_tableconstructor(tokens: &Vec<Token>, i: &mut usize) -> Result<TableconstructorNode, &'static str>
{
todo!()
}
fn parse_explist(tokens: &Vec<Token>, i: &mut usize) -> Result<ExplistNode, &'static str>
{
let mut exps: Vec<ExpNode> = Vec::from([parse_exp(tokens, i)?]);
while *i < tokens.len() && tokens[*i] == Token::Comma
{
*i += 1;
exps.push(parse_exp(tokens, i)?);
}
return Ok(ExplistNode { exps });
}
fn parse_funcname(tokens: &Vec<Token>, i: &mut usize) -> Result<FuncnameNode, &'static str>
{
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing funcname");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
let mut dotted_names = Vec::new();
while *i < tokens.len() && tokens[*i] == Token::Dot
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing dotted part of funcname");
}
if let Token::Name(dotted_name) = &tokens[*i]
{
*i += 1;
dotted_names.push(dotted_name.clone());
}
else
{
return Err("Expected name in dotted funcname");
}
}
let first_arg = if *i < tokens.len() && tokens[*i] == Token::Colon
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing funcname first arg");
}
if let Token::Name(arg_name) = &tokens[*i]
{
*i += 1;
Some(arg_name.clone())
}
else
{
return Err("Expected name of first arg in funcname");
}
}
else
{
None
};
return Ok(FuncnameNode { name: name.clone(), dotted_names, first_arg });
}
else
{
return Err("Expected func name");
}
}
fn parse_funcbody(tokens: &Vec<Token>, i: &mut usize) -> Result<FuncbodyNode, &'static str>
{
if *i >= tokens.len() || tokens[*i] != Token::RoundOpen
{
return Err("Expected '(' to start funcbody");
}
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing funcbody parlist");
}
let pars = if tokens[*i] == Token::RoundClosed
{
*i += 1;
None
}
else
{
let ret = Some(parse_parlist(tokens, i)?);
if *i >= tokens.len() || tokens[*i] != Token::RoundClosed
{
return Err("Expected ')' to close funcbody parlist");
}
*i += 1;
ret
};
let block = parse_block(tokens, i)?;
if *i >= tokens.len() || tokens[*i] != Token::End
{
println!("{:?}", &tokens[(*i - 10)..(*i + 10)]);
return Err("Expected 'end' to close funcbody");
}
*i += 1;
return Ok(FuncbodyNode { pars, body: block });
}
fn parse_parlist(tokens: &Vec<Token>, i: &mut usize) -> Result<ParlistNode, &'static str>
{
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing parlist");
}
if tokens[*i] == Token::DotDotDot
{
*i += 1;
return Ok(ParlistNode { names: Vec::new(), has_varargs: true });
}
let first_name = if let Token::Name(name) = &tokens[*i]
{
*i += 1;
name.clone()
}
else
{
return Err("Expected name to start parlist");
};
let mut names = Vec::from([first_name]);
let mut has_varargs = false;
while *i < tokens.len() && tokens[*i] == Token::Comma
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens while parsing parlist name list");
}
match &tokens[*i]
{
Token::Name(name) =>
{
*i += 1;
names.push(name.clone());
}
Token::DotDotDot =>
{
*i += 1;
has_varargs = true;
break;
}
_ => return Err("Unexpected token while parsing parlist name list"),
}
}
return Ok(ParlistNode { names, has_varargs });
}
fn parse_attnamelist(tokens: &Vec<Token>, i: &mut usize) -> Result<AttnamelistNode, &'static str>
{
let mut attnames: Vec<AttnameNode> = Vec::from([parse_attname(tokens, i)?]);
while *i < tokens.len() && tokens[*i] == Token::Comma
{
*i += 1;
attnames.push(parse_attname(tokens, i)?);
}
return Ok(AttnamelistNode { attnames });
}
fn parse_attname(tokens: &Vec<Token>, i: &mut usize) -> Result<AttnameNode, &'static str>
{
if *i >= tokens.len()
{
return Err("Reached end of tokens but expected name for attrib name");
}
if let Token::Name(name) = &tokens[*i]
{
*i += 1;
let attribute = if *i < tokens.len() && tokens[*i] == Token::Lt
{
*i += 1;
if *i >= tokens.len()
{
return Err("Reached end of tokens but expected attribute");
}
if let Token::Name(attrib) = &tokens[*i]
{
*i += 1;
if *i >= tokens.len() || tokens[*i] != Token::Gt
{
return Err("Exptected '>' to close attribute name");
}
Some(attrib.clone())
}
else
{
return Err("Expected attribute in attrib name");
}
}
else
{
None
};
return Ok(AttnameNode { name: name.clone(), attribute });
}
else
{
return Err("Expected name for attrib name");
}
}
//===============================================================================================================================================
//===============================================================================================================================================
//===============================================================================================================================================
//===============================================================================================================================================
//===============================================================================================================================================
//===============================================================================================================================================
#[derive(Debug, Clone, Copy)]
pub struct Node
{
}
#[derive(Debug, Clone, Copy)]
pub struct AmbiguousNode
{
}
pub fn cyk(tokens: Vec<Token>) -> Result<ChunkNode, &'static str>
{
let r = NONTERMINAL_NAMES.len();
let n = tokens.len();
macro_rules! index {
($x:expr, $y:expr, $z:expr) => {
($x + $y * n + ($z as usize) * n * n)
};
}
let mut p = vec![false; n * n * r];
//let mut back: Vec<Vec<(usize, u8, u8)>> = vec![Vec::new(); n * n * r];
println!("{n}, {r}, {}", p.len());
for s in 0..n
{
for (index, token) in TERMINAL_RULES
{
if let Token::Name(_) = tokens[s]
{
if let Token::Name(_) = token
{
p[index!(0, s, index)] = true
}
}
else if let Token::StringLiteral(_) = tokens[s]
{
if let Token::StringLiteral(_) = token
{
p[index!(0, s, index)] = true
}
}
else if let Token::Numeral(_) = tokens[s]
{
if let Token::Numeral(_) = token
{
p[index!(0, s, index)] = true
}
}
else if token == tokens[s]
{
p[index!(0, s, index)] = true
}
}
}
println!("Done initializing");
for l in 2..=n
{
for s in 1..=(n - l + 1)
{
for _p in 1..=(l-1)
{
for &(a, b, c) in &NONTERMINAL_RULES
{
if p[index!(_p - 1, s - 1, b)] && p[index!(l - _p - 1, s + _p - 1, c)]
{
let index = index!(l - 1, s - 1, a);
p[index] = true;
/* if !back[index].contains(&(_p, b, c))
{
back[index].push((_p, b, c));
}*/
}
}
}
}
println!("{l}");
}
let start_index = NONTERMINAL_NAMES.iter().position(|x| x == &"S_0").expect("no start index found");
if p[index!(n - 1, 0, start_index)]
{
println!("Is part of the language");
todo!()
//return Ok(disambiguate(traverse_back(back, tokens, n, 1, start_index)));
}
else
{
return Err("Input is not part of the language")
}
}
fn traverse_back(back: Vec<Vec<(usize, u8, u8)>>, tokens: Vec<Token>, l: usize, s: usize, a: usize) -> AmbiguousNode
{
todo!()
}
fn disambiguate(root: AmbiguousNode) -> Node
{
todo!()
}
-1210
View File
@@ -1,1210 +0,0 @@
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
pub enum Token
{
Name(String),
And, Break, Do, Else, Elseif, End,
False, For, Function, Goto, If, In,
Local, Nil, Not, Or, Repeat, Return,
Then, True, Until, While,
Plus, Minus, Star, Slash, Percent, Caret, Hash,
Ampersand, Tilde, Pipe, LtLt, GtGt, SlashSlash,
EqualsEquals, TildeEquals, LtEquals, GtEquals, Lt, Gt, Equals,
RoundOpen, RoundClosed, CurlyOpen, CurlyClosed, SquareOpen, SquareClosed, ColonColon,
Semicolon, Colon, Comma, Dot, DotDot, DotDotDot,
Numeral(String),
StringLiteral(String),
}
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum TokenizerState
{
Start,
Quote, SingleQuote, Name, Number, Zero,
A, B, D, E, F, G, I, L, N, O, R, T, U, W,
Plus, Minus, Star, Slash, Percent, Caret, Hash,
Ampersand, Tilde, Pipe, Lt, Gt, Equals, RoundOpen, RoundClosed, CurlyOpen, CurlyClosed, SquareOpen, SquareClosed,
Colon, Semicolon, Comma, Dot,
An, Br, Do, El, En, Fa, Fo, Fu, Go, If, In, Lo, Ni, No, Or, Re, Th, Tr, Un, Wh,
LtLt, GtGt, SlashSlash, EqualsEquals, TildeEquals, LtEquals, GtEquals, ColonColon, DotDot,
SmallCommentStart, QuoteBackslash, SingleQuoteBackslash, String, HexNumberX, ExpNumber,
And, Bre, Els, End, Fal, For, Fun, Got, Loc, Nil, Not, Rep, Ret, The, Tru, Unt, Whi,
DotDotDot, HexNumber, QuoteBackslashZ, SingleQuoteBackslashZ,
BigCommentLongBracketStart, SmallComment,
Brea, Else, Fals, Func, Goto, Loca, Repe, Retu, Then, True, Unti, Whil, HexExpNumber,
BigComment, BigCommentLongBracketEnd,
Break, Elsei, False, Funct, Local, Repea, Retur, Until, While,
Elseif, Functi, Repeat, Return,
Functio,
Function,
}
fn tokenize_update_index_and_state(last_index: &mut i32, index: usize, state: &mut TokenizerState, new_state: TokenizerState)
{
*last_index = index as i32;
*state = new_state;
}
fn tokenize_terminal_no_str(last_index: &mut i32, index: usize, token: &mut Option<Token>, state: &mut TokenizerState, new_token: Option<Token>, new_state: TokenizerState)
{
tokenize_update_index_and_state(last_index, index, state, new_state);
*token = new_token;
}
fn tokenize_terminal_no_token(last_index: &mut i32, index: usize, state: &mut TokenizerState, new_state: TokenizerState, token_str: &mut String, ch: char)
{
tokenize_update_index_and_state(last_index, index, state, new_state);
token_str.push(ch);
}
fn tokenize_terminal(last_index: &mut i32, index: usize, token: &mut Option<Token>, state: &mut TokenizerState, new_token: Option<Token>, new_state: TokenizerState, token_str: &mut String, ch: char)
{
tokenize_terminal_no_str(last_index, index, token, state, new_token, new_state);
token_str.push(ch);
}
fn tokenize_backtrack(last_index: &mut i32, index: &mut usize, tokens: &mut Vec<Token>, token: &mut Option<Token>, token_str: &mut String, state: &mut TokenizerState) -> Result<(), &'static str>
{
return tokenize_backtrack_custom_token(last_index, index, tokens, token, token_str, state, token.clone().unwrap());
}
fn tokenize_backtrack_name(last_index: &mut i32, index: &mut usize, tokens: &mut Vec<Token>, token: &mut Option<Token>, token_str: &mut String, state: &mut TokenizerState) -> Result<(), &'static str>
{
if *last_index == -1 || token.is_none()
{
println!("{}|{}|{:?} | {:?}", last_index, index, token, tokens);
return Err("Lexerr");
}
*index = *last_index as usize;
*last_index = -1;
tokens.push(Token::Name(token_str.clone()));
*token = None;
token_str.clear();
*state = TokenizerState::Start;
return Ok(());
}
fn tokenize_backtrack_custom_token(last_index: &mut i32, index: &mut usize, tokens: &mut Vec<Token>, token: &mut Option<Token>, token_str: &mut String, state: &mut TokenizerState, new_token: Token) -> Result<(), &'static str>
{
if *last_index == -1 || token.is_none()
{
println!("{}|{}|{:?} | {:?}", last_index, index, token, tokens);
return Err("Lexerr");
}
*index = *last_index as usize;
*last_index = -1;
tokens.push(new_token);
*token = None;
token_str.clear();
*state = TokenizerState::Start;
return Ok(());
}
fn tokenize_alphanumeric_nonstart(last_index: &mut i32, index: &mut usize, tokens: &mut Vec<Token>, token: &mut Option<Token>, token_str: &mut String, state: &mut TokenizerState, ch: char) -> Result<(), &'static str>
{
if ch.is_ascii_alphanumeric() || ch == '_'
{
tokenize_update_index_and_state(last_index, *index, state, TokenizerState::Name);
token_str.push(ch);
}
else
{
tokenize_backtrack_name(last_index, index, tokens, token, token_str, state)?;
}
return Ok(());
}
fn tokenize_alphanumeric_nonstart_custom(last_index: &mut i32, index: &mut usize, tokens: &mut Vec<Token>, token: &mut Option<Token>, token_str: &mut String, state: &mut TokenizerState, ch: char, new_token: Token) -> Result<(), &'static str>
{
if ch.is_ascii_alphanumeric() || ch == '_'
{
tokenize_update_index_and_state(last_index, *index, state, TokenizerState::Name);
token_str.push(ch);
}
else
{
tokenize_backtrack_custom_token(last_index, index, tokens, token, token_str, state, new_token)?;
}
return Ok(());
}
fn tokenize_char(state: &mut TokenizerState, ch: char, last_index: &mut i32, index: &mut usize, token: &mut Option<Token>, token_str: &mut String, tokens: &mut Vec<Token>, long_bracket_level: &mut u32) -> Result<(), &'static str>
{
match state
{
TokenizerState::Start =>
{
match ch
{
'-' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Minus), TokenizerState::Minus),
'a' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("a".to_string())), TokenizerState::A, token_str, ch),
'b' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("b".to_string())), TokenizerState::B, token_str, ch),
'd' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("d".to_string())), TokenizerState::D, token_str, ch),
'e' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("e".to_string())), TokenizerState::E, token_str, ch),
'f' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("f".to_string())), TokenizerState::F, token_str, ch),
'i' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("i".to_string())), TokenizerState::I, token_str, ch),
'g' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("g".to_string())), TokenizerState::G, token_str, ch),
'l' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("l".to_string())), TokenizerState::L, token_str, ch),
'n' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("n".to_string())), TokenizerState::N, token_str, ch),
'o' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("o".to_string())), TokenizerState::O, token_str, ch),
'r' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("r".to_string())), TokenizerState::R, token_str, ch),
't' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("t".to_string())), TokenizerState::T, token_str, ch),
'u' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("u".to_string())), TokenizerState::U, token_str, ch),
'w' => tokenize_terminal(last_index, *index, token, state, Some(Token::Name("w".to_string())), TokenizerState::W, token_str, ch),
',' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Comma), TokenizerState::Comma),
'=' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Equals), TokenizerState::Equals),
'(' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::RoundOpen), TokenizerState::RoundOpen),
')' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::RoundClosed), TokenizerState::RoundClosed),
'.' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Dot), TokenizerState::Dot),
':' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Colon), TokenizerState::Colon),
'{' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::CurlyOpen), TokenizerState::CurlyOpen),
'}' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::CurlyClosed), TokenizerState::CurlyClosed),
'[' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::SquareOpen), TokenizerState::SquareOpen),
']' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::SquareClosed), TokenizerState::SquareClosed),
'+' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Plus), TokenizerState::Plus),
'~' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Tilde), TokenizerState::Tilde),
'>' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Gt), TokenizerState::Gt),
'<' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Lt), TokenizerState::Lt),
'#' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Hash), TokenizerState::Hash),
'|' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Pipe), TokenizerState::Pipe),
'&' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Ampersand), TokenizerState::Ampersand),
'%' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Percent), TokenizerState::Percent),
'*' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Star), TokenizerState::Star),
'/' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Slash), TokenizerState::Slash),
';' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Semicolon), TokenizerState::Semicolon),
'^' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::Caret), TokenizerState::Caret),
'0' => tokenize_terminal(last_index, *index, token, state, Some(Token::Numeral("0".to_string())), TokenizerState::Zero, token_str, ch),
'"' =>
{
*token = None;
*state = TokenizerState::Quote;
}
'\'' =>
{
*token = None;
*state = TokenizerState::SingleQuote;
}
_ =>
{
if ch.is_whitespace() { }
else if ch.is_ascii_alphabetic() || ch == '_'
{
tokenize_terminal(last_index, *index, token, state, Some(Token::Name(ch.to_string())), TokenizerState::Name, token_str, ch);
}
else if ch.is_numeric() && ch.is_ascii()
{
tokenize_terminal(last_index, *index, token, state, Some(Token::Numeral(ch.to_string())), TokenizerState::Number, token_str, ch);
}
else
{
todo!("State {:?}, Char {}", state, ch);
}
}
}
}
TokenizerState::Quote =>
{
match ch
{
'\\' =>
{
*state = TokenizerState::QuoteBackslash;
}
'"' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::StringLiteral(token_str.clone())), TokenizerState::String),
_ =>
{
token_str.push(ch);
}
}
}
TokenizerState::QuoteBackslash =>
{
match ch
{
'a' =>
{
token_str.push('\u{0007}');
*state = TokenizerState::Quote;
}
'b' =>
{
token_str.push('\u{0008}');
*state = TokenizerState::Quote;
}
't' =>
{
token_str.push('\t');
*state = TokenizerState::Quote;
}
'n' | '\n' =>
{
token_str.push('\n');
*state = TokenizerState::Quote;
}
'v' =>
{
token_str.push('\u{000b}');
*state = TokenizerState::Quote;
}
'f' =>
{
token_str.push('\u{000c}');
*state = TokenizerState::Quote;
}
'r' =>
{
token_str.push('\r');
*state = TokenizerState::Quote;
}
'\\' =>
{
token_str.push('\\');
*state = TokenizerState::Quote;
}
'"' =>
{
token_str.push('\"');
*state = TokenizerState::Quote;
}
'\'' =>
{
token_str.push('\'');
*state = TokenizerState::Quote;
}
'z' =>
{
*state = TokenizerState::QuoteBackslashZ;
}
_ => return Err("Unknown escape sequence"),
}
}
TokenizerState::QuoteBackslashZ =>
{
match ch
{
'\\' =>
{
*state = TokenizerState::QuoteBackslash;
}
'"' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::StringLiteral(token_str.clone())), TokenizerState::String),
_ =>
{
if !ch.is_whitespace()
{
token_str.push(ch);
*state = TokenizerState::Quote;
}
}
}
}
TokenizerState::SingleQuote =>
{
match ch
{
'\\' =>
{
*state = TokenizerState::SingleQuoteBackslash;
}
'\'' =>
{
*last_index = *index as i32;
*token = Some(Token::StringLiteral(token_str.clone()));
*state = TokenizerState::String;
}
_ =>
{
token_str.push(ch);
}
}
}
TokenizerState::SingleQuoteBackslash =>
{
match ch
{
'a' =>
{
token_str.push('\u{0007}');
*state = TokenizerState::SingleQuote;
}
'b' =>
{
token_str.push('\u{0008}');
*state = TokenizerState::SingleQuote;
}
't' =>
{
token_str.push('\t');
*state = TokenizerState::SingleQuote;
}
'n' | '\n' =>
{
token_str.push('\n');
*state = TokenizerState::SingleQuote;
}
'v' =>
{
token_str.push('\u{000b}');
*state = TokenizerState::SingleQuote;
}
'f' =>
{
token_str.push('\u{000c}');
*state = TokenizerState::SingleQuote;
}
'r' =>
{
token_str.push('\r');
*state = TokenizerState::SingleQuote;
}
'\\' =>
{
token_str.push('\\');
*state = TokenizerState::SingleQuote;
}
'"' =>
{
token_str.push('\"');
*state = TokenizerState::SingleQuote;
}
'\'' =>
{
token_str.push('\'');
*state = TokenizerState::SingleQuote;
}
'z' =>
{
*state = TokenizerState::SingleQuoteBackslashZ;
}
_ => return Err("Unknown escape sequence"),
}
}
TokenizerState::SingleQuoteBackslashZ =>
{
match ch
{
'\\' =>
{
*state = TokenizerState::SingleQuoteBackslash;
}
'\'' =>
{
*last_index = *index as i32;
*token = Some(Token::StringLiteral(token_str.clone()));
*state = TokenizerState::String;
}
_ =>
{
if !ch.is_whitespace()
{
token_str.push(ch);
*state = TokenizerState::SingleQuote;
}
}
}
}
TokenizerState::String =>
{
let content = token_str.clone();
tokenize_backtrack_custom_token(last_index, index, tokens, token, token_str, state, Token::StringLiteral(content))?;
}
TokenizerState::Name => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
TokenizerState::Zero =>
{
match ch
{
'x' =>
{
token_str.push(ch);
*token = None;
*state = TokenizerState::HexNumberX;
}
_ =>
{
if ch.is_numeric() && ch.is_ascii()
{
*last_index = *index as i32;
token_str.push(ch);
*token = Some(Token::Numeral(token_str.clone()));
}
else
{
tokenize_backtrack(last_index, index, tokens, token, token_str, state)?;
}
}
}
}
TokenizerState::HexNumberX =>
{
if ch.is_ascii() && ch.is_numeric() || match ch
{
'A'..='F' | 'a'..='f' => true,
_ => false,
}
{
*last_index = *index as i32;
token_str.push(ch);
*token = Some(Token::Numeral(token_str.clone()));
*state = TokenizerState::HexNumber;
}
else
{
tokenize_backtrack(last_index, index, tokens, token, token_str, state)?;
}
}
TokenizerState::HexNumber =>
{
match ch
{
'p' =>
{
token_str.push(ch);
*token = None;
*state = TokenizerState::HexExpNumber;
}
_ =>
{
if ch.is_ascii() && ch.is_numeric() || match ch
{
'A'..='F' | 'a'..='f' => true,
_ => false,
}
{
*last_index = *index as i32;
token_str.push(ch);
*token = Some(Token::Numeral(token_str.clone()));
}
else
{
tokenize_backtrack(last_index, index, tokens, token, token_str, state)?;
}
}
}
}
TokenizerState::Number =>
{
match ch
{
'e' =>
{
token_str.push(ch);
*token = None;
*state = TokenizerState::ExpNumber;
}
_ =>
{
if ch.is_numeric() && ch.is_ascii()
{
*last_index = *index as i32;
token_str.push(ch);
*token = Some(Token::Numeral(token_str.clone()));
}
else
{
tokenize_backtrack(last_index, index, tokens, token, token_str, state)?;
}
}
}
}
TokenizerState::Comma | TokenizerState::RoundOpen | TokenizerState::RoundClosed |
TokenizerState::CurlyOpen | TokenizerState::CurlyClosed | TokenizerState::Plus |
TokenizerState::TildeEquals | TokenizerState::EqualsEquals | TokenizerState::Hash |
TokenizerState::GtEquals | TokenizerState::LtEquals | TokenizerState::SquareOpen |
TokenizerState::SquareClosed | TokenizerState::Pipe | TokenizerState::Ampersand |
TokenizerState::Percent | TokenizerState::Star | TokenizerState::Semicolon |
TokenizerState::Caret | TokenizerState::DotDotDot | TokenizerState::GtGt |
TokenizerState::LtLt | TokenizerState::SlashSlash => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
TokenizerState::Tilde =>
{
match ch
{
'=' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::TildeEquals), TokenizerState::TildeEquals),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Gt =>
{
match ch
{
'>' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::GtGt), TokenizerState::GtGt),
'=' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::GtEquals), TokenizerState::GtEquals),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Lt =>
{
match ch
{
'>' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::LtLt), TokenizerState::LtLt),
'=' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::LtEquals), TokenizerState::LtEquals),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Slash =>
{
match ch
{
'/' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::SlashSlash), TokenizerState::SlashSlash),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Dot =>
{
match ch
{
'.' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::DotDot), TokenizerState::DotDot),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::DotDot =>
{
match ch
{
'.' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::DotDotDot), TokenizerState::DotDotDot),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Colon =>
{
match ch
{
':' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::ColonColon), TokenizerState::ColonColon),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Equals =>
{
match ch
{
'=' => tokenize_terminal_no_str(last_index, *index, token, state, Some(Token::EqualsEquals), TokenizerState::EqualsEquals),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::Minus =>
{
match ch
{
'-' => tokenize_terminal_no_str(last_index, *index, token, state, None, TokenizerState::SmallCommentStart),
_ => tokenize_backtrack(last_index, index, tokens, token, token_str, state)?,
}
}
TokenizerState::SmallCommentStart =>
{
match ch
{
'[' =>
{
*token = None;
*state = TokenizerState::BigCommentLongBracketStart;
}
'\n' =>
{
*state = TokenizerState::Start;
*last_index = -1;
}
_ =>
{
*state = TokenizerState::SmallComment;
}
}
}
TokenizerState::SmallComment =>
{
match ch
{
'\n' =>
{
*state = TokenizerState::Start;
*last_index = -1;
}
_ => { }
}
}
TokenizerState::BigCommentLongBracketStart =>
{
match ch
{
'=' =>
{
*long_bracket_level += 1;
}
'[' =>
{
*state = TokenizerState::BigComment;
}
_ => return Err("Malformed long bracket at the beginning of a big comment"),
}
}
TokenizerState::BigComment =>
{
match ch
{
']' =>
{
*state = TokenizerState::BigCommentLongBracketEnd;
}
_ => { }
}
}
TokenizerState::BigCommentLongBracketEnd =>
{
match ch
{
'=' =>
{
if *long_bracket_level == 0
{
return Err("Long bracket level too big when ending big comment");
}
*long_bracket_level -= 1;
}
']' =>
{
if *long_bracket_level != 0
{
return Err("Long bracket level too small when ending big comment");
}
*state = TokenizerState::Start;
}
_ => return Err("Malformed long bracket when ending big comment"),
}
}
TokenizerState::A =>
{
match ch
{
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::An, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::An =>
{
match ch
{
'd' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::And, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::And => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::And)?,
TokenizerState::W =>
{
match ch
{
'h' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Wh, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Wh =>
{
match ch
{
'i' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Whi, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Whi =>
{
match ch
{
'l' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Whil, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Whil =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::While, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::While => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::While)?,
TokenizerState::B =>
{
match ch
{
'r' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Br, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Br =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Bre, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Bre =>
{
match ch
{
'a' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Brea, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Brea =>
{
match ch
{
'k' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Break, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Break => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Break)?,
TokenizerState::G =>
{
match ch
{
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Go, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Go =>
{
match ch
{
't' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Got, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Got =>
{
match ch
{
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Goto, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Goto => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Goto)?,
TokenizerState::R =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Re, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Re =>
{
match ch
{
't' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Ret, token_str, ch),
'p' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Rep, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Ret =>
{
match ch
{
'u' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Retu, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Retu =>
{
match ch
{
'r' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Retur, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Retur =>
{
match ch
{
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Return, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Return => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Return)?,
TokenizerState::Rep =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Repe, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Repe =>
{
match ch
{
'a' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Repea, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Repea =>
{
match ch
{
't' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Repeat, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Repeat => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Repeat)?,
TokenizerState::N =>
{
match ch
{
'i' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Ni, token_str, ch),
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::No, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::No =>
{
match ch
{
't' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Not, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Not => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Not)?,
TokenizerState::Ni =>
{
match ch
{
'l' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Nil, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Nil => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Nil)?,
TokenizerState::T =>
{
match ch
{
'h' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Th, token_str, ch),
'r' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Tr, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Th =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::The, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::The =>
{
match ch
{
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Then, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Then => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Then)?,
TokenizerState::Tr =>
{
match ch
{
'u' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Tru, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Tru =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::True, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::True => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::True)?,
TokenizerState::E =>
{
match ch
{
'l' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::El, token_str, ch),
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::En, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::En =>
{
match ch
{
'd' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::End, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::End => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::End)?,
TokenizerState::El =>
{
match ch
{
's' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Els, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Els =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Else, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Else =>
{
match ch
{
'i' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Elsei, token_str, ch),
_ => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Else)?,
}
}
TokenizerState::Elsei =>
{
match ch
{
'f' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Elseif, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Elseif => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Elseif)?,
TokenizerState::O =>
{
match ch
{
'r' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Or, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Or => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Or)?,
TokenizerState::D =>
{
match ch
{
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Do, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Do => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Do)?,
TokenizerState::I =>
{
match ch
{
'f' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::If, token_str, ch),
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::In, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::In => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::In)?,
TokenizerState::If => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::If)?,
TokenizerState::F =>
{
match ch
{
'a' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Fa, token_str, ch),
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Fo, token_str, ch),
'u' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Fu, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Fu =>
{
match ch
{
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Fun, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Fun =>
{
match ch
{
'c' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Func, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Func =>
{
match ch
{
't' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Funct, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Funct =>
{
match ch
{
'i' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Functi, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Functi =>
{
match ch
{
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Functio, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Functio =>
{
match ch
{
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Function, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Function => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Function)?,
TokenizerState::Fa =>
{
match ch
{
'l' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Fal, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Fal =>
{
match ch
{
's' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Fals, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Fals =>
{
match ch
{
'e' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::False, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::False => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::False)?,
TokenizerState::Fo =>
{
match ch
{
'r' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::For, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::For => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::For)?,
TokenizerState::L =>
{
match ch
{
'o' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Lo, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Lo =>
{
match ch
{
'c' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Loc, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Loc =>
{
match ch
{
'a' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Loca, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Loca =>
{
match ch
{
'l' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Local, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Local => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Local)?,
TokenizerState::U =>
{
match ch
{
'n' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Un, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Un =>
{
match ch
{
't' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Unt, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Unt =>
{
match ch
{
'i' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Unti, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Unti =>
{
match ch
{
'l' => tokenize_terminal_no_token(last_index, *index, state, TokenizerState::Until, token_str, ch),
_ => tokenize_alphanumeric_nonstart(last_index, index, tokens, token, token_str, state, ch)?,
}
}
TokenizerState::Until => tokenize_alphanumeric_nonstart_custom(last_index, index, tokens, token, token_str, state, ch, Token::Until)?,
_ => todo!("State: {:?}", state),
}
return Ok(());
}
pub fn tokenize(file_content: &String) -> Result<Vec<Token>, &'static str>
{
let mut tokens: Vec<Token> = Vec::new();
let mut state = TokenizerState::Start;
let char_vec: Vec<char> = file_content.chars().collect();
let mut last_index: i32 = -1;
let mut index = 0;
let mut token: Option<Token> = None;
let mut token_str: String = String::new();
let mut long_bracket_level = 0;
while index < char_vec.len()
{
let ch = char_vec[index];
tokenize_char(&mut state, ch, &mut last_index, &mut index, &mut token, &mut token_str, &mut tokens, &mut long_bracket_level)?;
index += 1;
}
match state
{
TokenizerState::Name => tokenize_backtrack_name(&mut last_index, &mut index, &mut tokens, &mut token, &mut token_str, &mut state)?,
TokenizerState::End => tokenize_backtrack_custom_token(&mut last_index, &mut index, &mut tokens, &mut token, &mut token_str, &mut state, Token::End)?,
TokenizerState::And => tokenize_backtrack_custom_token(&mut last_index, &mut index, &mut tokens, &mut token, &mut token_str, &mut state, Token::And)?,
TokenizerState::Semicolon => tokenize_backtrack_custom_token(&mut last_index, &mut index, &mut tokens, &mut token, &mut token_str, &mut state, Token::Semicolon)?,
TokenizerState::Number =>
{
if let Some(numeral_token) = token
{
if let Token::Numeral(_) = numeral_token
{
tokens.push(numeral_token);
}
else
{
return Err("In number state but current token is not a numeral")
}
}
else
{
return Err("In number state but no current token")
}
}
TokenizerState::Start =>
{
if token.is_some()
{
return Err("Finished tokenizing in the start state but the token was non-empty");
}
}
_ => todo!("state: {:?} {:?}", state, token),
}
return Ok(tokens);
}
+6
View File
@@ -0,0 +1,6 @@
if T == nil then
(Message or print)('\n >>> testC not active: \z
skipping some generational tests <<<\n')
print 'OK'
return
end
+8
View File
@@ -0,0 +1,8 @@
a, b = 12, test(32, 4)
local t=(string.find(originalField.af,'m') and originalField.tableAction) or c.tableAction or originalField.tableAction or tableActionGeneric
b = {["a"] = 23}
for i=0, 10 do b[i] = 2^23 end
print("asdf")
function test(a, b)
return 42 + a / b
end
+5
View File
@@ -0,0 +1,5 @@
0.000001
0.1
0.12345
1234566788.21212
+22
View File
@@ -0,0 +1,22 @@
0xff11 123 , () ~ ~= > >> >= < << <= / // . .. ... : :: = == - --
-- asdf
--[[askf]]zzz
--[==========[asdfasdf]==========] zzzz
[[fsad]]
[=
]
[===[test]]=]==]===] zzzzz
a ab an anb ando and
w wo wh who whi whio whil whilo while whileo
b bo br bro bre breo brea breao break breako
g gz go goz got gotz goto gotoz
r ro re reo ret rep reto repo retu repe retuo repeo retur repea returo repeao return repeat returno repeato
n nz ni no niz noz nil not nilz notz
t to th tr tho the tro tru then true theno trueo
e eo el en elo eno els end elso endo else elso elsei elseif elseio elseifo
o oo or oro
d dz do doz
i io if in ifo ino
f fz fo fa fu foz faz fuz for fal fun forz falz funz fals func falsz funcz false funct falsez functz functi functiz functio functioz function functionz
l lz lo loz loc locz loca locaz local localz
u uo un uno unt unto unti untio until untilo
+13
View File
@@ -0,0 +1,13 @@
"test" "\z
abc" "123" "sdlfkgj<3" "asldkfj" zzz "" "" "" "" "" "fasd!" "afd" "" "as" zzzz
"\xf7\xAff\x43"
"\u{fa4}\u{1234}\u{12000}\u{123}"