Compare commits

...
6 Commits
Author SHA1 Message Date
HEL 8252f452f2 feat(parser): add base parser class
the parser was adapted from another project (see docstring on the Parser class)
2026-05-13 22:40:27 +02:00
HEL cc4b5dabf2 feat(parser): add midas lexer to test script 2026-05-13 22:40:26 +02:00
HEL 1fc842e23f feat(parser): add basic lexer for type definitions 2026-05-13 22:40:26 +02:00
HEL fcbea218a4 feat(parser): add a test script for the annotation lexer 2026-05-13 22:40:26 +02:00
HEL 10ee4991c3 feat(parser): add a basic lexer for annotations 2026-05-13 22:40:25 +02:00
HEL fedc582e16 feat(parser): add base lexer class
the lexer and token structures were adapted from another project (see docstring on the Lexer class)
2026-05-13 22:40:19 +02:00
12 changed files with 653 additions and 2 deletions

No files matched your search

+6 -1
View File
@@ -1 +1,6 @@
.vscode
.vscode
__pycache__
.env
venv
.venv
*.pyc
@@ -21,4 +21,4 @@ type Age<int + (0 <= _ < 150)>
// Predefined custom constraints that can be referenced in other definitions
constraint Positive = _ >= 0
constraint StrictlyPositive = _ > 0
constraint Even = _ % 2 == 0
//constraint Even = _ % 2 == 0
View File
Whitespace-only changes.
+81
View File
@@ -0,0 +1,81 @@
from lexer.base import Lexer
from lexer.token import TokenType
class AnnotationLexer(Lexer):
def scan_token(self) -> None:
char: str = self.advance()
match char:
case "(":
self.add_token(TokenType.LEFT_PAREN)
case ")":
self.add_token(TokenType.RIGHT_PAREN)
case "[":
self.add_token(TokenType.LEFT_BRACKET)
case "]":
self.add_token(TokenType.RIGHT_BRACKET)
case ":":
self.add_token(TokenType.COLON)
case ",":
self.add_token(TokenType.COMMA)
case "_":
self.add_token(TokenType.UNDERSCORE)
case "+":
self.add_token(TokenType.PLUS)
case "#":
self.scan_comment()
case "\n":
self.add_token(TokenType.NEWLINE)
case " " | "\r" | "\t":
# Consume all whitespace characters until EOL or EOF
while (
self.peek().isspace()
and self.peek() != "\n"
and not self.is_at_end()
):
self.advance()
self.add_token(TokenType.WHITESPACE)
case _:
if char.isdigit():
self.scan_number()
elif char.isalpha():
self.scan_identifier()
else:
self.error("Unexpected character")
return None
def scan_number(self):
"""Scan the rest of number and add it as a token
This method handles both simple integers and floats. Scientific notation
and base prefixes (0x, 0b, 0o) are not supported
"""
while self.peek().isdigit():
self.advance()
if self.peek() == "." and self.peek_next().isdigit():
self.advance()
while self.peek().isdigit():
self.advance()
value: float = float(self.source[self.start : self.idx])
self.add_token(TokenType.NUMBER, value)
def scan_identifier(self):
"""Scan the rest of an identifier and add it as a token
An identifier starts with a letter, followed by any number of
alphanumerical characters or underscores
"""
while self.peek().isalnum() or self.peek() == "_":
self.advance()
self.add_token(TokenType.IDENTIFIER)
def scan_comment(self):
"""Scan the rest of a comment and add it as a token
A comment starts with a `#` character and ends at the EOL/EOF
"""
while self.peek() != "\n" and not self.is_at_end():
self.advance()
self.add_token(TokenType.COMMENT)
+166
View File
@@ -0,0 +1,166 @@
from abc import ABC, abstractmethod
from typing import Any, Callable, Optional
from lexer.position import Position
from lexer.token import Token, TokenType
class Lexer(ABC):
"""An abstract lexer which provides methods to easily extend it into a concrete one
This implementation is based on the [_Crafting Interpreters_][1] book by Robert Nystrom,
more specifically on my [previous Python implementation](https://git.kb28.ch/HEL/pebble)
[1]: https://craftinginterpreters.com/
"""
def __init__(self, source: str, file: Optional[str] = None) -> None:
"""Create a new lexer to scan for tokens in the given source
Args:
source (str): the source to scan
file (Optional[str], optional): the path of the given source. Can be a file path or any string identifier. Defaults to None.
"""
self.source: str = source
self.file: Optional[str] = file
self.tokens: list[Token] = []
self.start: int = 0
self.idx: int = 0
self.length: int = len(self.source)
self.line: int = 1
self.column: int = 1
self.start_pos: Position = self.get_position()
def error(self, msg: str):
"""Raise a syntax error
Args:
msg (str): the error message
Raises:
SyntaxError
"""
raise SyntaxError(f"[ERROR] Error at {self.start_pos}: {msg}")
def process(self) -> list[Token]:
"""Scan tokens out of the source text
Returns:
list[Token]: all the tokens that could be scanned
Raises:
SyntaxError: if a syntax error is found
"""
self.scan_tokens()
self.tokens.append(Token(TokenType.EOF, "", None, self.get_position()))
return self.tokens
def is_at_end(self) -> bool:
"""Whether the lexer is at the end of the source
Returns:
bool: True if the current index is at the end of the source
"""
return self.idx >= self.length
def get_position(self) -> Position:
"""Get the current position
Returns:
Position: the current position
"""
return Position(file=self.file, line=self.line, column=self.column)
def peek(self) -> str:
"""Get the current character without advancing, if any
Returns:
str: the current character, or an empty string if at EOF
"""
if self.idx < self.length:
return self.source[self.idx]
return ""
def peek_next(self) -> str:
"""Get the next character without advancing, if any
Returns:
str: the next character, or an empty string if at EOF
"""
if self.idx + 1 < self.length:
return self.source[self.idx + 1]
return ""
def advance(self) -> str:
"""Get the new character and advance
Returns:
str: the current character, before advancing
"""
char: str = self.peek()
self.idx += 1
self.column += 1
if char == "\n":
self.newline()
return char
def newline(self):
"""Update the current position after encountering a newline character"""
self.line += 1
self.column = 1
def match(self, expected: str) -> bool:
"""Consume the next character if it matches the given value
Args:
expected (str): the expected character
Returns:
bool: whether a character was matched and consumed
"""
if self.peek() == expected:
self.advance()
return True
return False
def update_start(self):
"""Update the starting position of the current lexeme
The cursor marking the start of the lexeme currently being scanned is
moved to the current position
"""
self.start_pos = self.get_position()
self.start = self.idx
def add_token(self, token_type: TokenType, value: Optional[Any] = None):
"""Add the current lexeme to the list of scanned tokens
Args:
token_type (TokenType): the type of token to add
value (Optional[Any], optional): the value of the token (useful for numbers or constants). Defaults to None.
"""
lexeme: str = self.source[self.start : self.idx]
self.tokens.append(
Token(position=self.start_pos, type=token_type, lexeme=lexeme, value=value)
)
def scan_tokens(self, condition: Optional[Callable[[], bool]] = None):
"""Scan tokens until EOF is reached or the given condition becomes False
Args:
condition (Optional[Callable[[], bool]], optional): the condition to continue scanning tokens.
If None, defaults to always being True, effectively scanning tokens until EOF is reached. Defaults to None.
"""
if condition is None:
condition = lambda: True # noqa: E731
while condition() and not self.is_at_end():
self.update_start()
self.scan_token()
@abstractmethod
def scan_token(self) -> None:
"""Scan a token
This function should (at least) consume the current character and produce the appropriate token(s), using `add_token`
"""
pass
+9
View File
@@ -0,0 +1,9 @@
from lexer.token import TokenType
KEYWORDS: dict[str, TokenType] = {
"type": TokenType.TYPE,
"op": TokenType.OP,
"constraint": TokenType.CONSTRAINT,
"true": TokenType.TRUE,
"false": TokenType.FALSE,
}
+126
View File
@@ -0,0 +1,126 @@
from lexer.base import Lexer
from lexer.keyword import KEYWORDS
from lexer.token import TokenType
class MidasLexer(Lexer):
def scan_token(self) -> None:
char: str = self.advance()
match char:
case "(":
self.add_token(TokenType.LEFT_PAREN)
case ")":
self.add_token(TokenType.RIGHT_PAREN)
case "[":
self.add_token(TokenType.LEFT_BRACKET)
case "]":
self.add_token(TokenType.RIGHT_BRACKET)
case "{":
self.add_token(TokenType.LEFT_BRACE)
case "}":
self.add_token(TokenType.RIGHT_BRACE)
case "<":
self.add_token(
TokenType.LESS_EQUAL if self.match("=") else TokenType.LESS
)
case ">":
self.add_token(
TokenType.GREATER_EQUAL if self.match("=") else TokenType.GREATER
)
case "=":
self.add_token(
TokenType.EQUAL_EQUAL if self.match("=") else TokenType.EQUAL
)
case ":":
self.add_token(TokenType.COLON)
case ",":
self.add_token(TokenType.COMMA)
case "_":
self.add_token(TokenType.UNDERSCORE)
case "+":
self.add_token(TokenType.PLUS)
case "-":
self.add_token(TokenType.MINUS)
case "*":
self.add_token(TokenType.STAR)
case "/":
if self.match("/"):
self.scan_comment()
elif self.match("*"):
self.scan_comment_multiline()
else:
self.add_token(TokenType.SLASH)
case "\n":
self.add_token(TokenType.NEWLINE)
case " " | "\r" | "\t":
# Consume all whitespace characters until EOL or EOF
while (
self.peek().isspace()
and self.peek() != "\n"
and not self.is_at_end()
):
self.advance()
self.add_token(TokenType.WHITESPACE)
case _:
if char.isdigit():
self.scan_number()
elif char.isalpha():
self.scan_identifier()
else:
self.error("Unexpected character")
return None
def scan_number(self):
"""Scan the rest of number and add it as a token
This method handles both simple integers and floats. Scientific notation
and base prefixes (0x, 0b, 0o) are not supported
"""
while self.peek().isdigit():
self.advance()
if self.peek() == "." and self.peek_next().isdigit():
self.advance()
while self.peek().isdigit():
self.advance()
value: float = float(self.source[self.start : self.idx])
self.add_token(TokenType.NUMBER, value)
def scan_identifier(self):
"""Scan the rest of an identifier and add it as a token
An identifier starts with a letter, followed by any number of
alphanumerical characters or underscores
"""
while self.peek().isalnum() or self.peek() == "_":
self.advance()
lexeme: str = self.source[self.start : self.idx]
token_type: TokenType = KEYWORDS.get(lexeme, TokenType.IDENTIFIER)
self.add_token(token_type)
def scan_comment(self):
"""Scan the rest of a comment and add it as a token
A comment starts with `//` and ends at the EOL/EOF
"""
while self.peek() != "\n" and not self.is_at_end():
self.advance()
self.add_token(TokenType.COMMENT)
def scan_comment_multiline(self):
"""Scan the rest of a multiline comment and add it as a token
A multiline comment starts with `/*` and ends with `*/` or at the EOF
"""
while (
not (self.peek() == "*" and self.peek_next() == "/")
and not self.is_at_end()
):
self.advance()
if not self.is_at_end():
self.advance()
if not self.is_at_end():
self.advance()
self.add_token(TokenType.COMMENT)
+13
View File
@@ -0,0 +1,13 @@
from dataclasses import dataclass
from typing import Optional
@dataclass(frozen=True)
class Position:
"""A simple structure to store the position of a token"""
file: Optional[str]
line: int
column: int
def __repr__(self):
return f"{self.file or ''}L{self.line}:{self.column}"
+58
View File
@@ -0,0 +1,58 @@
from dataclasses import dataclass
from enum import Enum, auto
from typing import Any
from lexer.position import Position
class TokenType(Enum):
# Punctuation
LEFT_PAREN = auto()
RIGHT_PAREN = auto()
LEFT_BRACKET = auto()
RIGHT_BRACKET = auto()
LEFT_BRACE = auto()
RIGHT_BRACE = auto()
COLON = auto()
COMMA = auto()
UNDERSCORE = auto()
# Operators
PLUS = auto()
MINUS = auto()
STAR = auto()
SLASH = auto()
GREATER = auto()
GREATER_EQUAL = auto()
LESS = auto()
LESS_EQUAL = auto()
EQUAL = auto()
EQUAL_EQUAL = auto()
# Literals
IDENTIFIER = auto()
NUMBER = auto()
TRUE = auto()
FALSE = auto()
NONE = auto()
# Keywords
TYPE = auto()
OP = auto()
CONSTRAINT = auto()
# Misc
COMMENT = auto()
WHITESPACE = auto()
EOF = auto()
NEWLINE = auto()
@dataclass(frozen=True)
class Token:
"""A scanned token"""
type: TokenType
lexeme: str
value: Any
position: Position
+163
View File
@@ -0,0 +1,163 @@
from abc import ABC, abstractmethod
from dataclasses import dataclass
from typing import Generic, TypeVar
from lexer.token import Token, TokenType
from parser.errors import ParsingError
@dataclass(frozen=True)
class TokenError:
"""A parsing error linked to a particular token"""
token: Token
message: str
def get_report(self) -> str:
"""Get a detailed error message
Returns:
str: the complete error message
"""
where: str = f"'{self.token.lexeme}'"
if self.token.type == TokenType.EOF:
where = "end"
return f"({self.token.position}) Error at {where}: {self.message}"
T = TypeVar("T")
class Parser(ABC, Generic[T]):
"""An abstract parser which provides methods to easily extend it into a concrete one
This implementation is based on the [_Crafting Interpreters_][1] book by Robert Nystrom,
more specifically on my [previous Python implementation](https://git.kb28.ch/HEL/pebble)
[1]: https://craftinginterpreters.com/
"""
IGNORE: set[TokenType] = {
TokenType.WHITESPACE,
TokenType.COMMENT,
TokenType.NEWLINE,
}
def __init__(self, tokens: list[Token]) -> None:
"""Create a new parser to parse the given tokens
Args:
tokens (list[Token]): the tokens to parse
"""
self.tokens: list[Token] = list(
filter(lambda t: t.type not in self.IGNORE, tokens)
)
self.current: int = 0
self.length: int = len(self.tokens)
self.errors: list[TokenError]
def error(self, token: Token, message: str):
"""Record an error
Args:
token (Token): the token at which the error was detected
message (str): a message explaining the error
Returns:
ParsingError: the parsing error to raise
"""
self.errors.append(TokenError(token=token, message=message))
return ParsingError()
@abstractmethod
def parse(self) -> T:
"""Parse the tokens
Returns:
T: the parsed element(s)
"""
pass
def is_at_end(self) -> bool:
"""Whether the parser is at the end of the token list
Returns:
bool: True if the current index is at the end of the token list
"""
return self.peek().type == TokenType.EOF
def peek(self) -> Token:
"""Get the current token without advancing
Returns:
Token: the current token
"""
return self.tokens[self.current]
def previous(self) -> Token:
"""Get the previous token
This function is unsafe and will raise an IndexError if called when
the parser is at the begin of the token list
Returns:
Token: the previous token
"""
return self.tokens[self.current - 1]
def check(self, token_type: TokenType) -> bool:
"""Check whether the current token is of the given type
This function always returns False if the parser is at the EOF token
Args:
token_type (TokenType): the type of token to check
Returns:
bool: True if the current token is of the given type and not EOF
"""
if self.is_at_end():
return False
return self.peek().type == token_type
def advance(self) -> Token:
"""Consume and return the current token, if not at the EOF
Returns:
Token: the current token, before advancing
"""
if not self.is_at_end():
self.current += 1
return self.previous()
def match(self, *types: TokenType) -> bool:
"""Consume the next token if it matches one of the given types
Returns:
bool: whether a token was matched and consumed
"""
for token_type in types:
if self.check(token_type):
self.advance()
return True
return False
def consume(self, token_type: TokenType, error_msg: str) -> Token:
"""Consume the current token if it matches the given type or raise an error
If the current token doesn't match the given type, an error is raised
with the provided message
Args:
token_type (TokenType): the expected token type
error_msg (str): the error message if the token doesn't match
Raises:
SyntaxError: if the current token doesn't match the given type
Returns:
Token: the current token which matched the given type
"""
if self.check(token_type):
return self.advance()
raise self.error(self.peek(), error_msg)
+2
View File
@@ -0,0 +1,2 @@
class ParsingError(RuntimeError):
pass
+28
View File
@@ -0,0 +1,28 @@
import importlib
from pathlib import Path
from lexer.annotations import AnnotationLexer
from lexer.midas import MidasLexer
from lexer.token import Token
# Frame annotation
mod = importlib.import_module("examples.00_syntax_prototype.01_simple_types")
annotation: str = mod.__annotations__["df"]
lexer: AnnotationLexer = AnnotationLexer(annotation, "01_simple_types.py")
tokens: list[Token] = lexer.process()
print([
f"{t.type.name}('{t.lexeme}')"
for t in tokens
])
# Midas type definitions
path: Path = Path("examples") / "00_syntax_prototype" / "02_custom_types.midas"
definitions: str = path.read_text()
midas_lexer: MidasLexer = MidasLexer(definitions, path.name)
tokens = midas_lexer.process()
print([
f"{t.type.name}('{t.lexeme}')"
for t in tokens
])