Compare commits

...
5 Commits
Author SHA1 Message Date
HEL 4d9358c696 feat(parser): add midas lexer to test script 2026-05-13 22:07:23 +02:00
HEL 81a0c5d175 feat(parser): add basic lexer for type definitions 2026-05-13 22:06:57 +02:00
HEL c79fc7f909 feat(parser): add a test script for the annotation lexer 2026-05-13 21:31:28 +02:00
HEL e103469c79 feat(parser): add a basic lexer for annotations 2026-05-13 19:26:09 +02:00
HEL d0526110e4 feat(parser): add base lexer class
the lexer and token structures where adapted from another project (see docstring on the Lexer class)
2026-05-13 19:17:55 +02:00
10 changed files with 488 additions and 2 deletions

No files matched your search

+6 -1
View File
@@ -1 +1,6 @@
.vscode
.vscode
__pycache__
.env
venv
.venv
*.pyc
@@ -21,4 +21,4 @@ type Age<int + (0 <= _ < 150)>
// Predefined custom constraints that can be referenced in other definitions
constraint Positive = _ >= 0
constraint StrictlyPositive = _ > 0
constraint Even = _ % 2 == 0
//constraint Even = _ % 2 == 0
View File
Whitespace-only changes.
+81
View File
@@ -0,0 +1,81 @@
from lexer.base import Lexer
from lexer.token import TokenType
class AnnotationLexer(Lexer):
def scan_token(self) -> None:
char: str = self.advance()
match char:
case "(":
self.add_token(TokenType.LEFT_PAREN)
case ")":
self.add_token(TokenType.RIGHT_PAREN)
case "[":
self.add_token(TokenType.LEFT_BRACKET)
case "]":
self.add_token(TokenType.RIGHT_BRACKET)
case ":":
self.add_token(TokenType.COLON)
case ",":
self.add_token(TokenType.COMMA)
case "_":
self.add_token(TokenType.UNDERSCORE)
case "+":
self.add_token(TokenType.PLUS)
case "#":
self.scan_comment()
case "\n":
self.add_token(TokenType.NEWLINE)
case " " | "\r" | "\t":
# Consume all whitespace characters until EOL or EOF
while (
self.peek().isspace()
and self.peek() != "\n"
and not self.is_at_end()
):
self.advance()
self.add_token(TokenType.WHITESPACE)
case _:
if char.isdigit():
self.scan_number()
elif char.isalpha():
self.scan_identifier()
else:
self.error("Unexpected character")
return None
def scan_number(self):
"""Scan the rest of number and add it as a token
This method handles both simple integers and floats. Scientific notation
and base prefixes (0x, 0b, 0o) are not supported
"""
while self.peek().isdigit():
self.advance()
if self.peek() == "." and self.peek_next().isdigit():
self.advance()
while self.peek().isdigit():
self.advance()
value: float = float(self.source[self.start : self.idx])
self.add_token(TokenType.NUMBER, value)
def scan_identifier(self):
"""Scan the rest of an identifier and add it as a token
An identifier starts with a letter, followed by any number of
alphanumerical characters or underscores
"""
while self.peek().isalnum() or self.peek() == "_":
self.advance()
self.add_token(TokenType.IDENTIFIER)
def scan_comment(self):
"""Scan the rest of a comment and add it as a token
A comment starts with a `#` character and ends at the EOL/EOF
"""
while self.peek() != "\n" and not self.is_at_end():
self.advance()
self.add_token(TokenType.COMMENT)
+166
View File
@@ -0,0 +1,166 @@
from abc import ABC, abstractmethod
from typing import Any, Callable, Optional
from lexer.position import Position
from lexer.token import Token, TokenType
class Lexer(ABC):
"""An abstract lexer which provides methods to easily extend it into a concrete one
This implementation is based on the [_Crafting Interpreters_][1] book by Robert Nystrom,
more specifically on my [previous Python implementation](https://git.kb28.ch/HEL/pebble)
[1]: https://craftinginterpreters.com/
"""
def __init__(self, source: str, file: Optional[str] = None) -> None:
"""Create a new lexer to scan for tokens in the given source
Args:
source (str): the source to scan
file (Optional[str], optional): the path of the given source. Can be a file path or any string identifier. Defaults to None.
"""
self.source: str = source
self.file: Optional[str] = file
self.tokens: list[Token] = []
self.start: int = 0
self.idx: int = 0
self.length: int = len(self.source)
self.line: int = 1
self.column: int = 1
self.start_pos: Position = self.get_position()
def error(self, msg: str):
"""Raise a syntax error
Args:
msg (str): the error message
Raises:
SyntaxError
"""
raise SyntaxError(f"[ERROR] Error at {self.start_pos}: {msg}")
def process(self) -> list[Token]:
"""Scan tokens out of the source text
Returns:
list[Token]: all the tokens that could be scanned
Raises:
SyntaxError: if a syntax error is found
"""
self.scan_tokens()
self.tokens.append(Token(TokenType.EOF, "", None, self.get_position()))
return self.tokens
def is_at_end(self) -> bool:
"""Whether the lexer is at the end of the source
Returns:
bool: True if the current index is at the end of the source
"""
return self.idx >= self.length
def get_position(self) -> Position:
"""Get the current position
Returns:
Position: the current position
"""
return Position(file=self.file, line=self.line, column=self.column)
def peek(self) -> str:
"""Get the current character without advancing, if any
Returns:
str: the current character, or an empty string if at EOF
"""
if self.idx < self.length:
return self.source[self.idx]
return ""
def peek_next(self) -> str:
"""Get the next character without advancing, if any
Returns:
str: the next character, or an empty string if at EOF
"""
if self.idx + 1 < self.length:
return self.source[self.idx + 1]
return ""
def advance(self) -> str:
"""Get the new character and advance
Returns:
str: the current character, before advancing
"""
char: str = self.peek()
self.idx += 1
self.column += 1
if char == "\n":
self.newline()
return char
def newline(self):
"""Update the current position after encountering a newline character"""
self.line += 1
self.column = 1
def match(self, expected: str) -> bool:
"""Consume the next character if it matches the given value
Args:
expected (str): the expected character
Returns:
bool: whether a character was matched and consumed
"""
if self.peek() == expected:
self.advance()
return True
return False
def update_start(self):
"""Update the starting position of the current lexeme
The cursor marking the start of the lexeme currently being scanned is
moved to the current position
"""
self.start_pos = self.get_position()
self.start = self.idx
def add_token(self, token_type: TokenType, value: Optional[Any] = None):
"""Add the current lexeme to the list of scanned tokens
Args:
token_type (TokenType): the type of token to add
value (Optional[Any], optional): the value of the token (useful for numbers or constants). Defaults to None.
"""
lexeme: str = self.source[self.start : self.idx]
self.tokens.append(
Token(position=self.start_pos, type=token_type, lexeme=lexeme, value=value)
)
def scan_tokens(self, condition: Optional[Callable[[], bool]] = None):
"""Scan tokens until EOF is reached or the given condition becomes False
Args:
condition (Optional[Callable[[], bool]], optional): the condition to continue scanning tokens.
If None, defaults to always being True, effectively scanning tokens until EOF is reached. Defaults to None.
"""
if condition is None:
condition = lambda: True # noqa: E731
while condition() and not self.is_at_end():
self.update_start()
self.scan_token()
@abstractmethod
def scan_token(self) -> None:
"""Scan a token
This function should (at least) consume the current character and produce the appropriate token(s), using `add_token`
"""
pass
+9
View File
@@ -0,0 +1,9 @@
from lexer.token import TokenType
KEYWORDS: dict[str, TokenType] = {
"type": TokenType.TYPE,
"op": TokenType.OP,
"constraint": TokenType.CONSTRAINT,
"true": TokenType.TRUE,
"false": TokenType.FALSE,
}
+126
View File
@@ -0,0 +1,126 @@
from lexer.base import Lexer
from lexer.keyword import KEYWORDS
from lexer.token import TokenType
class MidasLexer(Lexer):
def scan_token(self) -> None:
char: str = self.advance()
match char:
case "(":
self.add_token(TokenType.LEFT_PAREN)
case ")":
self.add_token(TokenType.RIGHT_PAREN)
case "[":
self.add_token(TokenType.LEFT_BRACKET)
case "]":
self.add_token(TokenType.RIGHT_BRACKET)
case "{":
self.add_token(TokenType.LEFT_BRACE)
case "}":
self.add_token(TokenType.RIGHT_BRACE)
case "<":
self.add_token(
TokenType.LESS_EQUAL if self.match("=") else TokenType.LESS
)
case ">":
self.add_token(
TokenType.GREATER_EQUAL if self.match("=") else TokenType.GREATER
)
case "=":
self.add_token(
TokenType.EQUAL_EQUAL if self.match("=") else TokenType.EQUAL
)
case ":":
self.add_token(TokenType.COLON)
case ",":
self.add_token(TokenType.COMMA)
case "_":
self.add_token(TokenType.UNDERSCORE)
case "+":
self.add_token(TokenType.PLUS)
case "-":
self.add_token(TokenType.MINUS)
case "*":
self.add_token(TokenType.STAR)
case "/":
if self.match("/"):
self.scan_comment()
elif self.match("*"):
self.scan_comment_multiline()
else:
self.add_token(TokenType.SLASH)
case "\n":
self.add_token(TokenType.NEWLINE)
case " " | "\r" | "\t":
# Consume all whitespace characters until EOL or EOF
while (
self.peek().isspace()
and self.peek() != "\n"
and not self.is_at_end()
):
self.advance()
self.add_token(TokenType.WHITESPACE)
case _:
if char.isdigit():
self.scan_number()
elif char.isalpha():
self.scan_identifier()
else:
self.error("Unexpected character")
return None
def scan_number(self):
"""Scan the rest of number and add it as a token
This method handles both simple integers and floats. Scientific notation
and base prefixes (0x, 0b, 0o) are not supported
"""
while self.peek().isdigit():
self.advance()
if self.peek() == "." and self.peek_next().isdigit():
self.advance()
while self.peek().isdigit():
self.advance()
value: float = float(self.source[self.start : self.idx])
self.add_token(TokenType.NUMBER, value)
def scan_identifier(self):
"""Scan the rest of an identifier and add it as a token
An identifier starts with a letter, followed by any number of
alphanumerical characters or underscores
"""
while self.peek().isalnum() or self.peek() == "_":
self.advance()
lexeme: str = self.source[self.start : self.idx]
token_type: TokenType = KEYWORDS.get(lexeme, TokenType.IDENTIFIER)
self.add_token(token_type)
def scan_comment(self):
"""Scan the rest of a comment and add it as a token
A comment starts with `//` and ends at the EOL/EOF
"""
while self.peek() != "\n" and not self.is_at_end():
self.advance()
self.add_token(TokenType.COMMENT)
def scan_comment_multiline(self):
"""Scan the rest of a multiline comment and add it as a token
A multiline comment starts with `/*` and ends with `*/` or at the EOF
"""
while (
not (self.peek() == "*" and self.peek_next() == "/")
and not self.is_at_end()
):
self.advance()
if not self.is_at_end():
self.advance()
if not self.is_at_end():
self.advance()
self.add_token(TokenType.COMMENT)
+13
View File
@@ -0,0 +1,13 @@
from dataclasses import dataclass
from typing import Optional
@dataclass(frozen=True)
class Position:
"""A simple structure to store the position of a token"""
file: Optional[str]
line: int
column: int
def __repr__(self):
return f"{self.file or ''}L{self.line}:{self.column}"
+58
View File
@@ -0,0 +1,58 @@
from dataclasses import dataclass
from enum import Enum, auto
from typing import Any
from lexer.position import Position
class TokenType(Enum):
# Punctuation
LEFT_PAREN = auto()
RIGHT_PAREN = auto()
LEFT_BRACKET = auto()
RIGHT_BRACKET = auto()
LEFT_BRACE = auto()
RIGHT_BRACE = auto()
COLON = auto()
COMMA = auto()
UNDERSCORE = auto()
# Operators
PLUS = auto()
MINUS = auto()
STAR = auto()
SLASH = auto()
GREATER = auto()
GREATER_EQUAL = auto()
LESS = auto()
LESS_EQUAL = auto()
EQUAL = auto()
EQUAL_EQUAL = auto()
# Literals
IDENTIFIER = auto()
NUMBER = auto()
TRUE = auto()
FALSE = auto()
NONE = auto()
# Keywords
TYPE = auto()
OP = auto()
CONSTRAINT = auto()
# Misc
COMMENT = auto()
WHITESPACE = auto()
EOF = auto()
NEWLINE = auto()
@dataclass(frozen=True)
class Token:
"""A scanned token"""
type: TokenType
lexeme: str
value: Any
position: Position
+28
View File
@@ -0,0 +1,28 @@
import importlib
from pathlib import Path
from lexer.annotations import AnnotationLexer
from lexer.midas import MidasLexer
from lexer.token import Token
# Frame annotation
mod = importlib.import_module("examples.00_syntax_prototype.01_simple_types")
annotation: str = mod.__annotations__["df"]
lexer: AnnotationLexer = AnnotationLexer(annotation, "01_simple_types.py")
tokens: list[Token] = lexer.process()
print([
f"{t.type.name}('{t.lexeme}')"
for t in tokens
])
# Midas type definitions
path: Path = Path("examples") / "00_syntax_prototype" / "02_custom_types.midas"
definitions: str = path.read_text()
midas_lexer: MidasLexer = MidasLexer(definitions, path.name)
tokens = midas_lexer.process()
print([
f"{t.type.name}('{t.lexeme}')"
for t in tokens
])