Compare commits
5
Commits
9b59306604
...
4d9358c696
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4d9358c696
|
||
|
|
81a0c5d175
|
||
|
|
c79fc7f909
|
||
|
|
e103469c79
|
||
|
|
d0526110e4
|
No files matched your search
+6
-1
@@ -1 +1,6 @@
|
||||
.vscode
|
||||
.vscode
|
||||
__pycache__
|
||||
.env
|
||||
venv
|
||||
.venv
|
||||
*.pyc
|
||||
@@ -21,4 +21,4 @@ type Age<int + (0 <= _ < 150)>
|
||||
// Predefined custom constraints that can be referenced in other definitions
|
||||
constraint Positive = _ >= 0
|
||||
constraint StrictlyPositive = _ > 0
|
||||
constraint Even = _ % 2 == 0
|
||||
//constraint Even = _ % 2 == 0
|
||||
Whitespace-only changes.
@@ -0,0 +1,81 @@
|
||||
from lexer.base import Lexer
|
||||
from lexer.token import TokenType
|
||||
|
||||
|
||||
class AnnotationLexer(Lexer):
|
||||
def scan_token(self) -> None:
|
||||
char: str = self.advance()
|
||||
match char:
|
||||
case "(":
|
||||
self.add_token(TokenType.LEFT_PAREN)
|
||||
case ")":
|
||||
self.add_token(TokenType.RIGHT_PAREN)
|
||||
case "[":
|
||||
self.add_token(TokenType.LEFT_BRACKET)
|
||||
case "]":
|
||||
self.add_token(TokenType.RIGHT_BRACKET)
|
||||
case ":":
|
||||
self.add_token(TokenType.COLON)
|
||||
case ",":
|
||||
self.add_token(TokenType.COMMA)
|
||||
case "_":
|
||||
self.add_token(TokenType.UNDERSCORE)
|
||||
case "+":
|
||||
self.add_token(TokenType.PLUS)
|
||||
case "#":
|
||||
self.scan_comment()
|
||||
case "\n":
|
||||
self.add_token(TokenType.NEWLINE)
|
||||
case " " | "\r" | "\t":
|
||||
# Consume all whitespace characters until EOL or EOF
|
||||
while (
|
||||
self.peek().isspace()
|
||||
and self.peek() != "\n"
|
||||
and not self.is_at_end()
|
||||
):
|
||||
self.advance()
|
||||
self.add_token(TokenType.WHITESPACE)
|
||||
case _:
|
||||
if char.isdigit():
|
||||
self.scan_number()
|
||||
elif char.isalpha():
|
||||
self.scan_identifier()
|
||||
else:
|
||||
self.error("Unexpected character")
|
||||
return None
|
||||
|
||||
def scan_number(self):
|
||||
"""Scan the rest of number and add it as a token
|
||||
|
||||
This method handles both simple integers and floats. Scientific notation
|
||||
and base prefixes (0x, 0b, 0o) are not supported
|
||||
"""
|
||||
while self.peek().isdigit():
|
||||
self.advance()
|
||||
|
||||
if self.peek() == "." and self.peek_next().isdigit():
|
||||
self.advance()
|
||||
while self.peek().isdigit():
|
||||
self.advance()
|
||||
|
||||
value: float = float(self.source[self.start : self.idx])
|
||||
self.add_token(TokenType.NUMBER, value)
|
||||
|
||||
def scan_identifier(self):
|
||||
"""Scan the rest of an identifier and add it as a token
|
||||
|
||||
An identifier starts with a letter, followed by any number of
|
||||
alphanumerical characters or underscores
|
||||
"""
|
||||
while self.peek().isalnum() or self.peek() == "_":
|
||||
self.advance()
|
||||
self.add_token(TokenType.IDENTIFIER)
|
||||
|
||||
def scan_comment(self):
|
||||
"""Scan the rest of a comment and add it as a token
|
||||
|
||||
A comment starts with a `#` character and ends at the EOL/EOF
|
||||
"""
|
||||
while self.peek() != "\n" and not self.is_at_end():
|
||||
self.advance()
|
||||
self.add_token(TokenType.COMMENT)
|
||||
+166
@@ -0,0 +1,166 @@
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from lexer.position import Position
|
||||
from lexer.token import Token, TokenType
|
||||
|
||||
|
||||
class Lexer(ABC):
|
||||
"""An abstract lexer which provides methods to easily extend it into a concrete one
|
||||
|
||||
This implementation is based on the [_Crafting Interpreters_][1] book by Robert Nystrom,
|
||||
more specifically on my [previous Python implementation](https://git.kb28.ch/HEL/pebble)
|
||||
|
||||
[1]: https://craftinginterpreters.com/
|
||||
"""
|
||||
|
||||
def __init__(self, source: str, file: Optional[str] = None) -> None:
|
||||
"""Create a new lexer to scan for tokens in the given source
|
||||
|
||||
Args:
|
||||
source (str): the source to scan
|
||||
file (Optional[str], optional): the path of the given source. Can be a file path or any string identifier. Defaults to None.
|
||||
"""
|
||||
self.source: str = source
|
||||
self.file: Optional[str] = file
|
||||
self.tokens: list[Token] = []
|
||||
self.start: int = 0
|
||||
self.idx: int = 0
|
||||
self.length: int = len(self.source)
|
||||
self.line: int = 1
|
||||
self.column: int = 1
|
||||
self.start_pos: Position = self.get_position()
|
||||
|
||||
def error(self, msg: str):
|
||||
"""Raise a syntax error
|
||||
|
||||
Args:
|
||||
msg (str): the error message
|
||||
|
||||
Raises:
|
||||
SyntaxError
|
||||
"""
|
||||
raise SyntaxError(f"[ERROR] Error at {self.start_pos}: {msg}")
|
||||
|
||||
def process(self) -> list[Token]:
|
||||
"""Scan tokens out of the source text
|
||||
|
||||
Returns:
|
||||
list[Token]: all the tokens that could be scanned
|
||||
|
||||
Raises:
|
||||
SyntaxError: if a syntax error is found
|
||||
"""
|
||||
self.scan_tokens()
|
||||
self.tokens.append(Token(TokenType.EOF, "", None, self.get_position()))
|
||||
return self.tokens
|
||||
|
||||
def is_at_end(self) -> bool:
|
||||
"""Whether the lexer is at the end of the source
|
||||
|
||||
Returns:
|
||||
bool: True if the current index is at the end of the source
|
||||
"""
|
||||
return self.idx >= self.length
|
||||
|
||||
def get_position(self) -> Position:
|
||||
"""Get the current position
|
||||
|
||||
Returns:
|
||||
Position: the current position
|
||||
"""
|
||||
return Position(file=self.file, line=self.line, column=self.column)
|
||||
|
||||
def peek(self) -> str:
|
||||
"""Get the current character without advancing, if any
|
||||
|
||||
Returns:
|
||||
str: the current character, or an empty string if at EOF
|
||||
"""
|
||||
if self.idx < self.length:
|
||||
return self.source[self.idx]
|
||||
return ""
|
||||
|
||||
def peek_next(self) -> str:
|
||||
"""Get the next character without advancing, if any
|
||||
|
||||
Returns:
|
||||
str: the next character, or an empty string if at EOF
|
||||
"""
|
||||
if self.idx + 1 < self.length:
|
||||
return self.source[self.idx + 1]
|
||||
return ""
|
||||
|
||||
def advance(self) -> str:
|
||||
"""Get the new character and advance
|
||||
|
||||
Returns:
|
||||
str: the current character, before advancing
|
||||
"""
|
||||
char: str = self.peek()
|
||||
self.idx += 1
|
||||
self.column += 1
|
||||
if char == "\n":
|
||||
self.newline()
|
||||
return char
|
||||
|
||||
def newline(self):
|
||||
"""Update the current position after encountering a newline character"""
|
||||
self.line += 1
|
||||
self.column = 1
|
||||
|
||||
def match(self, expected: str) -> bool:
|
||||
"""Consume the next character if it matches the given value
|
||||
|
||||
Args:
|
||||
expected (str): the expected character
|
||||
|
||||
Returns:
|
||||
bool: whether a character was matched and consumed
|
||||
"""
|
||||
if self.peek() == expected:
|
||||
self.advance()
|
||||
return True
|
||||
return False
|
||||
|
||||
def update_start(self):
|
||||
"""Update the starting position of the current lexeme
|
||||
|
||||
The cursor marking the start of the lexeme currently being scanned is
|
||||
moved to the current position
|
||||
"""
|
||||
self.start_pos = self.get_position()
|
||||
self.start = self.idx
|
||||
|
||||
def add_token(self, token_type: TokenType, value: Optional[Any] = None):
|
||||
"""Add the current lexeme to the list of scanned tokens
|
||||
|
||||
Args:
|
||||
token_type (TokenType): the type of token to add
|
||||
value (Optional[Any], optional): the value of the token (useful for numbers or constants). Defaults to None.
|
||||
"""
|
||||
lexeme: str = self.source[self.start : self.idx]
|
||||
self.tokens.append(
|
||||
Token(position=self.start_pos, type=token_type, lexeme=lexeme, value=value)
|
||||
)
|
||||
|
||||
def scan_tokens(self, condition: Optional[Callable[[], bool]] = None):
|
||||
"""Scan tokens until EOF is reached or the given condition becomes False
|
||||
|
||||
Args:
|
||||
condition (Optional[Callable[[], bool]], optional): the condition to continue scanning tokens.
|
||||
If None, defaults to always being True, effectively scanning tokens until EOF is reached. Defaults to None.
|
||||
"""
|
||||
if condition is None:
|
||||
condition = lambda: True # noqa: E731
|
||||
while condition() and not self.is_at_end():
|
||||
self.update_start()
|
||||
self.scan_token()
|
||||
|
||||
@abstractmethod
|
||||
def scan_token(self) -> None:
|
||||
"""Scan a token
|
||||
|
||||
This function should (at least) consume the current character and produce the appropriate token(s), using `add_token`
|
||||
"""
|
||||
pass
|
||||
@@ -0,0 +1,9 @@
|
||||
from lexer.token import TokenType
|
||||
|
||||
KEYWORDS: dict[str, TokenType] = {
|
||||
"type": TokenType.TYPE,
|
||||
"op": TokenType.OP,
|
||||
"constraint": TokenType.CONSTRAINT,
|
||||
"true": TokenType.TRUE,
|
||||
"false": TokenType.FALSE,
|
||||
}
|
||||
+126
@@ -0,0 +1,126 @@
|
||||
from lexer.base import Lexer
|
||||
from lexer.keyword import KEYWORDS
|
||||
from lexer.token import TokenType
|
||||
|
||||
|
||||
class MidasLexer(Lexer):
|
||||
def scan_token(self) -> None:
|
||||
char: str = self.advance()
|
||||
match char:
|
||||
case "(":
|
||||
self.add_token(TokenType.LEFT_PAREN)
|
||||
case ")":
|
||||
self.add_token(TokenType.RIGHT_PAREN)
|
||||
case "[":
|
||||
self.add_token(TokenType.LEFT_BRACKET)
|
||||
case "]":
|
||||
self.add_token(TokenType.RIGHT_BRACKET)
|
||||
case "{":
|
||||
self.add_token(TokenType.LEFT_BRACE)
|
||||
case "}":
|
||||
self.add_token(TokenType.RIGHT_BRACE)
|
||||
case "<":
|
||||
self.add_token(
|
||||
TokenType.LESS_EQUAL if self.match("=") else TokenType.LESS
|
||||
)
|
||||
case ">":
|
||||
self.add_token(
|
||||
TokenType.GREATER_EQUAL if self.match("=") else TokenType.GREATER
|
||||
)
|
||||
case "=":
|
||||
self.add_token(
|
||||
TokenType.EQUAL_EQUAL if self.match("=") else TokenType.EQUAL
|
||||
)
|
||||
case ":":
|
||||
self.add_token(TokenType.COLON)
|
||||
case ",":
|
||||
self.add_token(TokenType.COMMA)
|
||||
case "_":
|
||||
self.add_token(TokenType.UNDERSCORE)
|
||||
case "+":
|
||||
self.add_token(TokenType.PLUS)
|
||||
case "-":
|
||||
self.add_token(TokenType.MINUS)
|
||||
case "*":
|
||||
self.add_token(TokenType.STAR)
|
||||
case "/":
|
||||
if self.match("/"):
|
||||
self.scan_comment()
|
||||
elif self.match("*"):
|
||||
self.scan_comment_multiline()
|
||||
else:
|
||||
self.add_token(TokenType.SLASH)
|
||||
case "\n":
|
||||
self.add_token(TokenType.NEWLINE)
|
||||
case " " | "\r" | "\t":
|
||||
# Consume all whitespace characters until EOL or EOF
|
||||
while (
|
||||
self.peek().isspace()
|
||||
and self.peek() != "\n"
|
||||
and not self.is_at_end()
|
||||
):
|
||||
self.advance()
|
||||
self.add_token(TokenType.WHITESPACE)
|
||||
case _:
|
||||
if char.isdigit():
|
||||
self.scan_number()
|
||||
elif char.isalpha():
|
||||
self.scan_identifier()
|
||||
else:
|
||||
self.error("Unexpected character")
|
||||
return None
|
||||
|
||||
def scan_number(self):
|
||||
"""Scan the rest of number and add it as a token
|
||||
|
||||
This method handles both simple integers and floats. Scientific notation
|
||||
and base prefixes (0x, 0b, 0o) are not supported
|
||||
"""
|
||||
while self.peek().isdigit():
|
||||
self.advance()
|
||||
|
||||
if self.peek() == "." and self.peek_next().isdigit():
|
||||
self.advance()
|
||||
while self.peek().isdigit():
|
||||
self.advance()
|
||||
|
||||
value: float = float(self.source[self.start : self.idx])
|
||||
self.add_token(TokenType.NUMBER, value)
|
||||
|
||||
def scan_identifier(self):
|
||||
"""Scan the rest of an identifier and add it as a token
|
||||
|
||||
An identifier starts with a letter, followed by any number of
|
||||
alphanumerical characters or underscores
|
||||
"""
|
||||
while self.peek().isalnum() or self.peek() == "_":
|
||||
self.advance()
|
||||
|
||||
lexeme: str = self.source[self.start : self.idx]
|
||||
token_type: TokenType = KEYWORDS.get(lexeme, TokenType.IDENTIFIER)
|
||||
self.add_token(token_type)
|
||||
|
||||
def scan_comment(self):
|
||||
"""Scan the rest of a comment and add it as a token
|
||||
|
||||
A comment starts with `//` and ends at the EOL/EOF
|
||||
"""
|
||||
while self.peek() != "\n" and not self.is_at_end():
|
||||
self.advance()
|
||||
self.add_token(TokenType.COMMENT)
|
||||
|
||||
def scan_comment_multiline(self):
|
||||
"""Scan the rest of a multiline comment and add it as a token
|
||||
|
||||
A multiline comment starts with `/*` and ends with `*/` or at the EOF
|
||||
"""
|
||||
while (
|
||||
not (self.peek() == "*" and self.peek_next() == "/")
|
||||
and not self.is_at_end()
|
||||
):
|
||||
self.advance()
|
||||
if not self.is_at_end():
|
||||
self.advance()
|
||||
if not self.is_at_end():
|
||||
self.advance()
|
||||
self.add_token(TokenType.COMMENT)
|
||||
@@ -0,0 +1,13 @@
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Position:
|
||||
"""A simple structure to store the position of a token"""
|
||||
file: Optional[str]
|
||||
line: int
|
||||
column: int
|
||||
|
||||
def __repr__(self):
|
||||
return f"{self.file or ''}L{self.line}:{self.column}"
|
||||
@@ -0,0 +1,58 @@
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum, auto
|
||||
from typing import Any
|
||||
|
||||
from lexer.position import Position
|
||||
|
||||
|
||||
class TokenType(Enum):
|
||||
# Punctuation
|
||||
LEFT_PAREN = auto()
|
||||
RIGHT_PAREN = auto()
|
||||
LEFT_BRACKET = auto()
|
||||
RIGHT_BRACKET = auto()
|
||||
LEFT_BRACE = auto()
|
||||
RIGHT_BRACE = auto()
|
||||
COLON = auto()
|
||||
COMMA = auto()
|
||||
UNDERSCORE = auto()
|
||||
|
||||
# Operators
|
||||
PLUS = auto()
|
||||
MINUS = auto()
|
||||
STAR = auto()
|
||||
SLASH = auto()
|
||||
GREATER = auto()
|
||||
GREATER_EQUAL = auto()
|
||||
LESS = auto()
|
||||
LESS_EQUAL = auto()
|
||||
EQUAL = auto()
|
||||
EQUAL_EQUAL = auto()
|
||||
|
||||
# Literals
|
||||
IDENTIFIER = auto()
|
||||
NUMBER = auto()
|
||||
TRUE = auto()
|
||||
FALSE = auto()
|
||||
NONE = auto()
|
||||
|
||||
# Keywords
|
||||
TYPE = auto()
|
||||
OP = auto()
|
||||
CONSTRAINT = auto()
|
||||
|
||||
# Misc
|
||||
COMMENT = auto()
|
||||
WHITESPACE = auto()
|
||||
EOF = auto()
|
||||
NEWLINE = auto()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Token:
|
||||
"""A scanned token"""
|
||||
|
||||
type: TokenType
|
||||
lexeme: str
|
||||
value: Any
|
||||
position: Position
|
||||
@@ -0,0 +1,28 @@
|
||||
import importlib
|
||||
from pathlib import Path
|
||||
|
||||
from lexer.annotations import AnnotationLexer
|
||||
from lexer.midas import MidasLexer
|
||||
from lexer.token import Token
|
||||
|
||||
|
||||
# Frame annotation
|
||||
mod = importlib.import_module("examples.00_syntax_prototype.01_simple_types")
|
||||
|
||||
annotation: str = mod.__annotations__["df"]
|
||||
lexer: AnnotationLexer = AnnotationLexer(annotation, "01_simple_types.py")
|
||||
tokens: list[Token] = lexer.process()
|
||||
print([
|
||||
f"{t.type.name}('{t.lexeme}')"
|
||||
for t in tokens
|
||||
])
|
||||
|
||||
# Midas type definitions
|
||||
path: Path = Path("examples") / "00_syntax_prototype" / "02_custom_types.midas"
|
||||
definitions: str = path.read_text()
|
||||
midas_lexer: MidasLexer = MidasLexer(definitions, path.name)
|
||||
tokens = midas_lexer.process()
|
||||
print([
|
||||
f"{t.type.name}('{t.lexeme}')"
|
||||
for t in tokens
|
||||
])
|
||||
Reference in new issue
Block a user