import re
from bluebase.core import assignment
from .error import LpInvalidTokenError
from .type import LpToken
[docs]
class LpLexer:
r"""
A lexer class that converts an input query into tokens.
Attributes:
pattern: a pattern combining regular expression rules
.. rubric:: Rules
.. list-table::
:header-rows: 1
* - Group
- Pattern
* - ``FLOAT``
- ``\d+\.\d+``
* - ``INT``
- ``\d+``
* - ``BOOL``
- ``TRUE|FALSE``
* - ``STRING``
- ``'[^']*'``
* - ``WORD``
- ``[a-zA-Z_*][a-zA-Z0-9_]*``
* - ``COMP``
- ``=|<>|<=|>=|<|>``
* - ``COMMA``
- ``,``
* - ``LP``
- ``\(``
* - ``RP``
- ``\)``
* - ``DOT``
- ``\.``
* - ``SC``
- ``;``
* - ``WS``
- ``[ \t\n]+``
"""
RULES = (
('FLOAT', r"\d+\.\d+"),
('INT', r"\d+"),
('BOOL', r"TRUE|FALSE"),
('STRING', r"'[^']*'"),
('WORD', r"[a-zA-Z_*][a-zA-Z0-9_]*"),
('COMP', r"=|<>|<=|>=|<|>"),
('COMMA', r","),
('LP', r"\("),
('RP', r"\)"),
('DOT', r"\."),
('SC', r";"),
('WS', r"[ \t\n]+"),
)
pattern: re.Pattern[str]
def __repr__(self) -> str: # pragma: no cover
return "LpLexer()"
def __init__(self) -> None:
self.pattern = re.compile('|'.join([
f"(?P<{name}>{pattern})" for name, pattern in LpLexer.RULES
]))
[docs]
@assignment
def tokenize(self, query: str) -> list[LpToken]:
"""
Convert a query into tokens.
Raises:
LpInvalidTokenError: if a token is invalid
Hint:
Find matches sequentially using `pattern.finditer`.
If the current position does not match the start position of a match, the token is invalid.
Ignore ``WS`` tokens instead of adding them.
Update the current position to the end position of each match.
If the final position does not match the length of the query, the query is also invalid.
"""
raise NotImplementedError