Source code for bluebase.lp.lexer

import re

from bluebase.core import assignment

from .error import LpInvalidTokenError
from .type import LpToken


[docs] class LpLexer: r""" A lexer class that converts an input query into tokens. Attributes: pattern: a pattern combining regular expression rules .. rubric:: Rules .. list-table:: :header-rows: 1 * - Group - Pattern * - ``FLOAT`` - ``\d+\.\d+`` * - ``INT`` - ``\d+`` * - ``BOOL`` - ``TRUE|FALSE`` * - ``STRING`` - ``'[^']*'`` * - ``WORD`` - ``[a-zA-Z_*][a-zA-Z0-9_]*`` * - ``COMP`` - ``=|<>|<=|>=|<|>`` * - ``COMMA`` - ``,`` * - ``LP`` - ``\(`` * - ``RP`` - ``\)`` * - ``DOT`` - ``\.`` * - ``SC`` - ``;`` * - ``WS`` - ``[ \t\n]+`` """ RULES = ( ('FLOAT', r"\d+\.\d+"), ('INT', r"\d+"), ('BOOL', r"TRUE|FALSE"), ('STRING', r"'[^']*'"), ('WORD', r"[a-zA-Z_*][a-zA-Z0-9_]*"), ('COMP', r"=|<>|<=|>=|<|>"), ('COMMA', r","), ('LP', r"\("), ('RP', r"\)"), ('DOT', r"\."), ('SC', r";"), ('WS', r"[ \t\n]+"), ) pattern: re.Pattern[str] def __repr__(self) -> str: # pragma: no cover return "LpLexer()" def __init__(self) -> None: self.pattern = re.compile('|'.join([ f"(?P<{name}>{pattern})" for name, pattern in LpLexer.RULES ]))
[docs] @assignment def tokenize(self, query: str) -> list[LpToken]: """ Convert a query into tokens. Raises: LpInvalidTokenError: if a token is invalid Hint: Find matches sequentially using `pattern.finditer`. If the current position does not match the start position of a match, the token is invalid. Ignore ``WS`` tokens instead of adding them. Update the current position to the end position of each match. If the final position does not match the length of the query, the query is also invalid. """ raise NotImplementedError