Skip to content

scanner

Scanner

Scans a given text and returns tokens based on the provided configuration.

At each position the scanner picks the rule with the longest match; if several rules match the same length, the one configured first wins. Within a single rule, matching follows Python's re semantics (leftmost-first), so a pattern like a|ab matches only a.

Text that no rule matches raises UnmatchedTextError by default. Pass ignore_unmatched=True to skip it instead, or set a custom handler with set_unmatched_handler.

Example

from lectes import Rule, Configuration, Regex, Scanner

config = Configuration(
    [
        Rule(name="FOR", regex=Regex("for")),
        Rule(name="INT", regex=Regex("[1-9]+")),
        Rule(name="ID", regex=Regex("[a-zA-Z][a-zA-Z0-9]*")),
        Rule(name="WHITESPACE", regex=Regex("( )")),
    ]
)

scanner = Scanner(config)
program = "somevar in othervar for 9 let"

for token in scanner.scan(program):
    print(token)
Source code in src/lectes/scanner/scanner.py
class Scanner:
    """
    Scans a given text and returns tokens based on the provided configuration.

    At each position the scanner picks the rule with the longest match; if
    several rules match the same length, the one configured first wins. Within
    a single rule, matching follows Python's `re` semantics (leftmost-first),
    so a pattern like `a|ab` matches only `a`.

    Text that no rule matches raises `UnmatchedTextError` by default. Pass
    `ignore_unmatched=True` to skip it instead, or set a custom handler with
    `set_unmatched_handler`.

    ## Example

    ```python
    from lectes import Rule, Configuration, Regex, Scanner

    config = Configuration(
        [
            Rule(name="FOR", regex=Regex("for")),
            Rule(name="INT", regex=Regex("[1-9]+")),
            Rule(name="ID", regex=Regex("[a-zA-Z][a-zA-Z0-9]*")),
            Rule(name="WHITESPACE", regex=Regex("( )")),
        ]
    )

    scanner = Scanner(config)
    program = "somevar in othervar for 9 let"

    for token in scanner.scan(program):
        print(token)
    ```
    """

    def __init__(
        self,
        configuration: Configuration,
        debug: bool = False,
        ignore_unmatched: bool = False,
    ) -> None:
        self.configuration = configuration
        self._ignore_unmatched = ignore_unmatched
        self._unmatched_handler: Callable[[UnmatchedText], None] = (
            self._handle_unmatched
        )
        self._matched_handlers = {
            rule: self._handle_matched for rule in configuration.rules
        }
        self._debug = debug
        self._logger = None

    def scan(self, text: str) -> Generator[Token]:
        """
        Scan the given text and yield tokens as they are recognized.

        Each contiguous run of text not matched by any rule is passed to the
        unmatched handler as a single `UnmatchedText`, once the run ends.

        Raises:
            UnmatchedTextError: if part of the text matches no rule, unless
                `ignore_unmatched` is set or a custom unmatched handler is used.
                Tokens before the unmatched text have already been yielded.
        """
        cursor = _Cursor()
        unmatched_location = None
        logger = self.logger()
        debug = logger.enabled()

        while cursor.offset < len(text):
            rule, literal = self._longest_match(text, cursor.offset)

            if rule is None:
                if unmatched_location is None:
                    unmatched_location = cursor.location()

                cursor.advance(text[cursor.offset])
                continue

            if unmatched_location is not None:
                self._flush_unmatched(
                    UnmatchedText(
                        text=text[unmatched_location.offset : cursor.offset],
                        location=unmatched_location,
                    )
                )
                unmatched_location = None

            if debug:
                logger.debug("rule %s matched: %r", rule.name, literal)

            token = Token(rule=rule, literal=literal, location=cursor.location())
            result = self._matched_handlers[rule](token)

            if result is not None:
                yield result

            cursor.advance(literal)

        if unmatched_location is not None:
            self._flush_unmatched(
                UnmatchedText(
                    text=text[unmatched_location.offset :],
                    location=unmatched_location,
                )
            )

    def set_unmatched_handler(self, handler: Callable[[UnmatchedText], None]) -> None:
        """
        Set the given function as the handler that executes when a string is not
        matched to a configured rule.

        The handler receives the `UnmatchedText` and returns None. It replaces the
        default behaviour of raising `UnmatchedTextError`.

        Raises:
            ScannerConfigurationError: if the scanner was created with
                `ignore_unmatched=True`, since the handler would never run.
        """
        if self._ignore_unmatched:
            raise ScannerConfigurationError(
                "cannot set an unmatched handler on a scanner that ignores "
                "unmatched text"
            )

        self._unmatched_handler = handler

    def set_handler(self, rule: Rule, handler: Callable[[Token], Any]) -> None:
        """
        Set the given function as the handler that executes when a string is matched
        against rule.

        The handler receives the matched Token. Whatever it returns, except None,
        is yielded by `scan`; returning None skips the token.
        """
        self._matched_handlers[rule] = handler

    def logger(self) -> Logger:
        """
        Return the scanner's logger instance.
        """
        if self._logger is None:
            self._logger = self._build_logger()

        return self._logger

    def _build_logger(self) -> Logger:
        logger = Logger()

        if self._debug:
            logger.set_level(LogLevel.DEBUG)

        return logger

    def _longest_match(self, text: str, position: int) -> tuple[Rule | None, str]:
        best_rule = None
        best_literal = ""

        for rule in self.configuration.rules:
            literal = rule.regex.match_prefix(text, position)

            # Empty matches never produce tokens; ties go to the earlier rule.
            if literal and len(literal) > len(best_literal):
                best_rule = rule
                best_literal = literal

        return best_rule, best_literal

    def _flush_unmatched(self, unmatched: UnmatchedText) -> None:
        self.logger().debug("unmatched: %r", unmatched.text)

        if self._ignore_unmatched:
            return

        self._unmatched_handler(unmatched)

    @staticmethod
    def _handle_unmatched(unmatched: UnmatchedText) -> NoReturn:
        raise UnmatchedTextError(unmatched)

    @staticmethod
    def _handle_matched(token: Token) -> Token:
        return token

logger()

Return the scanner's logger instance.

Source code in src/lectes/scanner/scanner.py
def logger(self) -> Logger:
    """
    Return the scanner's logger instance.
    """
    if self._logger is None:
        self._logger = self._build_logger()

    return self._logger

scan(text)

Scan the given text and yield tokens as they are recognized.

Each contiguous run of text not matched by any rule is passed to the unmatched handler as a single UnmatchedText, once the run ends.

Raises:

Type Description
UnmatchedTextError

if part of the text matches no rule, unless ignore_unmatched is set or a custom unmatched handler is used. Tokens before the unmatched text have already been yielded.

Source code in src/lectes/scanner/scanner.py
def scan(self, text: str) -> Generator[Token]:
    """
    Scan the given text and yield tokens as they are recognized.

    Each contiguous run of text not matched by any rule is passed to the
    unmatched handler as a single `UnmatchedText`, once the run ends.

    Raises:
        UnmatchedTextError: if part of the text matches no rule, unless
            `ignore_unmatched` is set or a custom unmatched handler is used.
            Tokens before the unmatched text have already been yielded.
    """
    cursor = _Cursor()
    unmatched_location = None
    logger = self.logger()
    debug = logger.enabled()

    while cursor.offset < len(text):
        rule, literal = self._longest_match(text, cursor.offset)

        if rule is None:
            if unmatched_location is None:
                unmatched_location = cursor.location()

            cursor.advance(text[cursor.offset])
            continue

        if unmatched_location is not None:
            self._flush_unmatched(
                UnmatchedText(
                    text=text[unmatched_location.offset : cursor.offset],
                    location=unmatched_location,
                )
            )
            unmatched_location = None

        if debug:
            logger.debug("rule %s matched: %r", rule.name, literal)

        token = Token(rule=rule, literal=literal, location=cursor.location())
        result = self._matched_handlers[rule](token)

        if result is not None:
            yield result

        cursor.advance(literal)

    if unmatched_location is not None:
        self._flush_unmatched(
            UnmatchedText(
                text=text[unmatched_location.offset :],
                location=unmatched_location,
            )
        )

set_handler(rule, handler)

Set the given function as the handler that executes when a string is matched against rule.

The handler receives the matched Token. Whatever it returns, except None, is yielded by scan; returning None skips the token.

Source code in src/lectes/scanner/scanner.py
def set_handler(self, rule: Rule, handler: Callable[[Token], Any]) -> None:
    """
    Set the given function as the handler that executes when a string is matched
    against rule.

    The handler receives the matched Token. Whatever it returns, except None,
    is yielded by `scan`; returning None skips the token.
    """
    self._matched_handlers[rule] = handler

set_unmatched_handler(handler)

Set the given function as the handler that executes when a string is not matched to a configured rule.

The handler receives the UnmatchedText and returns None. It replaces the default behaviour of raising UnmatchedTextError.

Raises:

Type Description
ScannerConfigurationError

if the scanner was created with ignore_unmatched=True, since the handler would never run.

Source code in src/lectes/scanner/scanner.py
def set_unmatched_handler(self, handler: Callable[[UnmatchedText], None]) -> None:
    """
    Set the given function as the handler that executes when a string is not
    matched to a configured rule.

    The handler receives the `UnmatchedText` and returns None. It replaces the
    default behaviour of raising `UnmatchedTextError`.

    Raises:
        ScannerConfigurationError: if the scanner was created with
            `ignore_unmatched=True`, since the handler would never run.
    """
    if self._ignore_unmatched:
        raise ScannerConfigurationError(
            "cannot set an unmatched handler on a scanner that ignores "
            "unmatched text"
        )

    self._unmatched_handler = handler

Location dataclass

Represents where a piece of the scanned text starts.

The offset is 0-based; line and column are 1-based. Columns count characters, so a tab counts as one column, and only \n starts a new line.

Source code in src/lectes/scanner/models.py
@dataclass(frozen=True)
class Location:
    """
    Represents where a piece of the scanned text starts.

    The offset is 0-based; line and column are 1-based. Columns count
    characters, so a tab counts as one column, and only `\\n` starts a new line.
    """

    offset: int
    line: int
    column: int

Token dataclass

Represents a token returned by the scanner.

The scanned token is related to a configuration rule, a string literal and the location in the text where the literal starts.

Source code in src/lectes/scanner/models.py
@dataclass(frozen=True)
class Token:
    """
    Represents a token returned by the scanner.

    The scanned token is related to a configuration rule, a string literal and
    the location in the text where the literal starts.
    """

    rule: Rule
    literal: str
    location: Location

    @property
    def name(self) -> str:
        return self.rule.name

UnmatchedText dataclass

Represents a contiguous run of scanned text that no rule matched, and the location where it starts.

Source code in src/lectes/scanner/models.py
@dataclass(frozen=True)
class UnmatchedText:
    """
    Represents a contiguous run of scanned text that no rule matched, and the
    location where it starts.
    """

    text: str
    location: Location

ScannerConfigurationError

Bases: ScannerError

The scanner has been configured in a conflicting way.

Source code in src/lectes/scanner/errors.py
class ScannerConfigurationError(ScannerError):
    """
    The scanner has been configured in a conflicting way.
    """

ScannerError

Bases: LectesError

Base class for all errors occuring in the scanner.

Source code in src/lectes/scanner/errors.py
5
6
7
8
class ScannerError(LectesError):
    """
    Base class for all errors occuring in the scanner.
    """

UnmatchedTextError

Bases: ScannerError

Part of the scanned text does not match any configured rule.

The unmatched text and its location are available as unmatched.

Source code in src/lectes/scanner/errors.py
class UnmatchedTextError(ScannerError):
    """
    Part of the scanned text does not match any configured rule.

    The unmatched text and its location are available as `unmatched`.
    """

    def __init__(self, unmatched: UnmatchedText) -> None:
        location = unmatched.location
        super().__init__(
            f"unmatched text {unmatched.text!r} "
            f"at line {location.line}, column {location.column}"
        )
        self.unmatched = unmatched