Coverage for trlc/lexer_md.py: 86%
712 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 11:03 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 11:03 +0000
1#!/usr/bin/env python3
2#
3# TRLC - Treat Requirements Like Code
4# Copyright (C) 2026 Bayerische Motoren Werke Aktiengesellschaft (BMW AG)
5#
6# This file is part of the TRLC Python Reference Implementation.
7#
8# TRLC is free software: you can redistribute it and/or modify it
9# under the terms of the GNU General Public License as published by
10# the Free Software Foundation, either version 3 of the License, or
11# (at your option) any later version.
12#
13# TRLC is distributed in the hope that it will be useful, but WITHOUT
14# ANY WARRANTY; without even the implied warranty of MERCHANTABILITY
15# or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public
16# License for more details.
17#
18# You should have received a copy of the GNU General Public License
19# along with TRLC. If not, see <https://www.gnu.org/licenses/>.
21"""Lexer for Markdown TRLC (.trlc.md) files.
23Converts a Markdown representation of a TRLC requirements file into a
24token stream compatible with the TRLC parser.
26Markdown format
27---------------
28The file structure maps to TRLC constructs as follows:
30 # PackageName
31 → ``package PackageName``
33 import pkg
34 → ``import pkg``
36 The package heading and the import lines are the markdown spelling of
37 the TRLC preamble, so they follow the ``package_name`` and
38 ``import_clause`` grammar of the language: the name may be a nested
39 package name (``# ns.app``, ``import ns.base``) and an import may be a
40 wildcard (``import ns.*``).
42 ## Section name
43 → ``section "Section name" {``
45 <hr>
46 → Record separator (closes any open record within the section)
48 ### Record heading
49 → Starts a record; the heading text becomes the record identifier
50 (spaces and non-alphanumeric characters are replaced with ``_``).
52 Property table (under the ``###`` heading):
54 | Property | Value |
55 |----------|-------|
56 | type | Foo | ← ``type`` row gives the record type name
57 | field | value | ← subsequent rows are ``field = value`` assignments
59 #### Field heading
60 → Starts a free-text / string field whose name is the (lowercased)
61 heading identifier. All content that follows—including any Markdown
62 tables—is collected verbatim and emitted as a single STRING token.
63 The string field ends when the next heading (``##``/``###``/``####``),
64 ``<hr>``, or end-of-file is reached.
66Value inference
67---------------
68Values in the property table are interpreted as follows (in order):
70 * ``true`` / ``false`` / ``null`` → KEYWORD token
71 * Decimal integer (``42``) → INTEGER token
72 * Decimal number (``3.14``) → DECIMAL token
73 * Hex integer (``0x1A``) → INTEGER token
74 * Binary integer (``0b101``) → INTEGER token
75 * Dot-qualified identifier → chain of IDENTIFIER + DOT tokens
76 * Anything else → STRING token
77"""
79import re
80from fractions import Fraction
82from trlc import ast as trlc_ast
83from trlc.lexer import Token, TRLC_Lexer
84from trlc.errors import Location, Message_Handler
85from trlc.location_md import MD_Location
88class MD_Source_Reference(Location):
89 """A Location subclass that carries the source line and caret position
90 so Message_Handler can render the same visual caret output as TRLC."""
92 def __init__(self, file_name, line_no, col_no, source_line):
93 super().__init__(file_name, line_no, col_no)
94 self._source_line = source_line
96 def context_lines(self):
97 # Mirror Source_Reference.context_lines() from trlc/lexer.py:
98 # return [source_line_stripped, caret_string]
99 col = self.col_no if self.col_no else 1
100 stripped = self._source_line.lstrip()
101 leading = len(self._source_line) - len(stripped)
102 caret_col = max(col - 1 - leading, 0)
103 return [stripped, " " * caret_col + "^"]
106class MD_Lexer(TRLC_Lexer):
107 """Lexer that converts a ``.trlc.md`` file to a TRLC token stream.
109 The resulting token stream is compatible with :class:`trlc.parser.Parser`
110 when the parser is instantiated with a custom *lexer* argument.
112 Usage::
114 mh = Message_Handler()
115 lexer = MD_Lexer(mh, "path/to/file.trlc.md")
116 # pass lexer to Parser(mh, stab, file_name, ..., lexer=lexer)
117 """
119 # Boolean / null keywords (mirrors TRLC_Lexer.KEYWORDS)
120 KEYWORDS = frozenset(["true", "false", "null", "#", "##", "import"])
122 # Markdown-friendly aliases that map to TRLC block tokens.
123 MD_SECTION_START_TOKEN = "C_BRA"
124 MD_SECTION_END_TOKEN = "C_KET"
126 # ------------------------------------------------------------------ #
127 # Character classification (mirrors Lexer_Base / TRLC_Lexer helpers) #
128 # ------------------------------------------------------------------ #
130 @staticmethod
131 def _is_alpha(c):
132 return c.isascii() and c.isalpha()
134 @staticmethod
135 def _is_numeric(c):
136 return c.isascii() and c.isdigit()
138 @staticmethod
139 def _is_alnum(c):
140 return c.isascii() and c.isalnum()
142 @staticmethod
143 def _is_ident_start(c):
144 return c == "_" or (c.isascii() and c.isalpha())
146 @staticmethod
147 def _is_ident_cont(c):
148 return c == "_" or (c.isascii() and c.isalnum())
150 @staticmethod
151 def _is_hex_digit(c):
152 return c in "0123456789abcdefABCDEF"
154 # ------------------------------------------------------------------ #
155 # Value-token scanning helpers #
156 # ------------------------------------------------------------------ #
158 @staticmethod
159 def _scan_integer(text, start=0):
160 """Scan a run of decimal digits (and underscores) from *start*.
162 Returns the index one past the last scanned character, or ``-1``
163 when no digit is present at *start*.
164 """
165 i = start
166 n = len(text)
167 if i >= n or not text[i].isdigit():
168 return -1
169 while i < n and (text[i].isdigit() or text[i] == "_"):
170 i += 1
171 return i
173 @staticmethod
174 def _scan_hex(text, start=2):
175 """Scan hex digits from *start* (caller has consumed the ``0x`` prefix).
177 Returns the end index or ``-1`` if no valid hex digit at *start*.
178 """
179 i = start
180 n = len(text)
181 if i >= n or not MD_Lexer._is_hex_digit(text[i]): 181 ↛ 182line 181 didn't jump to line 182 because the condition on line 181 was never true
182 return -1
183 while i < n and (MD_Lexer._is_hex_digit(text[i]) or text[i] == "_"):
184 i += 1
185 return i
187 @staticmethod
188 def _scan_binary(text, start=2):
189 """Scan binary digits from *start* (caller has consumed the ``0b`` prefix).
191 Returns the end index or ``-1`` if no valid binary digit at *start*.
192 """
193 i = start
194 n = len(text)
195 if i >= n or text[i] not in "01": 195 ↛ 196line 195 didn't jump to line 196 because the condition on line 195 was never true
196 return -1
197 while i < n and (text[i] in "01" or text[i] == "_"):
198 i += 1
199 return i
201 @staticmethod
202 def _scan_ident(text, start=0):
203 """Scan one identifier segment from *text[start]*.
205 Returns the end index or ``-1`` when no identifier can start here.
206 """
207 i = start
208 n = len(text)
209 if i >= n or not MD_Lexer._is_ident_start(text[i]): 209 ↛ 210line 209 didn't jump to line 210 because the condition on line 209 was never true
210 return -1
211 i += 1
212 while i < n and MD_Lexer._is_ident_cont(text[i]):
213 i += 1
214 return i
216 # ------------------------------------------------------------------ #
217 # Heading / structural helpers #
218 # ------------------------------------------------------------------ #
220 @staticmethod
221 def _parse_heading(line):
222 """Return ``(level, content)`` for a Markdown heading, or ``(0, None)``.
224 *level* is the number of leading ``#`` characters; *content* is the
225 stripped heading text. Levels 1–4 are meaningful to this lexer.
226 """
227 i = 0
228 while i < len(line) and line[i] == "#":
229 i += 1
230 if i == 0 or i >= len(line) or not line[i].isspace():
231 return 0, None
232 content = line[i:].strip()
233 if not content: 233 ↛ 234line 233 didn't jump to line 234 because the condition on line 233 was never true
234 return 0, None
235 return i, content
237 @staticmethod
238 def _is_hr(stripped):
239 """Return True if *stripped* is a markdown record separator line.
241 Accepts ``<hr>``, ``<hr/>``, ``<br>``, ``<br/>`` combinations used in
242 exported markdown such as ``<hr><br><hr>``.
243 """
244 lower = stripped.lower().replace(" ", "")
245 if not lower:
246 return False
248 idx = 0
249 saw_hr = False
250 while idx < len(lower):
251 if lower.startswith("<hr>", idx):
252 idx += 4
253 saw_hr = True
254 elif lower.startswith("<hr/>", idx):
255 idx += 5
256 saw_hr = True
257 elif lower.startswith("<br>", idx):
258 idx += 4
259 elif lower.startswith("<br/>", idx):
260 idx += 5
261 else:
262 return False
264 return saw_hr
266 # ------------------------------------------------------------------ #
267 # Construction #
268 # ------------------------------------------------------------------ #
270 def __init__(self, mh, file_name, file_content=None):
271 assert isinstance(mh, Message_Handler)
272 assert isinstance(file_name, str)
274 self.file_name = file_name
276 if file_content is None:
277 with open(file_name, "r", encoding="UTF-8") as fd:
278 content = fd.read()
279 else:
280 assert isinstance(file_content, str)
281 content = file_content
283 # Initialise TRLC_Lexer to satisfy Parser lexer type checks.
284 super().__init__(mh, file_name, "")
286 self._raw_content = content # kept for Phase 2 reprocessing
287 self._stab = None # set by prepare_phase2() after RSL
288 self._md_tokens = []
289 self._tok_index = 0
291 # Phase 1: emit only preamble tokens (# PackageName, import lines).
292 # Phase 2 is triggered by Source_Manager after parse_rsl_files().
293 self._process_preamble(content)
294 self._preamble_end_index = len(self._md_tokens)
295 self.tokens = self._md_tokens
297 # ------------------------------------------------------------------ #
298 # Lexer_Base interface #
299 # ------------------------------------------------------------------ #
301 def file_location(self):
302 return Location(self.file_name, 1, 1)
304 def token(self):
305 if self._tok_index < len(self._md_tokens):
306 tok = self._md_tokens[self._tok_index]
307 self._tok_index += 1
308 # print("MD_Lexer: end of token stream reached", tok)
309 return tok
310 return None
312 def _process_preamble(self, content):
313 """Phase 1: emit only preamble tokens (# PackageName, import lines).
315 Body processing is deferred to prepare_phase2() so that RSL types
316 are available when field values are tokenised.
317 """
318 lines = content.splitlines()
319 preamble = []
320 for line in lines:
321 stripped = line.strip()
322 if not stripped:
323 preamble.append(line)
324 continue
325 if stripped.startswith("# ") or stripped == "#":
326 preamble.append(line)
327 continue
328 if stripped.startswith("import "):
329 preamble.append(line)
330 continue
331 break # first non-preamble line stops Phase 1
332 self._process("\n".join(preamble))
334 def prepare_phase2(self, stab):
335 """Phase 2: reprocess the full content with RSL types available.
337 Called by Source_Manager after parse_rsl_files() so the stab is
338 fully populated. After this returns, the caller must re-prime the
339 parser token cursor (set ct=None and call advance() once).
340 """
341 self._stab = stab
342 self._md_tokens = []
343 self._tok_index = 0
344 self._process(self._raw_content) # full reprocessing with types
345 self._tok_index = self._preamble_end_index # skip preamble
346 self.tokens = self._md_tokens
348 def _resolve_record_type(self, type_name, package_name=None):
349 """Look up the Record_Type AST node in the stab for *type_name*."""
350 if self._stab is None or not type_name: 350 ↛ 351line 350 didn't jump to line 351 because the condition on line 350 was never true
351 return None
352 if "." in type_name:
353 # The package part may itself be dotted (nested packages), so
354 # only the last segment is the type name.
355 pkg_name, local_name = type_name.rsplit(".", 1)
356 elif package_name: 356 ↛ 359line 356 didn't jump to line 359 because the condition on line 356 was always true
357 pkg_name, local_name = package_name, type_name
358 else:
359 return None
360 pkg_simple = trlc_ast.Symbol_Table.simplified_name(pkg_name)
361 pkg = self._stab.table.get(pkg_simple)
362 if not isinstance(pkg, trlc_ast.Package):
363 return None
364 type_simple = trlc_ast.Symbol_Table.simplified_name(local_name)
365 record_type = pkg.symbols.table.get(type_simple)
366 if isinstance(record_type, trlc_ast.Record_Type):
367 return record_type
368 return None
370 @staticmethod
371 def _get_field_type(record_type_ast, field_name):
372 """Return the declared type of *field_name*, or None."""
373 if record_type_ast is None: 373 ↛ 374line 373 didn't jump to line 374 because the condition on line 373 was never true
374 return None
375 simple = trlc_ast.Symbol_Table.simplified_name(field_name)
376 component = record_type_ast.components.table.get(simple)
377 return component.n_typ if component is not None else None
379 @staticmethod
380 def _is_tuple_array_field(record_type_ast, field_name):
381 """Return True when *field_name* is declared as an array of tuples."""
382 typ = MD_Lexer._get_field_type(record_type_ast, field_name)
383 return isinstance(typ, trlc_ast.Array_Type) and isinstance(
384 typ.element_type, trlc_ast.Tuple_Type
385 )
387 @staticmethod
388 def _is_string_field(record_type_ast, field_name):
389 """Return True when *field_name* is declared as a plain String."""
390 typ = MD_Lexer._get_field_type(record_type_ast, field_name)
391 return typ is not None and isinstance(typ, trlc_ast.Builtin_String)
393 def _emit_field_value(
394 self, raw_value, location, record_type_ast=None, field_name=None
395 ):
396 """Type-aware field value emission (used in Phase 2).
398 Dispatch rules when RSL type info is available:
399 - String field → emit STRING directly (never try array/tuple)
400 - Tuple-array field → type-confirmed array tokenisation
401 - Any other field → existing _emit_value() heuristics
403 Falls back to _emit_value() heuristics when type is unknown.
404 """
405 if record_type_ast is not None:
406 if self._is_string_field(record_type_ast, field_name):
407 self._emit(location, "STRING", raw_value.strip())
408 return
409 if self._is_tuple_array_field(record_type_ast, field_name):
410 if self._check_bracket_array(raw_value, location):
411 return
412 if not self._maybe_emit_array( 412 ↛ 415line 412 didn't jump to line 415 because the condition on line 412 was never true
413 raw_value, location, allow_unqualified=True
414 ):
415 self._emit(location, "STRING", raw_value.strip())
416 return
417 self._emit_value(raw_value, location)
419 # ------------------------------------------------------------------ #
420 # Internal helpers #
421 # ------------------------------------------------------------------ #
423 def _loc(self, line_no, col_no=1):
424 return Location(self.file_name, line_no, col_no)
426 def _source_ref(self, line_no, col_no, source_line):
427 """Return a location that renders a caret line, like TRLC Source_Reference."""
428 return MD_Source_Reference(self.file_name, line_no, col_no, source_line)
430 def _emit(self, location, kind, value=None):
431 if kind == "STRING" and not hasattr(location, "text"):
432 location = MD_Location(
433 file_name=location.file_name,
434 line_no=location.line_no,
435 col_no=location.col_no,
436 token_text=value,
437 mh=self.mh,
438 )
439 self._md_tokens.append(Token(location, kind, value))
441 @staticmethod
442 def _heading_to_identifier(text):
443 """Convert arbitrary heading text to a valid TRLC identifier.
445 Replaces any run of non-alphanumeric characters with a single
446 underscore and strips leading/trailing underscores.
447 """
448 result = []
449 in_sep = False
450 for c in text.strip():
451 if c.isascii() and c.isalnum():
452 result.append(c)
453 in_sep = False
454 else:
455 if not in_sep:
456 result.append("_")
457 in_sep = True
458 return "".join(result).strip("_")
460 @staticmethod
461 def _is_valid_identifier(name):
462 """Return True when *name* matches TRLC identifier syntax."""
463 if not name:
464 return False
465 if not MD_Lexer._is_alpha(name[0]):
466 return False
467 return all(MD_Lexer._is_alnum(ch) or ch == "_" for ch in name[1:])
469 @staticmethod
470 def _skip_spaces(line, start):
471 """Return the 0-based offset of the first non-whitespace character
472 in *line* at or after *start*, so callers can locate a token
473 positionally instead of via a substring search that could match an
474 earlier, unrelated occurrence."""
475 rest = line[start:]
476 return start + (len(rest) - len(rest.lstrip()))
478 def _validate_identifier(self, name, loc_line, heading_prefix_len, source_line=""):
479 """Validate name against TRLC identifier rules.
481 Raises lex_error pointing at the first offending character,
482 with the same caret-line visual as TRLC's 'unexpected character X'.
483 """
484 if not name:
485 self.mh.lex_error(
486 self._source_ref(loc_line, heading_prefix_len + 1, source_line),
487 "expected identifier",
488 )
489 if not MD_Lexer._is_alpha(name[0]):
490 self.mh.lex_error(
491 self._source_ref(loc_line, heading_prefix_len + 1, source_line),
492 f"unexpected character '{name[0]}'",
493 )
494 for i, ch in enumerate(name[1:], 1):
495 if not (MD_Lexer._is_alnum(ch) or ch == "_"): 495 ↛ 496line 495 didn't jump to line 496 because the condition on line 495 was never true
496 self.mh.lex_error(
497 self._source_ref(loc_line, heading_prefix_len + 1 + i, source_line),
498 f"unexpected character '{ch}'",
499 )
501 @staticmethod
502 def _is_separator_row(row):
503 """Return True if *row* is a Markdown table separator (``|---|``)."""
504 stripped = row.strip()
505 if not stripped.startswith("|"): 505 ↛ 506line 505 didn't jump to line 506 because the condition on line 505 was never true
506 return False
507 for c in stripped:
508 if c not in "|-: ":
509 return False
510 return True
512 @staticmethod
513 def _parse_table_row(row):
514 """Split a ``| key | value |`` row into (key, value) strings.
516 Returns ``None`` when the row cannot be parsed as a two-column
517 table entry.
518 """
519 stripped = row.strip()
520 if not stripped.startswith("|"):
521 return None
522 # Split on "|", ignore the empty strings at both ends
523 parts = [p.strip() for p in stripped.split("|")]
524 # After split: ["", cell0, cell1, ..., ""]
525 cells = parts[1:-1]
526 if len(cells) >= 2: 526 ↛ 528line 526 didn't jump to line 528 because the condition on line 526 was always true
527 return cells[0], cells[1]
528 return None
530 def _emit_qualified_identifier(self, location, value):
531 """Emit IDENTIFIER[/DOT/IDENTIFIER...] tokens for a type name."""
532 parts = [p for p in value.split(".") if p]
533 for idx, part in enumerate(parts):
534 self._emit(location, "IDENTIFIER", part)
535 if idx < len(parts) - 1:
536 self._emit(location, "DOT")
538 def _emit_package_name(
539 self, name, line_no, name_offset, source_line, allow_wildcard=False
540 ):
541 """Emit the token stream for a (possibly nested) package name.
543 Emits ``IDENTIFIER { DOT IDENTIFIER } [ DOT OPERATOR('*') ]``, i.e.
544 exactly what :meth:`trlc.parser.Parser.parse_dotted_name` and
545 :meth:`trlc.parser.Parser.parse_import_name` expect, so that
546 ``.trlc.md`` preambles share the TRLC grammar for package names.
548 *name_offset* is the 0-based column of *name* within *source_line*.
549 Each emitted token gets its own caret-capable location (rather than
550 one shared line-level location), so a parser error on e.g. the
551 second segment of ``ns.bad name`` points at that segment.
553 :returns: True if a trailing ``.*`` wildcard was emitted
554 """
555 # lobster-trace: LRM.Nested_Package_Names
556 # lobster-trace: LRM.Wildcard_Import
557 segments = name.split(".")
558 wildcard = allow_wildcard and len(segments) > 1 and segments[-1] == "*"
559 if wildcard:
560 segments = segments[:-1]
562 offset = name_offset
563 for idx, segment in enumerate(segments):
564 self._validate_identifier(segment, line_no, offset, source_line)
565 if idx:
566 self._emit(self._source_ref(line_no, offset, source_line), "DOT")
567 self._emit(
568 self._source_ref(line_no, offset + 1, source_line),
569 "IDENTIFIER",
570 segment,
571 )
572 offset += len(segment) + 1
574 if wildcard:
575 self._emit(self._source_ref(line_no, offset, source_line), "DOT")
576 self._emit(
577 self._source_ref(line_no, offset + 1, source_line), "OPERATOR", "*"
578 )
580 return wildcard
582 # Separator symbol token kinds (mirrors TRLC_Lexer.PUNCTUATION).
583 _SEPARATOR_PUNCTUATION = {"@": "AT", ":": "COLON", ";": "SEMICOLON"}
585 @staticmethod
586 def _normalize_array_value(raw_value):
587 """Normalize array value by converting various separators to comma format.
589 Handles:
590 - <br> tags (converted to newlines, then normalized)
591 - Multiple spaces around @, :, ; separators (normalized to single space)
592 - Identifier separators (multiple spaces collapsed to single space)
593 - Newlines (preserved initially for multiline detection, then normalized)
595 Returns normalized string with tuple refs as
596 ``"identifier <sep> integer, ..."``.
597 """
598 # Replace <br> and <BR> tags with newlines for uniform processing
599 normalized = re.sub(r"<[Bb][Rr]\s*/?>", "\n", raw_value)
601 # Split on newlines and commas, trim each part
602 parts = []
603 for line in normalized.split("\n"):
604 for item in line.split(","):
605 trimmed = item.strip()
606 if trimmed: 606 ↛ 604line 606 didn't jump to line 604 because the condition on line 606 was always true
607 parts.append(trimmed)
609 # Normalize whitespace around each part:
610 # - punctuation separators (@, :, ;) get exactly one space each side
611 # - identifier separators: collapse multiple spaces to one
612 normalized_parts = []
613 for part in parts:
614 part = re.sub(r"\s*([@:;])\s*", r" \1 ", part)
615 part = re.sub(r" +", " ", part.strip())
616 normalized_parts.append(part)
618 return ", ".join(normalized_parts)
620 # Tuple-reference pattern: package-qualified identifier, separator, integer.
621 # The reference MUST contain at least one dot (Package.name) so that plain
622 # free-text values are never falsely matched.
623 # Separator is @, :, ; or a plain identifier.
624 _TUPLE_REF_RE = re.compile(
625 r"^"
626 r"(?P<ref>[a-zA-Z_]\w*\.[a-zA-Z_]\w*(?:\.[a-zA-Z_]\w*)*)"
627 r" "
628 r"(?P<sep>[@:;]|[a-zA-Z_]\w*)"
629 r" "
630 r"(?P<ver>\d+)"
631 r"$"
632 )
634 # Like _TUPLE_REF_RE but the dot is optional, so unqualified same-package
635 # references (e.g. ``item @ 1``) are also accepted. Only safe in the
636 # Phase 2 type-aware path where the field type is already known.
637 _TUPLE_REF_FLEXIBLE_RE = re.compile(
638 r"^"
639 r"(?P<ref>[a-zA-Z_]\w*(?:\.[a-zA-Z_]\w*)*)"
640 r" "
641 r"(?P<sep>[@:;]|[a-zA-Z_]\w*)"
642 r" "
643 r"(?P<ver>\d+)"
644 r"$"
645 )
647 # Chunk patterns for the multi-separator tuple scanner.
648 # A numeric chunk: hex, binary, or decimal (integer or float).
649 _TUPLE_NUM_RE = re.compile(
650 r"0[xX][0-9a-fA-F][0-9a-fA-F_]*"
651 r"|0[bB][01][01_]*"
652 r"|\d+(?:\.\d+)?"
653 )
654 # A separator chunk: @, :, ; or a word identifier.
655 _TUPLE_SEP_RE = re.compile(r"[@:;]|[a-zA-Z_]\w*")
657 @staticmethod
658 def _looks_like_array(raw_value, allow_unqualified=False):
659 """Check if value looks like tuple-reference array without emitting.
661 Expected format (separator may be @, :, ;, or any identifier)::
663 identifier[.identifier]* sep integer
664 [, identifier[.identifier]* sep integer]*
666 When *allow_unqualified* is True, unqualified same-package references
667 (e.g. ``item @ 1``) are also accepted. Only pass True when the field
668 type is confirmed as a tuple-reference array.
669 """
670 normalized = MD_Lexer._normalize_array_value(raw_value)
671 parts = [p.strip() for p in normalized.split(",") if p.strip()]
672 pattern = (
673 MD_Lexer._TUPLE_REF_FLEXIBLE_RE
674 if allow_unqualified
675 else MD_Lexer._TUPLE_REF_RE
676 )
677 return bool(parts) and all(pattern.match(part) for part in parts)
679 @staticmethod
680 def _looks_like_simple_tuple(raw_value):
681 """Check if value looks like a separator-form tuple: num [sep num]+.
683 Handles multi-separator patterns like 0x500:12345@6.1 or 1@2:3;4.
684 Returns True only when there is at least one separator (bare numbers
685 are handled by the scalar integer/decimal paths in _emit_value).
686 """
687 value = raw_value.strip()
688 pos = 0
689 n = len(value)
691 def skip_ws():
692 nonlocal pos
693 while pos < n and value[pos] in (" ", "\t"): 693 ↛ 694line 693 didn't jump to line 694 because the condition on line 693 was never true
694 pos += 1
696 skip_ws()
697 m = MD_Lexer._TUPLE_NUM_RE.match(value, pos)
698 if not m: 698 ↛ 700line 698 didn't jump to line 700 because the condition on line 698 was always true
699 return False
700 pos = m.end()
701 has_sep = False
702 while pos < n:
703 skip_ws()
704 if pos >= n:
705 break
706 m = MD_Lexer._TUPLE_SEP_RE.match(value, pos)
707 if not m:
708 return False
709 pos = m.end()
710 has_sep = True
711 skip_ws()
712 m = MD_Lexer._TUPLE_NUM_RE.match(value, pos)
713 if not m:
714 return False
715 pos = m.end()
716 return pos == n and has_sep
718 def _maybe_emit_simple_tuple(self, raw_value, location):
719 """Try to emit tokens for a separator-form tuple: num [sep num]+.
721 Handles multi-separator patterns such as::
723 12345@42 → INTEGER AT INTEGER
724 0x500:12345@6.1 → INTEGER COLON INTEGER AT DECIMAL
725 1@2:3;4 → INTEGER AT INTEGER COLON INTEGER SEMICOLON INTEGER
727 Returns True if at least one separator was found and all tokens were
728 emitted, False to let _emit_value fall through to STRING.
729 """
730 value = raw_value.strip()
731 pos = 0
732 n = len(value)
733 chunks = [] # list of ('num', text) | ('sep', text)
735 def skip_ws():
736 nonlocal pos
737 while pos < n and value[pos] in (" ", "\t"): 737 ↛ 738line 737 didn't jump to line 738 because the condition on line 737 was never true
738 pos += 1
740 skip_ws()
741 m = MD_Lexer._TUPLE_NUM_RE.match(value, pos)
742 if not m:
743 return False
744 chunks.append(("num", m.group()))
745 pos = m.end()
747 while pos < n:
748 skip_ws()
749 if pos >= n: 749 ↛ 750line 749 didn't jump to line 750 because the condition on line 749 was never true
750 break
751 m = MD_Lexer._TUPLE_SEP_RE.match(value, pos)
752 if not m: 752 ↛ 753line 752 didn't jump to line 753 because the condition on line 752 was never true
753 return False
754 chunks.append(("sep", m.group()))
755 pos = m.end()
756 skip_ws()
757 m = MD_Lexer._TUPLE_NUM_RE.match(value, pos)
758 if not m: 758 ↛ 759line 758 didn't jump to line 759 because the condition on line 758 was never true
759 return False
760 chunks.append(("num", m.group()))
761 pos = m.end()
763 if pos != n: 763 ↛ 764line 763 didn't jump to line 764 because the condition on line 763 was never true
764 return False
766 # Must have at least one separator so bare numbers fall through to
767 # the integer/decimal scalar checks in _emit_value.
768 has_sep = any(kind == "sep" for kind, _ in chunks)
769 if not has_sep: 769 ↛ 770line 769 didn't jump to line 770 because the condition on line 769 was never true
770 return False
772 for kind, text in chunks:
773 if kind == "num":
774 if "." in text:
775 self._emit(location, "DECIMAL", Fraction(text.replace("_", "")))
776 else:
777 self._emit(location, "INTEGER", self._parse_number(text))
778 else: # sep
779 sep_kind = MD_Lexer._SEPARATOR_PUNCTUATION.get(text, "IDENTIFIER")
780 sep_val = text if sep_kind == "IDENTIFIER" else None
781 self._emit(location, sep_kind, sep_val)
783 return True
785 def _maybe_emit_parentheses_tuple(self, raw_value, location):
786 """Try to emit tokens for a tuple in parentheses form: (val1, val2, val3).
788 Returns True if a tuple was emitted, False for fallback to STRING.
789 """
790 value = raw_value.strip()
791 if not (value.startswith("(") and value.endswith(")")):
792 return False
794 inner = value[1:-1].strip()
795 if not inner: 795 ↛ 797line 795 didn't jump to line 797 because the condition on line 795 was never true
796 # Empty tuple () - treat as error or skip
797 return False
799 # Split by commas and parse each value
800 self._emit(location, "BRA")
802 parts = [p.strip() for p in inner.split(",") if p.strip()]
803 for idx, part in enumerate(parts):
804 if idx > 0:
805 self._emit(location, "COMMA")
807 # Try to infer the value type for each part
808 self._emit_value(part, location)
810 self._emit(location, "KET")
811 return True
813 @staticmethod
814 def _parse_number(text):
815 """Parse a number string (decimal, hex, or binary).
817 Returns the integer value or None if parsing fails.
818 """
819 try:
820 text_clean = text.replace("_", "")
821 if text_clean.startswith("0x") or text_clean.startswith("0X"):
822 return int(text_clean, 16)
823 elif text_clean.startswith("0b") or text_clean.startswith("0B"): 823 ↛ 824line 823 didn't jump to line 824 because the condition on line 823 was never true
824 return int(text_clean, 2)
825 else:
826 return int(text_clean, 10)
827 except ValueError:
828 return None
830 def _check_bracket_array(self, raw_value, location):
831 """Detect bracket notation for tuple arrays and emit a clear error.
833 Bracket syntax like ``[Pkg.item @ 1, Pkg.item @ 2]`` is not supported
834 because brackets conflict with Markdown URL syntax.
836 Returns True when bracket array syntax was detected (error was emitted),
837 so the caller can skip further processing of this value.
838 """
839 value = raw_value.strip()
840 if not (value.startswith("[") and value.endswith("]")):
841 return False
842 inner = value[1:-1].strip()
843 # Only reject if the content inside the brackets actually looks like
844 # tuple references – avoids false positives on URLs containing '@'
845 if not MD_Lexer._looks_like_array(inner): 845 ↛ 846line 845 didn't jump to line 846 because the condition on line 845 was never true
846 return False
847 self.mh.error(
848 location,
849 "bracket notation for tuple-reference arrays is not supported",
850 explanation=(
851 "Use comma-separated or newline-separated tuple references "
852 "without brackets instead, e.g. "
853 "'Pkg.item_a @ 1, Pkg.item_b @ 2'"
854 ),
855 fatal=False,
856 )
857 return True
859 def _maybe_emit_array(self, raw_value, location, allow_unqualified=False):
860 """Try to emit array tokens for tuple-reference format.
862 When *allow_unqualified* is True, unqualified same-package references
863 (e.g. ``item @ 1``) are also accepted. Only safe when the field type
864 is confirmed as a tuple-reference array (Phase 2 path).
866 Recognises values of the form::
868 Package.item sep integer [, Package.item sep integer]*
870 The reference must be package-qualified (contain at least one dot)
871 so that plain free-text values (e.g. ``Revision : 12``) are never
872 misinterpreted as tuple-reference arrays.
874 Returns True if array was emitted, False for STRING fallback.
875 """
876 if not self._looks_like_array(raw_value, allow_unqualified=allow_unqualified):
877 return False
879 normalized = self._normalize_array_value(raw_value)
880 parts = [p.strip() for p in normalized.split(",")]
881 self._emit(location, "S_BRA")
882 pattern = (
883 MD_Lexer._TUPLE_REF_FLEXIBLE_RE
884 if allow_unqualified
885 else MD_Lexer._TUPLE_REF_RE
886 )
888 for idx, part in enumerate(parts):
889 if idx > 0:
890 self._emit(location, "COMMA")
892 match = pattern.match(part)
893 if match: 893 ↛ 888line 893 didn't jump to line 888 because the condition on line 893 was always true
894 qual_ident = match.group("ref")
895 sep = match.group("sep")
896 integer = int(match.group("ver"))
898 # Emit qualified identifier (Pkg.item → IDENTIFIER DOT IDENTIFIER)
899 ident_parts = qual_ident.split(".")
900 for i, ident_part in enumerate(ident_parts):
901 self._emit(location, "IDENTIFIER", ident_part)
902 if i < len(ident_parts) - 1:
903 self._emit(location, "DOT")
905 # Emit separator: @→AT, :→COLON, ;→SEMICOLON, word→IDENTIFIER
906 sep_kind = MD_Lexer._SEPARATOR_PUNCTUATION.get(sep, "IDENTIFIER")
907 sep_val = sep if sep_kind == "IDENTIFIER" else None
908 self._emit(location, sep_kind, sep_val)
910 # Emit version integer
911 self._emit(location, "INTEGER", integer)
913 self._emit(location, "S_KET")
914 return True
916 def _emit_value(self, raw_value, location):
917 """Emit one or more tokens representing a property value.
919 Inference rules are applied in the order documented in the
920 module docstring.
921 """
922 value = raw_value.strip()
924 # Bracket array notation is not supported – emit a clear error
925 if self._check_bracket_array(raw_value, location):
926 return
928 # Parentheses tuple form: (val1, val2, val3)
929 if self._maybe_emit_parentheses_tuple(raw_value, location):
930 return
932 # Array of tuple references (before other checks)
933 if self._maybe_emit_array(raw_value, location):
934 return
936 # Boolean / null keywords
937 if value in MD_Lexer.KEYWORDS:
938 self._emit(location, "KEYWORD", value)
939 return
941 # Positive integers
942 end = MD_Lexer._scan_integer(value)
943 if end == len(value) and end > 0:
944 self._emit(location, "INTEGER", int(value.replace("_", "")))
945 return
947 # Positive decimals
948 dot_pos = value.find(".")
949 if dot_pos > 0:
950 int_end = MD_Lexer._scan_integer(value, 0)
951 if int_end == dot_pos:
952 dec_end = MD_Lexer._scan_integer(value, dot_pos + 1)
953 if dec_end == len(value) and dec_end > dot_pos + 1: 953 ↛ 958line 953 didn't jump to line 958 because the condition on line 953 was always true
954 self._emit(location, "DECIMAL", Fraction(value.replace("_", "")))
955 return
957 # Hexadecimal 0x…
958 if value.startswith("0x") and len(value) > 2:
959 end = MD_Lexer._scan_hex(value)
960 if end == len(value):
961 self._emit(location, "INTEGER", int(value.replace("_", ""), 16))
962 return
964 # Binary 0b…
965 if value.startswith("0b") and len(value) > 2:
966 end = MD_Lexer._scan_binary(value)
967 if end == len(value): 967 ↛ 972line 967 didn't jump to line 972 because the condition on line 967 was always true
968 self._emit(location, "INTEGER", int(value[2:].replace("_", ""), 2))
969 return
971 # Simple tuple syntax (e.g., 12345@42, 0x500:6.1)
972 if self._maybe_emit_simple_tuple(raw_value, location):
973 return
975 # Dot-qualified identifier or plain identifier
976 if value and MD_Lexer._is_ident_start(value[0]): 976 ↛ 1002line 976 didn't jump to line 1002 because the condition on line 976 was always true
977 parts = []
978 i = 0
979 valid = True
980 while i <= len(value): 980 ↛ 994line 980 didn't jump to line 994 because the condition on line 980 was always true
981 end = MD_Lexer._scan_ident(value, i)
982 if end < 0: 982 ↛ 983line 982 didn't jump to line 983 because the condition on line 982 was never true
983 valid = False
984 break
985 parts.append(value[i:end])
986 i = end
987 if i == len(value):
988 break
989 if value[i] == ".":
990 i += 1
991 else:
992 valid = False
993 break
994 if valid and parts:
995 for idx, part in enumerate(parts):
996 self._emit(location, "IDENTIFIER", part)
997 if idx < len(parts) - 1:
998 self._emit(location, "DOT")
999 return
1001 # Fall-back: treat the value as a plain string
1002 self._emit(location, "STRING", value)
1004 @staticmethod
1005 def _looks_like_scalar_value(value):
1006 """Return True when *value* should be inferred via _emit_value.
1008 This is used for single-line ``####`` field bodies so short scalar
1009 values (e.g. enum members, booleans, numbers) behave like table
1010 properties, while free text remains a STRING.
1011 """
1012 text = value.strip()
1013 if not text: 1013 ↛ 1014line 1013 didn't jump to line 1014 because the condition on line 1013 was never true
1014 return False
1016 if text in ("true", "false", "null"): 1016 ↛ 1017line 1016 didn't jump to line 1017 because the condition on line 1016 was never true
1017 return True
1019 # Parentheses tuple form (simple detection)
1020 if text.startswith("(") and text.endswith(")"): 1020 ↛ 1021line 1020 didn't jump to line 1021 because the condition on line 1020 was never true
1021 return True
1023 # Integer / decimal
1024 end = MD_Lexer._scan_integer(text)
1025 if end == len(text) and end > 0: 1025 ↛ 1026line 1025 didn't jump to line 1026 because the condition on line 1025 was never true
1026 return True
1028 dot_pos = text.find(".")
1029 if dot_pos > 0:
1030 int_end = MD_Lexer._scan_integer(text, 0)
1031 if int_end == dot_pos: 1031 ↛ 1032line 1031 didn't jump to line 1032 because the condition on line 1031 was never true
1032 dec_end = MD_Lexer._scan_integer(text, dot_pos + 1)
1033 if dec_end == len(text) and dec_end > dot_pos + 1:
1034 return True
1036 # Simple tuple syntax (e.g., 12345@42)
1037 if MD_Lexer._looks_like_simple_tuple(text): 1037 ↛ 1038line 1037 didn't jump to line 1038 because the condition on line 1037 was never true
1038 return True
1040 # Dot-qualified identifier (used e.g. for enum values)
1041 if "." in text and MD_Lexer._is_ident_start(text[0]):
1042 i = 0
1043 while i <= len(text): 1043 ↛ 1055line 1043 didn't jump to line 1055 because the condition on line 1043 was always true
1044 end = MD_Lexer._scan_ident(text, i)
1045 if end < 0: 1045 ↛ 1046line 1045 didn't jump to line 1046 because the condition on line 1045 was never true
1046 return False
1047 i = end
1048 if i == len(text): 1048 ↛ 1049line 1048 didn't jump to line 1049 because the condition on line 1048 was never true
1049 return True
1050 if text[i] == ".":
1051 i += 1
1052 else:
1053 return False
1055 return False
1057 # ------------------------------------------------------------------ #
1058 # Main processing #
1059 # ------------------------------------------------------------------ #
1061 def _process(self, content):
1062 """Transform *content* into the ``_md_tokens`` list."""
1064 lines = content.splitlines()
1065 total_lines = len(lines)
1067 # ── Section tracking ─────────────────────────────────────────── #
1068 in_section = False
1069 imported_packages = []
1071 # ── Type tracking for Phase 2 ─────────────────────────────────── #
1072 current_package_name = None # from # PackageName heading
1073 current_record_type_ast = None # resolved after type row found
1075 # ── Record tracking ───────────────────────────────────────────── #
1076 # When a ### heading is seen we buffer the name and then wait for
1077 # the properties table to discover the record type.
1078 in_record = False
1079 pending_name = None # identifier string for the record
1080 pending_name_loc = None # Location of the ### line
1081 pending_props = [] # [(key, value, line_no)] before "type" row
1082 record_type_found = False
1083 props_first_row = False # True while we should skip the header row
1085 # ── String-field tracking ─────────────────────────────────────── #
1086 in_string_field = False
1087 str_field_name = None
1088 str_field_loc = None
1089 str_field_lines = []
1090 str_field_first_content_line = None
1091 str_field_first_content_col = 1
1092 # #### fields seen before the "type" row are buffered here, then
1093 # emitted inside the record after C_BRA (mirrors pending_props).
1094 # Tuple: (name, loc, text, emit_as_scalar, is_array, is_bracket)
1095 pending_string_fields = []
1097 # ── Helpers (closures) ────────────────────────────────────────── #
1099 def flush_string_field():
1100 nonlocal in_string_field, str_field_name, str_field_lines
1101 nonlocal str_field_first_content_line, str_field_first_content_col
1102 if not in_string_field:
1103 return
1104 # Strip leading and trailing blank lines
1105 while str_field_lines and not str_field_lines[0].strip(): 1105 ↛ 1106line 1105 didn't jump to line 1106 because the condition on line 1105 was never true
1106 str_field_lines.pop(0)
1107 while str_field_lines and not str_field_lines[-1].strip():
1108 str_field_lines.pop()
1109 text = "\n".join(str_field_lines)
1111 emit_as_scalar = "\n" not in text and MD_Lexer._looks_like_scalar_value(
1112 text
1113 )
1115 value_loc = str_field_loc
1116 if str_field_first_content_line is not None: 1116 ↛ 1122line 1116 didn't jump to line 1122 because the condition on line 1116 was always true
1117 value_loc = self._loc(
1118 str_field_first_content_line,
1119 str_field_first_content_col,
1120 )
1122 if in_record: 1122 ↛ 1132line 1122 didn't jump to line 1132 because the condition on line 1122 was always true
1123 # Record already open – emit directly inside the block.
1124 self._emit(str_field_loc, "IDENTIFIER", str_field_name)
1125 self._emit(str_field_loc, "ASSIGN")
1126 self._emit_field_value(
1127 text, value_loc, current_record_type_ast, str_field_name
1128 )
1129 else:
1130 # "type" row not yet seen – buffer until the record opens.
1131 # Check if it looks like an array (but don't emit yet)
1132 is_bracket = (
1133 not emit_as_scalar
1134 and text.strip().startswith("[")
1135 and text.strip().endswith("]")
1136 and "@" in text
1137 )
1138 is_array = (
1139 not emit_as_scalar
1140 and not is_bracket
1141 and MD_Lexer._looks_like_array(text)
1142 )
1143 pending_string_fields.append(
1144 (
1145 str_field_name,
1146 value_loc,
1147 text,
1148 emit_as_scalar,
1149 is_array,
1150 is_bracket,
1151 )
1152 )
1153 in_string_field = False
1154 str_field_name = None
1155 str_field_lines = []
1156 str_field_first_content_line = None
1157 str_field_first_content_col = 1
1159 def flush_record(loc):
1160 nonlocal in_record, pending_name, pending_props
1161 nonlocal record_type_found, props_first_row
1162 if not in_record:
1163 # Nothing open – check for incomplete pending record
1164 if pending_name is not None and not record_type_found: 1164 ↛ 1165line 1164 didn't jump to line 1165 because the condition on line 1164 was never true
1165 self.mh.error(
1166 pending_name_loc,
1167 f"record heading '{pending_name}' has no 'type' property"
1168 " in its property table; record will be skipped",
1169 fatal=False,
1170 )
1171 pending_name = None
1172 pending_props = []
1173 pending_string_fields.clear()
1174 record_type_found = False
1175 props_first_row = False
1176 return
1177 flush_string_field()
1178 self._emit(loc, "C_KET")
1179 in_record = False
1180 pending_name = None
1181 pending_props = []
1182 pending_string_fields.clear()
1183 record_type_found = False
1184 props_first_row = False
1186 def open_section(name, loc):
1187 nonlocal in_section
1188 self._emit(loc, "KEYWORD", "##")
1189 self._emit(loc, "STRING", name)
1190 self._emit(loc, self.MD_SECTION_START_TOKEN)
1191 in_section = True
1193 def close_section(loc):
1194 nonlocal in_section
1195 if in_section:
1196 self._emit(loc, self.MD_SECTION_END_TOKEN)
1197 in_section = False
1199 # ── Line-by-line scan ─────────────────────────────────────────── #
1201 for i, line in enumerate(lines):
1202 line_no = i + 1
1203 loc = self._loc(line_no)
1204 stripped = line.strip()
1206 # ── While inside a string field, only headings / <hr> break out ─
1208 if in_string_field:
1209 level, _ = MD_Lexer._parse_heading(line)
1210 if level in (2, 3, 4) or MD_Lexer._is_hr(stripped):
1211 flush_string_field()
1212 # fall through so the line is processed normally
1213 elif not in_record and stripped.startswith("|"): 1213 ↛ 1218line 1213 didn't jump to line 1218 because the condition on line 1213 was never true
1214 # A property table row arrived while collecting a ####
1215 # string field but before "type" is known. Close the
1216 # string field (buffering its content) so the table row
1217 # is processed normally below.
1218 flush_string_field()
1219 # fall through to table-row handling
1220 else:
1221 if str_field_first_content_line is None and stripped:
1222 line_stripped = line.lstrip()
1223 str_field_first_content_line = line_no
1224 str_field_first_content_col = len(line) - len(line_stripped) + 1
1225 str_field_lines.append(line)
1226 continue
1228 # ── Heading dispatch (H1–H4) ─────────────────────────────────
1230 _h_level, _h_content = MD_Lexer._parse_heading(line)
1232 # ── H1: package declaration ──────────────────────────────────
1234 if _h_level == 1:
1235 parts = _h_content.split()
1236 if len(parts) != 1:
1237 self.mh.lex_error(loc, "package heading must be '# <PackageName>'")
1238 package_name = parts[0]
1239 self._emit(loc, "KEYWORD", "#")
1240 # Package name starts right after the '#' and its following
1241 # whitespace; computed positionally so a name that happens
1242 # to reoccur earlier on the line (e.g. inside "##") can't
1243 # be found at the wrong offset.
1244 name_offset = MD_Lexer._skip_spaces(line, _h_level)
1245 self._emit_package_name(package_name, line_no, name_offset, line)
1246 current_package_name = package_name
1247 continue
1249 # ── H2: section ──────────────────────────────────────────────
1250 # _parse_heading already distinguishes the levels by exact count
1251 # of leading "#" characters, so no prefix-collision is possible.
1253 if _h_level == 2:
1254 flush_record(loc)
1255 close_section(loc)
1256 open_section(_h_content, loc)
1257 continue
1259 # ── H3: record heading ───────────────────────────────────────
1261 if _h_level == 3:
1262 flush_record(loc)
1263 pending_name = _h_content.strip()
1264 # heading_prefix_len = level(3) + 1 space
1265 self._validate_identifier(pending_name, line_no, _h_level + 1, line)
1266 pending_name_loc = loc
1267 pending_props = []
1268 record_type_found = False
1269 props_first_row = True # skip the column-header row
1270 continue
1272 # ── H4: string field heading ─────────────────────────────────
1274 if _h_level == 4:
1275 # flush_string_field already called at the top of the loop
1276 str_field_name = _h_content.strip()
1277 # heading_prefix_len = level(4) + 1 space
1278 self._validate_identifier(str_field_name, line_no, _h_level + 1, line)
1279 str_field_loc = loc
1280 str_field_lines = []
1281 str_field_first_content_line = None
1282 str_field_first_content_col = 1
1283 in_string_field = True
1284 continue
1286 # ── Horizontal rule ──────────────────────────────────────────
1288 if MD_Lexer._is_hr(stripped):
1289 flush_record(loc)
1290 continue
1292 # ── Table rows ───────────────────────────────────────────────
1294 if stripped.startswith("|") and pending_name is not None:
1295 # Skip the column-header row (first row after ###)
1296 if props_first_row:
1297 props_first_row = False
1298 continue
1300 # Skip Markdown separator rows (|---|---|)
1301 if self._is_separator_row(stripped):
1302 continue
1304 row = self._parse_table_row(stripped)
1305 if row is None: 1305 ↛ 1306line 1305 didn't jump to line 1306 because the condition on line 1305 was never true
1306 continue
1307 key, value = row
1308 if not key: 1308 ↛ 1309line 1308 didn't jump to line 1309 because the condition on line 1308 was never true
1309 continue
1311 if key == "type":
1312 # Emit: RecordType RecordName {
1313 type_name = value.strip()
1314 if "." not in type_name and len(imported_packages) == 1:
1315 type_name = imported_packages[0] + "." + type_name
1316 self._emit_qualified_identifier(loc, type_name)
1317 self._emit(pending_name_loc, "IDENTIFIER", pending_name)
1318 self._emit(pending_name_loc, "C_BRA")
1319 in_record = True
1320 record_type_found = True
1321 current_record_type_ast = self._resolve_record_type(
1322 type_name, package_name=current_package_name
1323 )
1325 # Flush any properties that arrived before "type"
1326 for bkey, bval, bline in pending_props: 1326 ↛ 1327line 1326 didn't jump to line 1327 because the loop on line 1326 never started
1327 bloc = self._loc(bline)
1328 self._emit(bloc, "IDENTIFIER", bkey)
1329 self._emit(bloc, "ASSIGN")
1330 self._emit_field_value(
1331 bval, bloc, current_record_type_ast, bkey
1332 )
1333 pending_props = []
1335 # Flush any #### string fields that arrived
1336 # before "type"
1337 for ( 1337 ↛ 1345line 1337 didn't jump to line 1345 because the loop on line 1337 never started
1338 fname,
1339 floc,
1340 ftext,
1341 _fis_scalar,
1342 _fis_array,
1343 _fis_bracket,
1344 ) in pending_string_fields:
1345 self._emit(floc, "IDENTIFIER", fname)
1346 self._emit(floc, "ASSIGN")
1347 self._emit_field_value(
1348 ftext, floc, current_record_type_ast, fname
1349 )
1350 pending_string_fields.clear()
1352 elif record_type_found: 1352 ↛ 1359line 1352 didn't jump to line 1359 because the condition on line 1352 was always true
1353 self._emit(loc, "IDENTIFIER", key)
1354 self._emit(loc, "ASSIGN")
1355 self._emit_field_value(value, loc, current_record_type_ast, key)
1357 else:
1358 # Buffer: "type" has not appeared yet
1359 pending_props.append((key, value, line_no))
1361 continue
1363 # ── Import statement ─────────────────────────────────────────
1365 if stripped.startswith("import "):
1366 parts = stripped.split()
1367 if len(parts) == 2: 1367 ↛ 1394line 1367 didn't jump to line 1394 because the condition on line 1367 was always true
1368 self._emit(loc, "KEYWORD", "import")
1369 # The package name starts right after "import" and its
1370 # following whitespace; computed positionally so a name
1371 # that happens to also occur inside the keyword itself
1372 # (e.g. "rt" in "import rt") isn't found at the wrong
1373 # offset.
1374 indent = len(line) - len(line.lstrip())
1375 name_offset = MD_Lexer._skip_spaces(line, indent + len("import"))
1376 wildcard = self._emit_package_name(
1377 parts[1],
1378 line_no,
1379 name_offset,
1380 line,
1381 allow_wildcard=True,
1382 )
1383 # Only a plain import identifies a single package that
1384 # unqualified type names can be completed with.
1385 if not wildcard:
1386 imported_packages.append(parts[1])
1387 continue
1389 # ── Everything else: delegate to TRLC_Lexer ──────────────────
1390 # Tokenize the raw line so the parser receives real tokens and
1391 # can report a meaningful error (e.g. "expected keyword #,
1392 # encountered OPERATOR instead" for "* foo", or "encountered
1393 # IDENTIFIER instead" for "sadsad foo").
1394 if stripped: 1394 ↛ 1395line 1394 didn't jump to line 1395 because the condition on line 1394 was never true
1395 trlc_lex = TRLC_Lexer(self.mh, self.file_name, line)
1396 tok = trlc_lex.token()
1397 while tok is not None:
1398 self._emit(
1399 self._loc(line_no, tok.location.col_no),
1400 tok.kind,
1401 tok.value,
1402 )
1403 tok = trlc_lex.token()
1405 # ── End-of-file cleanup ──────────────────────────────────────────
1407 eof_loc = self._loc(total_lines + 1)
1408 flush_record(eof_loc)
1409 close_section(eof_loc)