diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 0ae9aac..bf0c749 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -30,3 +30,6 @@ jobs: - name: Tests run: .venv/bin/mojo run -I src test/test_markdown.mojo + + - name: Unit tests — parse-error positions + run: .venv/bin/mojo run -I src test/test_errors.mojo diff --git a/CHANGELOG.md b/CHANGELOG.md index c736898..fd3ea77 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,22 @@ # Changelog +## Unreleased + +New `markdown.errors` module (exported from the package), adopting the +error-reporting pattern shared across the mojo-* parser suite: +`line_col(source, offset)` maps a byte offset to a 1-based (line, column) +pair — the column is the 1-based BYTE offset within the line, no UTF-8 +decoding — and `parse_error(msg, source, offset)` builds an `Error` reading +` at line , column : ''`, where the snippet is up to +~30 bytes of the offending line centered on the column, +whitespace-trimmed, with `...` where truncated, and never multi-line. + +No parser call sites were wired: CommonMark treats every input as valid +markdown, so the parser has no input-error `raise` sites (the container +nesting cap degrades silently rather than raising). The module ships for +suite consistency and for downstream users who validate markdown-adjacent +input themselves. + ## 0.1.0 — 2026-07-05 Initial release. CommonMark block parser (ATX/setext headings, diff --git a/pixi.toml b/pixi.toml index 7f4b2f7..df24f9b 100644 --- a/pixi.toml +++ b/pixi.toml @@ -7,7 +7,7 @@ platforms = ["linux-64", "osx-arm64"] version = "0.1.0" [tasks] -test = "mojo run -I src test/test_markdown.mojo" +test = "mojo run -I src test/test_markdown.mojo && mojo run -I src test/test_errors.mojo" conformance = "mojo run -I src test/conformance.mojo" fuzz = "mojo run -I src test/fuzz_runner.mojo" bench = "mojo build -I src bench/bench_parse.mojo -o .bench_parse && ./.bench_parse" diff --git a/src/markdown/__init__.mojo b/src/markdown/__init__.mojo index 9b6d4ab..844c9ea 100644 --- a/src/markdown/__init__.mojo +++ b/src/markdown/__init__.mojo @@ -14,5 +14,6 @@ from markdown.block import ( B_ITEM, B_BREAK, ) +from markdown.errors import line_col, parse_error from markdown.inline import render_inlines from markdown.render import render_html diff --git a/src/markdown/errors.mojo b/src/markdown/errors.mojo new file mode 100644 index 0000000..a9c0039 --- /dev/null +++ b/src/markdown/errors.mojo @@ -0,0 +1,135 @@ +"""Position-aware parse errors. + +`line_col` maps a byte offset in a source buffer to a 1-based +(line, column) pair, and `parse_error` builds an `Error` whose message +carries that position plus a short snippet of the offending line: + + at line , column : '' + +Positions are byte-based: the column is the 1-based BYTE offset within +the line, not a codepoint or display column. That keeps the computation +deterministic and free of UTF-8 decode cost; for ASCII-heavy markup the +byte column and the visual column coincide. + +This module is the error-reporting pattern shared across the mojo-* +parser suite. +""" + +comptime _LF = UInt8(0x0A) +comptime _CR = UInt8(0x0D) + +# Snippet size budget, in bytes, before the `...` truncation markers are +# added. Wide enough to show meaningful context, narrow enough that error +# messages stay one readable line. +comptime _SNIPPET_BUDGET = 30 + + +def _is_ws(b: UInt8) -> Bool: + return b == 0x20 or b == 0x09 or b == _CR or b == _LF + + +def line_col(source: Span[UInt8, _], offset: Int) -> Tuple[Int, Int]: + """1-based (line, column) of byte `offset` in `source`. + + The column is the 1-based BYTE offset within the line — codepoints + are never decoded, so the result is cheap and deterministic even on + invalid UTF-8. Only LF (0x0A) terminates a line: after a CRLF + sequence the next byte is column 1 of the next line, with no phantom + column contributed by the CR. An offset pointing AT an LF reports + the line that newline terminates (column = line length + 1). Offsets + past the end of `source` clamp to the end; an empty source yields + (1, 1). + """ + var limit = offset + if limit > len(source): + limit = len(source) + if limit < 0: + limit = 0 + var line = 1 + var last_nl = -1 + for i in range(limit): + if source[i] == _LF: + line += 1 + last_nl = i + return (line, limit - last_nl) + + +def _snippet(source: Span[UInt8, _], offset: Int) -> String: + """Up to ~`_SNIPPET_BUDGET` bytes of the line containing `offset`. + + The line is trimmed of leading/trailing whitespace (which also drops + the CR of a CRLF line ending), then windowed around the offset with + `...` marking each side that was cut. Window edges are nudged off + UTF-8 continuation bytes so the result is always valid UTF-8. The + result never contains a newline. + """ + var n = len(source) + var anchor = offset + if anchor > n: + anchor = n + if anchor < 0: + anchor = 0 + # Line bounds around the anchor; an anchor sitting AT an LF belongs + # to the line that newline terminates. + var line_start = anchor + while line_start > 0 and source[line_start - 1] != _LF: + line_start -= 1 + var line_end = anchor + while line_end < n and source[line_end] != _LF: + line_end += 1 + # Trim surrounding whitespace. + while line_start < line_end and _is_ws(source[line_start]): + line_start += 1 + while line_end > line_start and _is_ws(source[line_end - 1]): + line_end -= 1 + var win_start = line_start + var win_end = line_end + var cut_left = False + var cut_right = False + if line_end - line_start > _SNIPPET_BUDGET: + win_start = anchor - _SNIPPET_BUDGET // 2 + if win_start > line_end - _SNIPPET_BUDGET: + win_start = line_end - _SNIPPET_BUDGET + if win_start < line_start: + win_start = line_start + win_end = win_start + _SNIPPET_BUDGET + # Never split a multi-byte UTF-8 sequence at a window edge. + while win_start < win_end and (source[win_start] & 0xC0) == 0x80: + win_start += 1 + while win_end < line_end and (source[win_end] & 0xC0) == 0x80: + win_end += 1 + cut_left = win_start > line_start + cut_right = win_end < line_end + var out = String() + if cut_left: + out += "..." + out += String(StringSlice(unsafe_from_utf8=source[win_start:win_end])) + if cut_right: + out += "..." + return out^ + + +def parse_error(msg: String, source: Span[UInt8, _], offset: Int) -> Error: + """An `Error` locating `msg` at byte `offset` of `source`. + + The message is exactly: + + at line , column : '' + + where line/column come from `line_col` (1-based; column is a byte + offset within the line) and the snippet is the offending line, + whitespace-trimmed and truncated to ~30 bytes centered on the + column, with `...` where truncated. The message never contains a + newline, so it renders on one line in test output and logs. + """ + var lc = line_col(source, offset) + return Error( + msg + + " at line " + + String(lc[0]) + + ", column " + + String(lc[1]) + + ": '" + + _snippet(source, offset) + + "'" + ) diff --git a/test/test_errors.mojo b/test/test_errors.mojo new file mode 100644 index 0000000..812c1bc --- /dev/null +++ b/test/test_errors.mojo @@ -0,0 +1,174 @@ +from std.testing import assert_equal, assert_true, TestSuite + +from markdown import line_col, parse_error + + +def _assert_lc(source: String, offset: Int, line: Int, col: Int) raises: + var lc = line_col(source.as_bytes(), offset) + assert_equal(lc[0], line) + assert_equal(lc[1], col) + + +def _msg(e: Error) -> String: + return String.write(e) + + +# -------------------------------------------------------------------------- +# line_col unit tests — every documented edge case. +# -------------------------------------------------------------------------- + + +def test_line_col_offset_zero() raises: + _assert_lc("abc", 0, 1, 1) + + +def test_line_col_empty_source() raises: + _assert_lc("", 0, 1, 1) + _assert_lc("", 7, 1, 1) + + +def test_line_col_negative_offset_clamps() raises: + _assert_lc("abc", -5, 1, 1) + + +def test_line_col_middle_of_lines() raises: + # "ab\ncd": a=0 b=1 \n=2 c=3 d=4 + _assert_lc("ab\ncd", 1, 1, 2) + _assert_lc("ab\ncd", 3, 2, 1) + _assert_lc("ab\ncd", 4, 2, 2) + + +def test_line_col_offset_at_newline() raises: + # An offset AT a '\n' reports the line that newline terminates. + _assert_lc("ab\ncd", 2, 1, 3) + + +def test_line_col_offset_past_end_clamps() raises: + _assert_lc("ab\ncd", 5, 2, 3) # == len + _assert_lc("ab\ncd", 99, 2, 3) # > len + + +def test_line_col_crlf_no_phantom_column() raises: + # "ab\r\ncd": a=0 b=1 \r=2 \n=3 c=4 d=5. The byte after a CRLF is + # column 1 of the next line — the '\r' contributes no phantom column. + _assert_lc("ab\r\ncd", 4, 2, 1) + _assert_lc("ab\r\ncd", 5, 2, 2) + _assert_lc("ab\r\ncd", 2, 1, 3) # at the '\r' + _assert_lc("ab\r\ncd", 3, 1, 4) # at the '\n' of the CRLF + + +def test_line_col_consecutive_newlines() raises: + # "a\n\nb": a=0 \n=1 \n=2 b=3 + _assert_lc("a\n\nb", 2, 2, 1) + _assert_lc("a\n\nb", 3, 3, 1) + + +def test_line_col_trailing_newline() raises: + _assert_lc("ab\n", 3, 2, 1) + + +# -------------------------------------------------------------------------- +# parse_error unit tests — exact message format + snippet behavior. +# -------------------------------------------------------------------------- + + +def test_parse_error_exact_format() raises: + var e = parse_error("boom", "hello".as_bytes(), 2) + assert_equal(_msg(e), "boom at line 1, column 3: 'hello'") + + +def test_parse_error_multiline_source_single_line_message() raises: + # Offset 8 is the 'n' of "line2"; snippet is that line only — the + # message never embeds a newline. + var e = parse_error("bad", "line1\nline2\nline3".as_bytes(), 8) + assert_equal(_msg(e), "bad at line 2, column 3: 'line2'") + + +def test_parse_error_snippet_trims_whitespace() raises: + # " pad ": offset 4 is the 'a'. Column counts the raw bytes, the + # snippet is the trimmed line content. + var e = parse_error("boom", " pad ".as_bytes(), 4) + assert_equal(_msg(e), "boom at line 1, column 5: 'pad'") + + +def test_parse_error_offset_at_newline_snippet_is_ended_line() raises: + var e = parse_error("boom", "ab\ncd".as_bytes(), 2) + assert_equal(_msg(e), "boom at line 1, column 3: 'ab'") + + +def test_parse_error_crlf_line_has_no_stray_cr() raises: + # The snippet for a CRLF-terminated line drops the '\r' (whitespace + # trim), so the quoted snippet is clean. + var e = parse_error("boom", "ab\r\ncd".as_bytes(), 1) + assert_equal(_msg(e), "boom at line 1, column 2: 'ab'") + + +def test_parse_error_offset_past_end_clamps() raises: + var e = parse_error("eof", "ab\ncd".as_bytes(), 99) + assert_equal(_msg(e), "eof at line 2, column 3: 'cd'") + + +def test_parse_error_empty_source() raises: + var e = parse_error("boom", "".as_bytes(), 0) + assert_equal(_msg(e), "boom at line 1, column 1: ''") + + +def test_parse_error_truncates_both_sides() raises: + # An 80-byte line with the offending '!' at offset 40: the snippet is + # a 30-byte window centered on it, with '...' on both cut sides. + var source = String() + for _ in range(40): + source += "x" + source += "!" + for _ in range(39): + source += "x" + var e = parse_error("bang", source.as_bytes(), 40) + assert_equal( + _msg(e), + "bang at line 1, column 41: '...xxxxxxxxxxxxxxx!xxxxxxxxxxxxxx...'", + ) + + +def test_parse_error_truncates_right_only() raises: + var source = String("abcde") + for _ in range(75): + source += "x" + var e = parse_error("bang", source.as_bytes(), 0) + assert_equal( + _msg(e), + "bang at line 1, column 1: 'abcdexxxxxxxxxxxxxxxxxxxxxxxxx...'", + ) + + +def test_parse_error_truncates_left_only() raises: + var source = String() + for _ in range(79): + source += "x" + source += "!" + var e = parse_error("bang", source.as_bytes(), 79) + assert_equal( + _msg(e), + "bang at line 1, column 80: '...xxxxxxxxxxxxxxxxxxxxxxxxxxxxx!'", + ) + + +def test_parse_error_snippet_never_splits_utf8() raises: + # 40 'é' (2 bytes each, 80 bytes total); an offset landing mid-sequence + # still yields a valid-UTF-8 snippet: window edges are nudged off + # continuation bytes, so the snippet is whole codepoints only. + var source = String() + for _ in range(40): + source += "é" + var e = parse_error("bang", source.as_bytes(), 40) + var rendered = _msg(e) + assert_true(rendered.startswith("bang at line 1, column 41: '...")) + # 30-byte budget at a 2-byte-char boundary nudge = 15 whole 'é'. + var expected_snippet = String("'...") + for _ in range(15): + expected_snippet += "é" + expected_snippet += "...'" + assert_true(expected_snippet in rendered) + + +def main() raises: + TestSuite.discover_tests[__functions_in_module()]().run()