From 4775df3d04f0fbf5c58941f9d58ce285d9f7b01f Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 10:19:30 +0000 Subject: [PATCH 01/30] Add load(path) for opening a document; parse replaces parse_document (deprecated, removed in 0.5.0) load() reads a file as UTF-8 and sets Document.filename and the new Document.path, the base later multi-file resolution can use. parse() is the string form; parse_document() stays as a warning alias. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_013WmBAc5T7UCKVdpUmg9qxz --- CONFORMANCE.md | 2 +- README.md | 28 +- src/legaldown/__init__.py | 10 +- src/legaldown/assembly.py | 10 +- src/legaldown/cli.py | 9 +- src/legaldown/models.py | 14 +- src/legaldown/parser.py | 43 +- src/legaldown/positions.py | 2 +- src/legaldown/serializer.py | 2 +- src/legaldown/validator/result.py | 2 +- tests/conformance/test_legaldown_fixtures.py | 10 +- tests/test_assembly.py | 10 +- tests/test_cli.py | 17 +- tests/test_conditions.py | 6 +- tests/test_identifiers.py | 4 +- tests/test_positions.py | 22 +- tests/test_public_api.py | 95 +++- tests/test_spec_alignment.py | 520 +++++++++---------- tests/test_templates.py | 96 ++-- 19 files changed, 517 insertions(+), 385 deletions(-) diff --git a/CONFORMANCE.md b/CONFORMANCE.md index e21353d..66d326e 100644 --- a/CONFORMANCE.md +++ b/CONFORMANCE.md @@ -162,7 +162,7 @@ failed to resolve. From assembly, which reads the file, it means the file could and for one whose `---` block holds YAML that is a scalar or a list rather than a mapping of fields: that block is not frontmatter, its `---` lines are thematic breaks, and the whole document is validated as body. A block that cannot be read at all — YAML that is malformed, nested too -deep, or a mapping of another kind (`!!set`) — is `frontmatter-invalid-yaml`: `parse_document` +deep, or a mapping of another kind (`!!set`) — is `frontmatter-invalid-yaml`: `parse` (and `load`) raises `FrontmatterError` for it, which the CLI reports and assembly returns as a diagnostic, for the template or for a fragment or attachment file it reads. Any other exception while parsing is a fault of this implementation: the CLI reports it as an internal error (exit status 2), never as diff --git a/README.md b/README.md index df63881..07b2114 100644 --- a/README.md +++ b/README.md @@ -179,9 +179,9 @@ off. Three functions cover the common path: ```python -from legaldown import parse_document, validate_document, serialize_document +from legaldown import load, validate_document, serialize_document -document = parse_document(open("contract.lgd").read(), filename="contract.lgd") +document = load("contract.lgd") result = validate_document(document) for diagnostic in result.diagnostics: @@ -191,9 +191,19 @@ if result.is_valid: # no Error-level diagnostics print(serialize_document(document)) ``` -`parse_document` raises `FrontmatterError` (a `yaml.YAMLError` and a `ValueError`) when the -frontmatter cannot be read — the `frontmatter-invalid-yaml` rule, which `validate_document`, -given a parsed document, cannot report. +`load(path)` takes a `str` or `os.PathLike`, reads the file as UTF-8 and returns the `Document`. +It names the document for you: `document.filename` is the file's name and `document.path` its +absolute path. For text that is not in a file (an editor buffer, an HTTP body, a test), use +`parse(text)`, which `load` is built on; it takes an optional `filename=` to name the document in +its diagnostics. + +`load` raises `FileNotFoundError` (or another `OSError`) when the file cannot be read and +`UnicodeDecodeError` when it is not UTF-8. Both it and `parse` raise `FrontmatterError` (a +`yaml.YAMLError` and a `ValueError`) when the frontmatter cannot be read — the +`frontmatter-invalid-yaml` rule, which `validate_document`, given a parsed document, cannot report. + +> **Deprecated:** `parse_document` is now `parse` (string) or `load` (file). It still works and +> raises a `DeprecationWarning`; it is deprecated since 0.4.0 and will be removed in 0.5.0. The public API is what the `legaldown` and `legaldown.validator` packages export (their `__all__`). Changes to it are listed in the notes of each @@ -234,13 +244,13 @@ validator reads the document with. ### Reading and editing the document model -`parse_document` returns a `Document` of plain dataclasses — `Metadata`, `Section`, `Block`, +`load` and `parse` return a `Document` of plain dataclasses — `Metadata`, `Section`, `Block`, `Side`, `Party`, `Attachment` — that you can inspect, edit, and write back out: ```python -from legaldown import parse_document, serialize_document +from legaldown import load, serialize_document -document = parse_document(source) +document = load("contract.lgd") document.metadata.governing_law = "Czech Republic" with open("contract.lgd", "w", encoding="utf-8") as handle: @@ -262,7 +272,7 @@ structures, except the fields that describe the parsed source rather than the do Each diagnostic names its `line` (from 1) and its `file` (the document's `filename`), as §16.9 requires: the line of the directive, marker, heading or block it is about, or of the frontmatter key — for a missing key, the key that holds it, or the frontmatter's first. Lines -come from the `source_map` that `parse_document` gives a document. A document built from a dict +come from the `source_map` that `load` and `parse` give a document. A document built from a dict has none, and one changed after parsing no longer fits its map: their diagnostics have no line (`None`) rather than a stale one. `FrontmatterError.line` is the line of YAML that cannot be read. `render_block()` renders a single block when you are driving your own layout. diff --git a/src/legaldown/__init__.py b/src/legaldown/__init__.py index eae6f84..dc55a0f 100644 --- a/src/legaldown/__init__.py +++ b/src/legaldown/__init__.py @@ -6,9 +6,9 @@ Quick start:: - from legaldown import parse_document, serialize_document, validate_document + from legaldown import load, serialize_document, validate_document - doc = parse_document(open("contract.lgd").read()) + doc = load("contract.lgd") result = validate_document(doc) for diagnostic in result.diagnostics: print(diagnostic.level, diagnostic.rule, diagnostic.message) @@ -77,7 +77,7 @@ ) # Parser & serializer -from .parser import FrontmatterError, collect_source_directives, parse_document +from .parser import FrontmatterError, collect_source_directives, load, parse, parse_document from .serializer import render_block, render_item, serialize_document # The LegalDown specification version this implementation targets. @@ -117,7 +117,9 @@ "CONFORMANCE_LEVEL", "CAPABILITIES", # Core workflow - "parse_document", + "load", + "parse", + "parse_document", # deprecated since 0.4.0, removed in 0.5.0: use ``parse`` or ``load`` "FrontmatterError", "serialize_document", "validate_document", diff --git a/src/legaldown/assembly.py b/src/legaldown/assembly.py index 0d54648..411f8d0 100644 --- a/src/legaldown/assembly.py +++ b/src/legaldown/assembly.py @@ -59,7 +59,7 @@ _may_interrupt, _opens_paragraph, _row_width_end, - parse_document, + parse, quote_content, ) from .validator import validate_document @@ -303,7 +303,7 @@ def _read_body( ) -> _Source: lines = body.split("\n") if lines[-1] == "": - lines.pop() # the last line's ending, not a line -- as parse_document reads it + lines.pop() # the last line's ending, not a line -- as parse reads it layout = _layout(lines) found: dict[tuple[int | None, int], list] = {} for marker in find_markers(document, cache(lex)): @@ -661,7 +661,7 @@ class _Template: def _read(template: str, load_file: LoadFile | None) -> _Template: bom = "\ufeff" if template.startswith("\ufeff") else "" source, newline = _unix(template) - document = parse_document(source) + document = parse(source) declared = document.metadata.questions if isinstance(document.metadata.questions, dict) else {} template_like = is_template(document) front, body = _split(source, document) @@ -711,7 +711,7 @@ def _read_files(main: _Source, front: _Frontmatter | None, load_file: LoadFile | continue text, newline = _unix(text) try: - sub_document = parse_document(text) + sub_document = parse(text) except FrontmatterError as exc: problems.append(Diagnostic( "frontmatter-invalid-yaml", "error", @@ -1392,7 +1392,7 @@ def _identifiers(head: str, rows: list[_Row]) -> dict[_Key, tuple[str, bool]]: """Each heading's identifier in the combined document, and whether it is explicit — generated by the validator, which knows alternatives (§5.5).""" lines = [text for text, _key in rows] - document = parse_document(head + "\n".join(lines)) + document = parse(head + "\n".join(lines)) entries = validate_document(document).sections headings = _layout(lines).headings if len(headings) != len(entries): diff --git a/src/legaldown/cli.py b/src/legaldown/cli.py index ae0b810..bb82bfa 100644 --- a/src/legaldown/cli.py +++ b/src/legaldown/cli.py @@ -25,7 +25,7 @@ from . import SPEC_VERSION, __version__ from .assembly import assemble -from .parser import FrontmatterError, parse_document +from .parser import FrontmatterError, load from .validator import validate_document # Exit codes: 0 clean, 1 diagnostics found, 2 usage/IO failure. @@ -39,12 +39,9 @@ def _validate_path(path: Path, *, final: bool = False) -> tuple[list[dict], str | None]: """Validate one file; return (diagnostics, read/parse failure message).""" try: - source = path.read_text(encoding="utf-8") - except OSError as exc: + document = load(path) + except (OSError, UnicodeDecodeError) as exc: return [], f"cannot read {path}: {exc}" - - try: - document = parse_document(source, filename=path.name) except FrontmatterError as exc: return [ { diff --git a/src/legaldown/models.py b/src/legaldown/models.py index 18ed8df..50e9685 100644 --- a/src/legaldown/models.py +++ b/src/legaldown/models.py @@ -7,6 +7,7 @@ from collections.abc import Iterator from dataclasses import asdict, dataclass, field +from pathlib import Path from typing import Any from .markdown import LINE_ENDING_RE, is_blank, strip_text @@ -105,12 +106,12 @@ class Metadata: #: Frontmatter keys (``questions``, ``attachments``) the parsed source #: does not write as §15.2 requires for assembly to edit them line by #: line: in YAML block style, each attachment entry beginning with - #: ``id``. Only ``parse_document`` sets it: a document built or rebuilt + #: ``id``. Only ``parse`` and ``load`` set it: a document built or rebuilt #: from a dict has no source, and the serializer writes block style. not_line_editable: list[str] = field(default_factory=list) #: True when the parsed source has no frontmatter (§3.1): no ``---`` #: block opens it, or the block's YAML is not a mapping of fields, in - #: which case the source is all body. Only ``parse_document`` sets it, + #: which case the source is all body. Only ``parse`` and ``load`` set it, #: like ``not_line_editable``. frontmatter_absent: bool = False @@ -189,9 +190,14 @@ class Document: sections: list[Section] = field(default_factory=list) filename: str = "" preamble: list[Block] = field(default_factory=list) + #: The file the document was loaded from, absolute (``load`` sets it): the + #: base that files it refers to resolve against. ``None`` for a document + #: parsed from a string or built in code. Not part of equality or of + #: ``document_to_dict``: where a document lives is not what it says. + path: Path | None = field(default=None, compare=False) #: Where the parsed source holds its parts (``positions.SourceMap``), for - #: diagnostics that name their line (§16.9). Only ``parse_document`` sets - #: it; it describes the source as parsed, so a document changed afterwards + #: diagnostics that name their line (§16.9). Only ``parse`` and ``load`` + #: set it; it describes the source as parsed, so a document changed afterwards #: may no longer fit it (then its diagnostics name no line). Not part of #: equality or of ``document_to_dict``. source_map: Any = field(default=None, compare=False, repr=False) diff --git a/src/legaldown/parser.py b/src/legaldown/parser.py index b653fe6..e6f443e 100644 --- a/src/legaldown/parser.py +++ b/src/legaldown/parser.py @@ -9,10 +9,13 @@ from __future__ import annotations import bisect +import os import re +import warnings from collections.abc import Iterator from dataclasses import asdict, dataclass, field from functools import lru_cache +from pathlib import Path from typing import Any import yaml @@ -152,7 +155,7 @@ class FrontmatterError(yaml.YAMLError, ValueError): """The frontmatter cannot be read (§3.1, frontmatter-invalid-yaml): its YAML is malformed, nested too deep, or a mapping of another kind (a ``!!set``). A ``yaml.YAMLError`` and a ``ValueError``, as what - ``parse_document`` raised for it before. ``line`` is the file line (from + ``parse`` raised for it before. ``line`` is the file line (from 1) of the problem, when the YAML reader gives one (§16.9).""" def __init__(self, message: str, line: int | None = None) -> None: @@ -1366,9 +1369,34 @@ def parse_item_content(text: str) -> list[Block]: return blocks -def parse_document(source: str, *, filename: str = "") -> Document: +def load(path: str | os.PathLike[str]) -> Document: + """Open the LegalDown file at *path* as a Document. + + The one call for a document that lives in a file: it reads the file as + UTF-8 (a byte-order mark is dropped) and parses it as ``parse`` does, + naming it in the result: ``Document.filename`` is the file's name and + ``Document.path`` its absolute path, the base that files it refers to + (``amends``, includes, attachments) resolve against. + + Raises ``FileNotFoundError`` (or another ``OSError``) when the file cannot + be read, ``UnicodeDecodeError`` when it is not UTF-8, and + ``FrontmatterError`` when its frontmatter cannot be read. + """ + file = Path(path).resolve() + # Bytes, decoded once: line endings are the parser's to read (LF, CR and + # CRLF alike), so the text layer's own translation would only be a second, + # redundant pass over the file. + document = parse(file.read_bytes().decode("utf-8-sig"), filename=file.name) + document.path = file + return document + + +def parse(source: str, *, filename: str = "") -> Document: """Parse a LegalDown source string into a Document object. + For a file, ``load`` reads and parses it in one call. ``filename`` only + names the document in its diagnostics. + The parser is deliberately faithful to the source: nothing is rewritten to make a document valid, so the validator reports what the document actually says (a bare ``unit=M`` surfaces as duration-invalid-unit rather than being @@ -1413,6 +1441,17 @@ def parse_document(source: str, *, filename: str = "") -> Document: return document +def parse_document(source: str, *, filename: str = "") -> Document: + """Deprecated alias of ``parse``; to be removed in 0.5.0.""" + warnings.warn( + "legaldown.parse_document() is deprecated since 0.4.0 and will be removed in 0.5.0; " + "use legaldown.parse() for a string, or legaldown.load() for a file", + DeprecationWarning, + stacklevel=2, + ) + return parse(source, filename=filename) + + def collect_source_directives(document: Document) -> tuple[set[str], set[str]]: """Collect all ref and term targets used in a document. diff --git a/src/legaldown/positions.py b/src/legaldown/positions.py index a8c9621..6045973 100644 --- a/src/legaldown/positions.py +++ b/src/legaldown/positions.py @@ -1,7 +1,7 @@ """Where a parsed document's parts lie in its source, for diagnostics that name their line (§16.9). -``parse_document`` gives each document a ``SourceMap``; a document built from +``parse`` gives each document a ``SourceMap``; a document built from a dict has none, and its diagnostics carry no line. """ from __future__ import annotations diff --git a/src/legaldown/serializer.py b/src/legaldown/serializer.py index 15168d2..92a900e 100644 --- a/src/legaldown/serializer.py +++ b/src/legaldown/serializer.py @@ -186,7 +186,7 @@ def _paragraph_line(text: str, *, in_item: bool, alone: bool) -> str: Anything else that would open a block (a fence, a heading, an HTML block, a block quote, a thematic break, a list item) gets a backslash before it instead, which renders as nothing (CommonMark): only a model built in - code, not one from ``parse_document``, holds that, since the parser + code, not one from ``parse``, holds that, since the parser never turns the start of a block into paragraph text.""" if alone and (split := split_lone_tag(text)): return split diff --git a/src/legaldown/validator/result.py b/src/legaldown/validator/result.py index 214bb04..78f25e4 100644 --- a/src/legaldown/validator/result.py +++ b/src/legaldown/validator/result.py @@ -31,7 +31,7 @@ class Diagnostic: part of a diagnostic that is stable across implementations and spec revisions (§16.9) — plus the severity level and human-readable message, and where it is (§16.9): the file, and the line (from 1) of what it - reports, which a document parsed from source has (``parse_document``) + reports, which a document parsed from source has (``parse``) and one built from a dict has not (None). Compare diagnostics by their fields, not with one built without a line. """ diff --git a/tests/conformance/test_legaldown_fixtures.py b/tests/conformance/test_legaldown_fixtures.py index a9454fc..2ec73e8 100644 --- a/tests/conformance/test_legaldown_fixtures.py +++ b/tests/conformance/test_legaldown_fixtures.py @@ -31,7 +31,7 @@ import yaml from legaldown import CAPABILITIES, assemble -from legaldown.parser import parse_document +from legaldown.parser import parse from legaldown.validator import validate_document FIXTURES_DIR = os.environ.get("LEGALDOWN_FIXTURES_DIR", "") @@ -163,7 +163,7 @@ def _validate_file(path: Path, config: dict): reports the answer rules (§16.12).""" if "answers" in config: return _assemble_case(path, path.parent / config["answers"]) - document = parse_document(path.read_text(encoding="utf-8"), filename=path.name) + document = parse(path.read_text(encoding="utf-8"), filename=path.name) return validate_document(document, final=bool(config.get("final"))) @@ -244,10 +244,10 @@ def test_a_fixture_keeps_its_diagnostics_when_written_back(path: Path): from legaldown.parser import FrontmatterError try: - document = parse_document(_read(path), filename=path.name) + document = parse(_read(path), filename=path.name) except FrontmatterError: pytest.skip("unreadable frontmatter: nothing to write back") - again = parse_document(serialize_document(document), filename=path.name) + again = parse(serialize_document(document), filename=path.name) # How the source writes its YAML is not the model's (block style when # written back). assert replace(again.metadata, not_line_editable=[]) == replace(document.metadata, not_line_editable=[]) @@ -305,4 +305,4 @@ def test_assembly_case_assembles_byte_for_byte(case: Path): assert result.ok, result.diagnostics assert {"template.lgd": result.output, **result.files} == _expected_tree(case) # The assembly guarantee (§15.7.4): the output has no Errors either. - assert not validate_document(parse_document(result.output)).errors + assert not validate_document(parse(result.output)).errors diff --git a/tests/test_assembly.py b/tests/test_assembly.py index d4c80b6..bda37d5 100644 --- a/tests/test_assembly.py +++ b/tests/test_assembly.py @@ -26,7 +26,7 @@ AssemblyResult, assemble, needed_questions, - parse_document, + parse, template_questions, validate_document, ) @@ -109,9 +109,9 @@ def test_fixture_template_and_its_output_have_no_errors(name): """The fixtures' premise (a template without Errors) and the assembly guarantee (§15.7.4): its output has none either.""" template = (FIXTURES / name / "template.lgd").read_text(encoding="utf-8") - assert validate_document(parse_document(template)).errors == [] + assert validate_document(parse(template)).errors == [] result = _assemble_case(name) - assert validate_document(parse_document(result.output)).errors == [] + assert validate_document(parse(result.output)).errors == [] # ── Helpers for the focused tests ──────────────────────────────── @@ -394,7 +394,7 @@ def test_an_unanswered_declared_blank_keeps_its_type_inline(self): "{{placeholder: day, type=date}}.\n" ) assert "questions" not in result.output - assert validate_document(parse_document(result.output)).errors == [] + assert validate_document(parse(result.output)).errors == [] def test_an_undeclared_or_already_typed_blank_is_left_as_written(self): template = _template("A {{placeholder: a}} and {{placeholder: b, type=date}}.\n") @@ -1041,7 +1041,7 @@ class TestHeadingBlocks: def test_a_blank_in_a_heading_block_is_filled(self, body): result = assemble(_template(body, _TEXT), {"name": "Acme"}) assert _body(result) == "\n" + body.replace("{{placeholder: name}}", "Acme") - assert "placeholder-unfilled" not in validate_document(parse_document(result.output)).rules() + assert "placeholder-unfilled" not in validate_document(parse(result.output)).rules() def test_a_marker_after_a_heading_block_is_no_condition(self): """An item's marker ends its first paragraph (§5.7): after a heading diff --git a/tests/test_cli.py b/tests/test_cli.py index 1b770d5..25f5aeb 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -5,7 +5,7 @@ import pytest -from legaldown import parse_document +from legaldown import load from legaldown.cli import EXIT_DIAGNOSTICS, EXIT_ERROR, EXIT_OK, main _VALID = """--- @@ -94,6 +94,13 @@ def test_missing_file_exits_two(tmp_path, capsys): assert "cannot read" in capsys.readouterr().err +def test_a_file_that_is_not_utf8_is_unreadable_not_a_crash(tmp_path, capsys): + path = tmp_path / "latin.lgd" + path.write_bytes(b"---\ntitle: caf\xe9\n---\n") + assert main(["validate", str(path)]) == EXIT_ERROR + assert "cannot read" in capsys.readouterr().err + + def test_malformed_frontmatter_is_reported_not_raised(write, capsys): path = write("bad.lgd", "---\ntitle: [unclosed\n---\n\n# Scope {#scope}\n") assert main(["validate", "--format", "json", str(path)]) == EXIT_DIAGNOSTICS @@ -122,12 +129,12 @@ def test_a_parser_fault_is_an_internal_error_not_a_diagnostic(write, capsys, mon # failure to report, and the other files are still validated. import legaldown.cli - def broken(source, *, filename=""): - if filename == "a.lgd": + def broken(path): + if path.name == "a.lgd": raise AttributeError("'set' object has no attribute 'get'") - return parse_document(source, filename=filename) + return load(path) - monkeypatch.setattr(legaldown.cli, "parse_document", broken) + monkeypatch.setattr(legaldown.cli, "load", broken) first, second = write("a.lgd", _VALID), write("b.lgd", _BROKEN) assert main(["validate", "--format", "json", str(first), str(second)]) == EXIT_ERROR captured = capsys.readouterr() diff --git a/tests/test_conditions.py b/tests/test_conditions.py index f08daf8..f9eaea1 100644 --- a/tests/test_conditions.py +++ b/tests/test_conditions.py @@ -6,7 +6,7 @@ from legaldown import serialize_document from legaldown.markers import Marker, parse_marker, split_heading -from legaldown.parser import parse_document +from legaldown.parser import parse from legaldown.validator import validate_document _SIDES = """sides: @@ -31,7 +31,7 @@ def _parse(body: str, frontmatter: str = _QUESTIONS): - return parse_document(f"---\ntitle: Fixture\n{_SIDES}{frontmatter}---\n\n{body}\n") + return parse(f"---\ntitle: Fixture\n{_SIDES}{frontmatter}---\n\n{body}\n") def _rules(body: str, frontmatter: str = _QUESTIONS, **options) -> set[str]: @@ -383,7 +383,7 @@ def test_a_comment_may_follow_a_heading_marker(): document = _parse("# Termination {when=vat} \n\nText.") [section] = document.sections assert (section.title, section.condition) == ("Termination ", "vat") - assert parse_document(serialize_document(document)).sections[0].condition == "vat" + assert parse(serialize_document(document)).sections[0].condition == "vat" def test_a_dotted_path_is_not_a_reference(): diff --git a/tests/test_identifiers.py b/tests/test_identifiers.py index 0d87b65..e5a7e3a 100644 --- a/tests/test_identifiers.py +++ b/tests/test_identifiers.py @@ -3,7 +3,7 @@ import pytest -from legaldown.parser import parse_document +from legaldown.parser import parse from legaldown.validator import validate_document from legaldown.validator.helpers import generate_identifier, slugify_identifier @@ -56,7 +56,7 @@ def test_a_slug_is_lossy_when_a_letter_or_digit_is_dropped(text, lossy): def _rules(body: str) -> list[str]: - document = parse_document(f"---\ntitle: Fixture\n---\n\n{body}\n") + document = parse(f"---\ntitle: Fixture\n---\n\n{body}\n") return [d.rule for d in validate_document(document).diagnostics] diff --git a/tests/test_positions.py b/tests/test_positions.py index 6ce2211..885f0ff 100644 --- a/tests/test_positions.py +++ b/tests/test_positions.py @@ -6,7 +6,7 @@ import pytest -from legaldown import document_from_dict, document_to_dict, parse_document +from legaldown import document_from_dict, document_to_dict, parse from legaldown.cli import main from legaldown.models import Block from legaldown.parser import FrontmatterError @@ -29,7 +29,7 @@ def _lines(source: str, rule: str) -> list[int | None]: - result = validate_document(parse_document(source, filename="t.lgd")) + result = validate_document(parse(source, filename="t.lgd")) return [d.line for d in result.diagnostics if d.rule == rule] @@ -73,7 +73,7 @@ def test_a_merged_key_is_where_it_is_written(): "title: T\nsides:\n - &base\n name: providers\n parties:\n - name: acme\n" " type: legal_entity\n - <<: *base\n name: clients\n" ) - document = parse_document(f"---\n{frontmatter}---\n\n# A\n") + document = parse(f"---\n{frontmatter}---\n\n# A\n") assert document.source_map.key("sides", 1, "parties", 0, "name") == 7 assert document.source_map.key("sides", 1, "name") == 10 assert _lines(f"---\n{frontmatter}---\n\n# A\n", "party-name-duplicate") == [7] @@ -85,7 +85,7 @@ def test_many_directives_in_one_block_are_found_quickly(): rows = "\n".join(f"| {{{{ref: bad-{n}}}}} | x |" for n in range(3000)) source = _HEAD + "| a | b |\n|---|---|\n" + rows + "\n\n" + " ".join(f"{{{{ref: p-{n}}}}}" for n in range(3000)) - document = parse_document(source) + document = parse(source) start = time.perf_counter() lines = [d.line for d in validate_document(document).diagnostics if d.rule == "ref-broken"] assert time.perf_counter() - start < 5 @@ -95,10 +95,10 @@ def test_many_directives_in_one_block_are_found_quickly(): def test_frontmatter_that_cannot_be_read_names_its_line(): with pytest.raises(FrontmatterError) as raised: - parse_document("---\ntitle: Fixture\n bad indent: [unclosed\n---\n\n# A\n") + parse("---\ntitle: Fixture\n bad indent: [unclosed\n---\n\n# A\n") assert raised.value.line == 3 with pytest.raises(FrontmatterError) as raised: - parse_document('---\ntitle: T\nsubtitle: "open\nlanguage: en\n---\n') + parse('---\ntitle: T\nsubtitle: "open\nlanguage: en\n---\n') assert raised.value.line == 3 # where the quote opens @@ -162,7 +162,7 @@ def test_a_document_without_frontmatter_counts_from_its_first_line(): def test_every_diagnostic_names_its_file(): - result = validate_document(parse_document(_HEAD + "{{ref: nope}}\n", filename="contract.lgd")) + result = validate_document(parse(_HEAD + "{{ref: nope}}\n", filename="contract.lgd")) assert {d.file for d in result.diagnostics} == {"contract.lgd"} @@ -170,7 +170,7 @@ def test_every_diagnostic_names_its_file(): def test_a_document_from_a_dict_has_no_lines(): - document = parse_document(_HEAD + "{{ref: nope}}\n") + document = parse(_HEAD + "{{ref: nope}}\n") rebuilt = document_from_dict(document_to_dict(document)) assert rebuilt == document and rebuilt.source_map is None assert "source_map" not in document_to_dict(document) @@ -180,10 +180,10 @@ def test_a_document_from_a_dict_has_no_lines(): def test_a_document_changed_after_parsing_names_no_line(): """Its source map describes the source as parsed: stale lines would point at the wrong text.""" - document = parse_document(_HEAD + "{{ref: nope}}\n\nText.\n") + document = parse(_HEAD + "{{ref: nope}}\n\nText.\n") document.sections[0].blocks.insert(0, Block(kind="paragraph", text="New.")) assert [d.line for d in validate_document(document).diagnostics] == [None] - document = parse_document(_HEAD + "Text {{ref: nope}} and {{term: nope}}.\n") + document = parse(_HEAD + "Text {{ref: nope}} and {{term: nope}}.\n") document.sections[0].blocks[0].suffix = " and {{term: other}}." assert {d.line for d in validate_document(document).diagnostics} == {None} @@ -245,7 +245,7 @@ def test_each_reference_is_reported_at_its_own_line(seed): body, refs = _piece(r, n) real += [(target, len(lines) + index + 1) for target, index in refs] lines += [*body, ""] - result = validate_document(parse_document("\n".join(lines))) + result = validate_document(parse("\n".join(lines))) got = sorted( (int(d.message.split("'")[1].split("-")[1]), d.line) for d in result.diagnostics if d.rule == "ref-broken" ) diff --git a/tests/test_public_api.py b/tests/test_public_api.py index 18e2403..99c8b81 100644 --- a/tests/test_public_api.py +++ b/tests/test_public_api.py @@ -18,7 +18,8 @@ is_drafting_note, is_template, list_fragments, - parse_document, + load, + parse, validate_document, ) from legaldown.directives import lex @@ -28,7 +29,7 @@ def _markers(body: str) -> list[PlacedMarker]: - return validate_document(parse_document(_FRONTMATTER + body)).placed_markers + return validate_document(parse(_FRONTMATTER + body)).placed_markers @pytest.mark.parametrize("module", [legaldown, legaldown.validator]) @@ -65,7 +66,7 @@ def test_a_list_items_marker_names_its_item(): markers = _markers(body) assert [(m.identifier, m.fragment, m.item) for m in markers] == [("a", 0, 0), ("b", 2, 1), ("c", 3, 2)] assert markers[2].condition == "x" - [block] = parse_document(_FRONTMATTER + body).sections[0].blocks + [block] = parse(_FRONTMATTER + body).sections[0].blocks assert [list_fragments(block)[m.fragment][2][-1] for m in markers] == [0, 1, 2] @@ -76,7 +77,7 @@ def test_an_include_only_paragraphs_identifier_does_not_apply(): def test_a_preamble_condition_applies_only_in_a_template(): assert _markers("Intro {when=a}\n\n# A\n\nText.\n") == [] - result = validate_document(parse_document(_FRONTMATTER + "Intro {when=a}\n\n# A {when=b}\n\nText.\n")) + result = validate_document(parse(_FRONTMATTER + "Intro {when=a}\n\n# A {when=b}\n\nText.\n")) assert result.is_template assert [(m.section, m.condition) for m in result.placed_markers] == [(None, "a")] @@ -104,12 +105,12 @@ def test_a_document_built_in_code_has_no_lines(): def test_is_template_on_its_own(): - assert not is_template(parse_document(_FRONTMATTER + "# A\n\nText.\n")) - assert is_template(parse_document(_FRONTMATTER + "# A\n\nPick {{choose: q}}.\n")) + assert not is_template(parse(_FRONTMATTER + "# A\n\nText.\n")) + assert is_template(parse(_FRONTMATTER + "# A\n\nPick {{choose: q}}.\n")) def test_is_drafting_note(): - [note, quote] = parse_document(_FRONTMATTER + "# A\n\n> [!drafting]\n> Check.\n\n> Plain.\n").sections[0].blocks + [note, quote] = parse(_FRONTMATTER + "# A\n\n> [!drafting]\n> Check.\n\n> Plain.\n").sections[0].blocks assert is_drafting_note(note) and not is_drafting_note(quote) @@ -127,7 +128,7 @@ def test_the_decisions_are_the_validators_own(path: Path): ``is_template``, the placed markers are the validator's placed findings, each where it says, at most one in a fragment.""" try: - document = parse_document(path.read_text(encoding="utf-8")) + document = parse(path.read_text(encoding="utf-8")) except legaldown.FrontmatterError: pytest.skip("unreadable frontmatter") result = validate_document(document) @@ -148,7 +149,7 @@ def test_the_decisions_are_the_validators_own(path: Path): def test_a_marker_after_a_lifted_reference_is_in_the_suffix(): [marker] = _markers("# A\n\nSee {{ref: a}} below. {#t}\n\n# B {#a}\n\nText.\n") - block = parse_document(_FRONTMATTER + "# A\n\nSee {{ref: a}} below. {#t}\n\n# B {#a}\n\nText.\n").sections[0].blocks[0] + block = parse(_FRONTMATTER + "# A\n\nSee {{ref: a}} below. {#t}\n\n# B {#a}\n\nText.\n").sections[0].blocks[0] assert (block.kind, marker.field) == ("ref", "suffix") assert block.suffix[marker.offset:] == "{#t}" @@ -175,7 +176,7 @@ def test_a_list_is_walked_once(monkeypatch): def test_a_document_changed_since_parsing_has_no_lines(): - document = parse_document(_FRONTMATTER + "# A\n\nText {#t}\n\n- a {#b}\n") + document = parse(_FRONTMATTER + "# A\n\nText {#t}\n\n- a {#b}\n") document.sections[0].blocks.insert(0, document.sections[0].blocks[0]) assert [m.line for m in validate_document(document).placed_markers] == [None, None, None] @@ -199,7 +200,7 @@ def test_two_lists_in_a_section_number_their_own_items(): def test_a_marker_after_a_lifted_term_is_in_the_suffix(): body = '"Thing" {{def: thing}} means a thing.\n\n# A\n\nUse the {{term: thing}} well. {#t}\n' [marker] = _markers(body) - block = parse_document(_FRONTMATTER + body).sections[0].blocks[0] + block = parse(_FRONTMATTER + body).sections[0].blocks[0] assert (block.kind, marker.field) == ("term", "suffix") assert block.suffix[marker.offset:] == "{#t}" @@ -212,8 +213,78 @@ def test_only_a_quote_block_is_a_drafting_note(): def test_fragments_have_names(): - [block] = parse_document(_FRONTMATTER + "# A\n\n- a {#a}\n - b\n").sections[0].blocks + [block] = parse(_FRONTMATTER + "# A\n\n- a {#a}\n - b\n").sections[0].blocks first = list_fragments(block)[0] assert (first.text, first.anchor, first.items) == ("a {#a}", True, (0,)) text, anchor = block_fragments(block)[1] assert (text, anchor) == ("b", True) + + +# load / parse ---------------------------------------------------------------- + +_SOURCE = _FRONTMATTER + "# A {#a}\n\nText.\n" + + +def test_load_reads_a_file_and_names_it(tmp_path): + file = tmp_path / "contract.lgd" + file.write_text(_SOURCE, encoding="utf-8") + document = load(file) + assert (document.filename, document.path) == ("contract.lgd", file.resolve()) + assert document.sections[0].title == "A" + + +def test_load_takes_a_string_path_and_a_relative_one(tmp_path, monkeypatch): + (tmp_path / "contract.lgd").write_text(_SOURCE, encoding="utf-8") + monkeypatch.chdir(tmp_path) + assert load("contract.lgd").path == (tmp_path / "contract.lgd").resolve() + assert load("contract.lgd").path.is_absolute() + + +@pytest.mark.parametrize("ending", ["\n", "\r\n", "\r"]) +def test_load_reads_the_same_whatever_the_line_endings(tmp_path, ending): + file = tmp_path / "x.lgd" + file.write_bytes(_SOURCE.replace("\n", ending).encode("utf-8")) + assert load(file) == parse(_SOURCE, filename="x.lgd") + + +def test_load_drops_a_byte_order_mark_and_keeps_utf8(tmp_path): + source = _FRONTMATTER + "# Čl. 1\n\nPlatba v €.\n" + file = tmp_path / "x.lgd" + file.write_bytes(b"\xef\xbb\xbf" + source.encode("utf-8")) + assert load(file) == parse(source, filename="x.lgd") + + +def test_load_has_the_source_map_of_a_parsed_document(tmp_path): + file = tmp_path / "x.lgd" + file.write_text(_FRONTMATTER + "# A\n\nSee {{ref: nowhere}}.\n", encoding="utf-8") + [diagnostic] = [d for d in validate_document(load(file)).diagnostics if d.rule == "ref-broken"] + assert (diagnostic.line, diagnostic.file) == (7, "x.lgd") + + +def test_load_fails_in_the_ways_the_documentation_says(tmp_path): + with pytest.raises(FileNotFoundError): + load(tmp_path / "nope.lgd") + bad = tmp_path / "bad.lgd" + bad.write_bytes(b"\xff\xfe") + with pytest.raises(UnicodeDecodeError): + load(bad) + bad.write_text("---\ntitle: [unclosed\n---\n", encoding="utf-8") + with pytest.raises(legaldown.FrontmatterError): + load(bad) + + +def test_where_a_document_lives_is_not_part_of_what_it_says(tmp_path): + file = tmp_path / "x.lgd" + file.write_text(_SOURCE, encoding="utf-8") + assert load(file).path is not None + assert parse(_SOURCE).path is None + assert load(file).path != parse(_SOURCE).path + assert load(file) == parse(_SOURCE, filename="x.lgd") + assert "path" not in legaldown.document_to_dict(load(file)) + + +def test_parse_document_is_a_deprecated_alias_of_parse(): + with pytest.warns(DeprecationWarning, match=r"removed in 0\.5\.0") as record: + document = legaldown.parse_document(_SOURCE, filename="x.lgd") + assert document == parse(_SOURCE, filename="x.lgd") + assert record[0].filename == __file__ # the caller is named, not the shim diff --git a/tests/test_spec_alignment.py b/tests/test_spec_alignment.py index 2f2c64b..f981b98 100644 --- a/tests/test_spec_alignment.py +++ b/tests/test_spec_alignment.py @@ -25,7 +25,7 @@ from legaldown.directives import format_value, lex from legaldown.markers import Marker, split_heading from legaldown.models import CustomField, empty_document, list_items -from legaldown.parser import collect_source_directives, parse_document +from legaldown.parser import collect_source_directives, parse from legaldown.validator import validate_document from legaldown.validator.helpers import generate_identifier @@ -65,7 +65,7 @@ def _shape(block: Block): def _validate(body: str): - return validate_document(parse_document(_FRONTMATTER + body, filename="t.lgd")) + return validate_document(parse(_FRONTMATTER + body, filename="t.lgd")) # ── Faithful parsing ────────────────────────────────────────────── @@ -103,7 +103,7 @@ def test_parser_does_not_silently_correct_party_metadata(): Text. """ - document = parse_document(source, filename="t.lgd") + document = parse(source, filename="t.lgd") side = document.metadata.sides[0] assert side.name == "Providing Party" # preserved verbatim assert side.parties[0].type == "company" # not mapped onto legal_entity @@ -133,7 +133,7 @@ def test_a_party_omitting_the_required_type_is_reported(): Text. """ - document = parse_document(source, filename="t.lgd") + document = parse(source, filename="t.lgd") assert document.metadata.sides[0].parties[0].type == "" assert "party-type-invalid" in validate_document(document).rules("error") @@ -257,21 +257,21 @@ def test_parameters_on_ref_and_def_are_reported(): result = _validate(source) assert [d.rule for d in result.diagnostics] == ["directive-unknown-param"] * 2 assert result.definition_lookup["foo"] == "Foo" - assert "{{ref: terms, format=long}}" in serialize_document(parse_document(_FRONTMATTER + source)) + assert "{{ref: terms, format=long}}" in serialize_document(parse(_FRONTMATTER + source)) def test_directive_in_code_span_is_not_lifted_or_checked(): """§11.4: the parser must not lift a code-span {{ref:}} into a ref block.""" - document = parse_document(_FRONTMATTER + "Write `{{ref: nope}}` or `{{term: nope}}`.") + document = parse(_FRONTMATTER + "Write `{{ref: nope}}` or `{{term: nope}}`.") assert document.sections[0].blocks[0].kind == "paragraph" assert validate_document(document).diagnostics == [] def test_lifted_term_label_round_trips(): source = _FRONTMATTER + '"Svc" {{def: svc}} x.\n\nSee {{term: svc, label="Services, as amended"}}.\n' - block = parse_document(source).sections[0].blocks[1] + block = parse(source).sections[0].blocks[1] assert (block.kind, block.target, block.label) == ("term", "svc", "Services, as amended") - assert '{{term: svc, label="Services, as amended"}}' in serialize_document(parse_document(source)) + assert '{{term: svc, label="Services, as amended"}}' in serialize_document(parse(source)) def test_placeholder_currency_is_defined_only_for_money(): @@ -285,7 +285,7 @@ def test_frontmatter_placeholder_with_unknown_parameter_is_checked(): source = _FRONTMATTER.replace( "title: Fixture", 'title: "{{placeholder: t, type=percentage, colour=red}}"' ) - result = validate_document(parse_document(source + "Text.\n", filename="t.lgd")) + result = validate_document(parse(source + "Text.\n", filename="t.lgd")) assert "placeholder-type-invalid" in result.rules("error") assert "directive-unknown-param" in result.rules("warning") @@ -323,9 +323,9 @@ def test_other_values_are_not_curly_quote_warnings(text): def test_a_curly_quoted_reference_stays_paragraph_text(): """A lifted {{ref:}} or {{term:}} is not lexed again, so a directive the validator warns about is not lifted.""" - document = parse_document(_FRONTMATTER + "See {{ref: “terms”}}.\n") + document = parse(_FRONTMATTER + "See {{ref: “terms”}}.\n") assert document.sections[0].blocks[0].kind == "paragraph" - assert parse_document(serialize_document(document)).sections == document.sections + assert parse(serialize_document(document)).sections == document.sections assert "value-curly-quote" in validate_document(document).rules("warning") @@ -334,15 +334,15 @@ def test_a_curly_quoted_reference_stays_paragraph_text(): ['"Fee" {{def: fee}} x.\n\nSee {{term: fee, label="“Curly”"}}.', 'See {{ref: "“terms”"}}.', '"Fee" {{def: "«fee»"}} x.'], ) def test_a_quoted_curly_value_stays_quoted_through_a_round_trip(text): - document = parse_document(_FRONTMATTER + text + "\n") - reparsed = parse_document(serialize_document(document)) + document = parse(_FRONTMATTER + text + "\n") + reparsed = parse(serialize_document(document)) assert reparsed.sections == document.sections assert "value-curly-quote" not in validate_document(reparsed).rules() def test_a_curly_quote_in_a_frontmatter_placeholder_is_a_warning(): source = _FRONTMATTER.replace("title: Fixture", "title: '{{placeholder: t, note=“x”}}'") - assert "value-curly-quote" in validate_document(parse_document(source + "Text.\n")).rules("warning") + assert "value-curly-quote" in validate_document(parse(source + "Text.\n")).rules("warning") def test_the_lexer_records_which_values_were_unquoted(): @@ -351,7 +351,7 @@ def test_the_lexer_records_which_values_were_unquoted(): def test_collect_source_directives_sees_every_parameter_shape(): - document = parse_document( + document = parse( _FRONTMATTER + "See {{ref: a, colour=red}} and {{term: b, label=\"x, y\"}}.\n" ) assert collect_source_directives(document) == ({"a"}, {"b"}) @@ -359,7 +359,7 @@ def test_collect_source_directives_sees_every_parameter_shape(): def test_explicitly_empty_label_round_trips(): source = _FRONTMATTER + '"X" {{def: x}} y.\n\nSee {{term: x, label=""}}.\n' - assert '{{term: x, label=""}}' in serialize_document(parse_document(source)) + assert '{{term: x, label=""}}' in serialize_document(parse(source)) def test_malformed_directive_with_unknown_name_is_malformed(): @@ -433,8 +433,8 @@ def test_quoted_value_may_contain_backticks(): def test_lifted_values_are_kept_as_written(): source = _FRONTMATTER + '"X" {{def: x}} y.\n\nSee {{term: x, label="a `b` c"}}.\n' - assert parse_document(source).sections[0].blocks[1].label == "a `b` c" - reparsed = parse_document(serialize_document(parse_document(source))) + assert parse(source).sections[0].blocks[1].label == "a `b` c" + reparsed = parse(serialize_document(parse(source))) assert reparsed.sections[0].blocks[1].label == "a `b` c" @@ -462,7 +462,7 @@ def test_malformed_directive_ends_at_the_next_opener(body): def test_definition_with_other_quotation_marks_round_trips(paragraph): """The serializer writes straight quotes, so other delimiters stay text.""" source = _FRONTMATTER + paragraph + "\n" - assert paragraph in serialize_document(parse_document(source)) + assert paragraph in serialize_document(parse(source)) def test_ref_inside_a_defined_term_is_not_lifted(): @@ -505,14 +505,14 @@ def test_unclosed_directive_does_not_swallow_the_next(): @pytest.mark.parametrize("paragraph", ['"" {{def: foo}} means x.', '" Foo " {{def: foo}} means x.']) def test_definition_the_block_cannot_hold_stays_text(paragraph): source = _FRONTMATTER + paragraph + "\n" - assert paragraph in serialize_document(parse_document(source)) + assert paragraph in serialize_document(parse(source)) def test_escaped_placeholder_in_metadata_is_not_a_placeholder(): source = _FRONTMATTER.replace( "title: Fixture", 'title: Fixture\neffective_date: "\\\\{{placeholder: d}} soon"' ) - result = validate_document(parse_document(source + "Text.\n")) + result = validate_document(parse(source + "Text.\n")) assert "metadata-date-invalid" in result.rules("error") @@ -630,7 +630,7 @@ def test_duration_unit_m_is_rejected_with_hint(): # Written directly in a directive the migration does not reach (already # canonical spelling context), the bare unit M is an error. result = validate_document( - parse_document( + parse( _FRONTMATTER + "Within {{duration: 5, unit=Q}}.", filename="t.lgd" ) ) @@ -649,7 +649,7 @@ def test_ref_to_attachment_id_suggests_attach(): "document_type: contract\nattachments:\n" " - id: schedule-a\n title: Schedule A\n file: schedule-a.pdf", ) - result = validate_document(parse_document(source + body, filename="t.lgd")) + result = validate_document(parse(source + body, filename="t.lgd")) assert "ref-targets-attachment" in result.rules("error") @@ -677,7 +677,7 @@ def _numbers(headings: str) -> list[str]: """The section numbers of a document with *headings*; one that opens below level 1 is also a heading-skip, numbered all the same.""" source = _FRONTMATTER.replace("# Terms {#terms}\n\n", "") + headings - return [entry.number for entry in validate_document(parse_document(source)).sections] + return [entry.number for entry in validate_document(parse(source)).sections] @pytest.mark.parametrize( @@ -697,7 +697,7 @@ def test_a_skipped_level_counts_as_its_first_so_numbers_stay_unique(headings, nu def test_a_fragment_starting_at_level_two_is_numbered_from_one(): """An attachment or include file has neither frontmatter nor a # heading (§12, §13.8).""" - result = validate_document(parse_document("## A\n\n## B\n\n### B1\n\n## C\n")) + result = validate_document(parse("## A\n\n## B\n\n### B1\n\n## C\n")) assert [entry.number for entry in result.sections] == ["1", "2", "2.1", "3"] assert "heading-skip" not in result.rules() @@ -709,7 +709,7 @@ def test_a_fragment_starting_at_level_two_is_numbered_from_one(): def test_a_main_document_opening_below_level_one_skips_a_level(body, level): """§4.1: a document with frontmatter is a main document, whose first heading is at level 1.""" - result = validate_document(parse_document(_BARE + body)) + result = validate_document(parse(_BARE + body)) assert [d.message for d in result.diagnostics if d.rule == "heading-skip"] == [ f"Heading levels must not skip. 'A' is at level {level}, but the document has no level-1 heading before it." ] @@ -725,7 +725,7 @@ def test_a_main_document_opening_below_level_one_skips_a_level(body, level): ], ) def test_where_a_first_heading_below_level_one_is_a_skip(source, skips): - assert ("heading-skip" in validate_document(parse_document(source)).rules()) is skips + assert ("heading-skip" in validate_document(parse(source)).rules()) is skips def test_numbers_are_unique_and_unchanged_without_a_skip(): @@ -756,7 +756,7 @@ def test_out_of_range_heading_keeps_its_index_entry(): """ body = "# One {#one}\n\nText.\n\n###### Deep {#deep}\n\nText.\n\n# Two {#two}\n\nText.\n" source = _FRONTMATTER.replace("# Terms {#terms}\n\n", "") + body - document = parse_document(source, filename="t.lgd") + document = parse(source, filename="t.lgd") result = validate_document(document) assert "heading-depth" in result.rules("error") assert len(result.sections) == len(document.sections) @@ -769,7 +769,7 @@ def test_amend_term_is_an_error_when_original_defines_nothing(): "document_type: contract", "document_type: contract\namends:\n title: Original\n file: original.lgd", ) + "Uses {{term: missing}}.\n" - document = parse_document(source, filename="amendment.lgd") + document = parse(source, filename="amendment.lgd") result = validate_document(document, import_definitions=lambda *_: {}) assert "amend-term-undefined" in result.rules("error") assert "amend-term-unresolvable" not in result.rules() @@ -819,7 +819,7 @@ def test_inline_value_indices_collect_field_spec_values(): def _validate_attachments(attachments: str, body: str = "Text."): source = _ATTACHMENT_FRONTMATTER.format(attachments=attachments, body=body) - return validate_document(parse_document(source, filename="t.lgd")) + return validate_document(parse(source, filename="t.lgd")) def test_duplicate_attachment_id_is_reported(): @@ -885,7 +885,7 @@ def test_amendment_redefining_an_imported_term_is_a_warning(): The {{term: services}} are amended. """ - document = parse_document(source, filename="first-amendment.lgd") + document = parse(source, filename="first-amendment.lgd") result = validate_document( document, import_definitions=lambda *_: {"services": "Services"} ) @@ -905,32 +905,32 @@ def test_amendment_redefining_an_imported_term_is_a_warning(): def test_preamble_is_kept_out_of_the_numbered_sections(): - document = parse_document(_PREAMBLE_SOURCE) + document = parse(_PREAMBLE_SOURCE) assert [s.identifier for s in document.sections] == ["confidentiality"] assert "entered into\nbetween" in document.preamble[0].text def test_definition_in_the_preamble_is_document_wide(): - result = validate_document(parse_document(_PREAMBLE_SOURCE)) + result = validate_document(parse(_PREAMBLE_SOURCE)) assert result.definition_lookup == {"agreement": "Agreement"} assert result.diagnostics == [] def test_directives_in_the_preamble_are_checked(): source = _PREAMBLE_SOURCE.replace("{{party: beta}}", "{{party: nobody}} on {{date: 2026-02-30}}") - assert {"party-unknown", "date-invalid"} <= validate_document(parse_document(source)).rules("error") + assert {"party-unknown", "date-invalid"} <= validate_document(parse(source)).rules("error") def test_preamble_round_trips(): - document = parse_document(_PREAMBLE_SOURCE) - reparsed = parse_document(serialize_document(document)) + document = parse(_PREAMBLE_SOURCE) + reparsed = parse(serialize_document(document)) assert reparsed.preamble == document.preamble assert reparsed.sections == document.sections def test_document_of_only_a_preamble(): source = _FRONTMATTER.replace("# Terms {#terms}\n\n", "") + "By {{party: nobody}}.\n" - document = parse_document(source) + document = parse(source) assert document.sections == [] and len(document.preamble) == 1 assert "party-unknown" in validate_document(document).rules("error") @@ -941,14 +941,14 @@ def test_preamble_paragraph_anchor_is_misplaced(marker): source = _PREAMBLE_SOURCE.replace("{{party: beta}}.", "{{party: beta}}. " + marker) + ( "See {{ref: intro}}.\n" ) - result = validate_document(parse_document(source)) + result = validate_document(parse(source)) assert "anchor-misplaced" in result.rules("warning") assert "ref-broken" in result.rules("error") assert "anchor-format" not in result.rules() def test_preamble_survives_the_dict_round_trip(): - document = parse_document(_PREAMBLE_SOURCE) + document = parse(_PREAMBLE_SOURCE) assert document_from_dict(document_to_dict(document)) == document ref = next(r for r in collect_definitions(document) if r.id == "agreement") assert ref.section_index is None @@ -1028,7 +1028,7 @@ def test_paragraph_anchor_colliding_with_an_attachment_id_is_reported(): def test_empty_frontmatter_and_byte_order_mark_are_not_body(source): """Empty frontmatter stops at its own closing ---, not at a later rule, and a byte-order mark never hides the first heading.""" - document = parse_document(source) + document = parse(source) assert document.preamble == [] assert document.sections[0].identifier == "a" @@ -1039,7 +1039,7 @@ def test_empty_frontmatter_and_byte_order_mark_are_not_body(source): def _presence(source: str) -> tuple[bool, list[str], list[str], set[str]]: """Whether the frontmatter is absent, the preamble's block kinds, the section titles, and the rules reported.""" - document = parse_document(source) + document = parse(source) return ( document.metadata.frontmatter_absent, [b.kind for b in document.preamble], @@ -1051,7 +1051,7 @@ def _presence(source: str) -> tuple[bool, list[str], list[str], set[str]]: def test_a_document_without_frontmatter_draws_only_the_warning(): """§16.6: not title-missing, not sides-absent.""" assert _presence("# Scope {#scope}\n\nBody.\n") == (True, [], ["Scope"], {"frontmatter-absent"}) - result = validate_document(parse_document("# Scope\n\nBody.\n")) + result = validate_document(parse("# Scope\n\nBody.\n")) assert [d.level for d in result.diagnostics] == ["warning"] @@ -1083,12 +1083,12 @@ def test_empty_frontmatter_is_present(block): def test_invalid_yaml_is_still_an_error(): with pytest.raises(yaml.YAMLError): - parse_document("---\ntitle: [unclosed\n---\n# A\n") + parse("---\ntitle: [unclosed\n---\n# A\n") def test_a_mapping_of_another_type_is_not_frontmatter_fields(): with pytest.raises(ValueError, match="mapping of fields"): - parse_document("---\n!!set {title: T}\n---\n# A\n") + parse("---\n!!set {title: T}\n---\n# A\n") @pytest.mark.parametrize("frontmatter", ["!!set {title: T}", "title: [x", "a: " + "{" * 3000]) @@ -1098,7 +1098,7 @@ def test_frontmatter_that_cannot_be_read_raises_frontmatter_error(frontmatter): from legaldown import FrontmatterError with pytest.raises(FrontmatterError) as raised: - parse_document(f"---\n{frontmatter}\n---\n# A\n") + parse(f"---\n{frontmatter}\n---\n# A\n") assert isinstance(raised.value, yaml.YAMLError) and isinstance(raised.value, ValueError) @@ -1112,23 +1112,23 @@ def test_frontmatter_that_cannot_be_read_raises_frontmatter_error(frontmatter): ], ) def test_a_document_without_frontmatter_is_written_without_it(source): - document = parse_document(source) + document = parse(source) written = serialize_document(document) assert not written.startswith("---") - assert parse_document(written) == document - assert validate_document(parse_document(written)).rules() <= {"frontmatter-absent"} + assert parse(written) == document + assert validate_document(parse(written)).rules() <= {"frontmatter-absent"} def test_metadata_set_on_a_bare_document_is_written_as_frontmatter(): - document = parse_document("# A\n\nText.\n") + document = parse("# A\n\nText.\n") document.metadata.title = "Terms" assert serialize_document(document).startswith("---\ntitle: Terms\n") def test_frontmatter_absent_is_not_read_from_frontmatter_or_a_dict(): - document = parse_document("---\ntitle: T\nfrontmatter_absent: true\n---\n\n# A\n") + document = parse("---\ntitle: T\nfrontmatter_absent: true\n---\n\n# A\n") assert document.metadata.frontmatter_absent is False - assert document_from_dict(document_to_dict(parse_document("# A\n"))).metadata.frontmatter_absent is False + assert document_from_dict(document_to_dict(parse("# A\n"))).metadata.frontmatter_absent is False assert "frontmatter-absent" not in validate_document(document_from_dict({"sections": []})).rules() @@ -1156,7 +1156,7 @@ def test_marker_after_a_malformed_directive_is_still_an_anchor(): def _outline(source: str) -> list[tuple[str, int, str]]: """Each heading's text, level, and identifier: explicit or generated.""" - entries = validate_document(parse_document(source)).sections + entries = validate_document(parse(source)).sections return [(entry.title, entry.level, entry.identifier) for entry in entries] @@ -1166,18 +1166,18 @@ def _outline(source: str) -> list[tuple[str, int, str]]: def test_only_spaces_and_tabs_and_nine_digits_make_a_heading_or_an_item(line): """CommonMark: other whitespace and longer or non-ASCII numbers are paragraph text (cmark-gfm).""" - document = parse_document(_FRONTMATTER + f"{line}\n") + document = parse(_FRONTMATTER + f"{line}\n") assert [(b.kind, b.text) for b in document.sections[0].blocks] == [ ("paragraph", line.strip(" \t\n\r")) ] - assert parse_document(serialize_document(document)) == document + assert parse(serialize_document(document)) == document @pytest.mark.parametrize(("line", "kind"), [ ("#\tTitle", None), ("123456789. item", "ordered_list"), ("-\titem", "unordered_list"), ]) def test_a_tab_or_nine_digits_still_make_a_heading_or_an_item(line, kind): - document = parse_document(_FRONTMATTER + f"{line}\n") + document = parse(_FRONTMATTER + f"{line}\n") if kind is None: assert [s.title for s in document.sections] == ["Terms", "Title"] else: @@ -1188,7 +1188,7 @@ def test_setext_headings_are_headings(): """§4.1: === and --- underlines make level-1 and level-2 headings.""" source = _BARE + "Intro.\n\nPayment\nTerms {#pay}\n=======\n\nText.\n\nLate Fees\n---\n\nMore.\n" assert _outline(source) == [("Payment Terms", 1, "pay"), ("Late Fees", 2, "late-fees")] - assert len(parse_document(source).preamble) == 1 + assert len(parse(source).preamble) == 1 @pytest.mark.parametrize("character", ["\x0c", "\x0b", "\x1c", "\x85", "\u2028"]) @@ -1200,7 +1200,7 @@ def test_only_lf_cr_and_crlf_end_a_line(character): def test_an_unclosed_fence_at_the_end_holds_no_extra_line(): - document = parse_document(_FRONTMATTER + "```\ncode\n") + document = parse(_FRONTMATTER + "```\ncode\n") assert document.sections[0].blocks[0].text == "```\ncode" @@ -1234,14 +1234,14 @@ def test_a_list_item_or_quote_that_is_one_fence_line_is_code(body): def test_a_nested_item_interrupts_its_items_paragraph_only_as_commonmark_allows(body, items): """Only a bullet with text, or an ordered item numbered 1, may interrupt a paragraph — the item's own too (cmark-gfm).""" - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert [(b.kind, _texts(b)) for b in document.sections[0].blocks] == items @pytest.mark.parametrize("number", ["1", "01", "001"]) def test_an_item_numbered_1_however_written_interrupts_a_paragraph(number): """#57: the number is compared as a number.""" - document = parse_document(_FRONTMATTER + f"Text\n{number}. item\n") + document = parse(_FRONTMATTER + f"Text\n{number}. item\n") assert [b.kind for b in document.sections[0].blocks] == ["paragraph", "ordered_list"] @@ -1255,8 +1255,8 @@ def test_an_anchor_after_a_heading_in_an_item_is_misplaced(): @pytest.mark.parametrize("text", [" See {{ref: x}}.", "See {{ref: x}} ", " Use {{term: x}}."]) def test_a_no_break_space_round_trips_around_a_lifted_directive(text): - document = parse_document(_FRONTMATTER + text + "\n") - assert parse_document(serialize_document(document)) == document + document = parse(_FRONTMATTER + text + "\n") + assert parse(serialize_document(document)) == document def test_a_paragraph_of_only_spaces_is_empty(): @@ -1268,7 +1268,7 @@ def test_a_paragraph_of_a_no_break_space_is_text(): # Not blank in CommonMark (#47): written back, it reads the same. document = document_from_dict({"sections": [{"title": "A", "blocks": [{"kind": "paragraph", "text": "\u00a0"}]}]}) assert document.sections[0].blocks[0].text == "\u00a0" - written = parse_document(serialize_document(document)) + written = parse(serialize_document(document)) assert [(b.kind, b.text) for b in written.sections[0].blocks] == [("paragraph", "\u00a0")] @@ -1296,11 +1296,11 @@ def test_a_marker_on_an_items_later_paragraph_is_misplaced(): ("- a\n\n b\n", ["unordered_list", "paragraph"], ["Terms"]), ]) def test_where_a_list_does_not_continue_past_a_blank_line(body, kinds, titles): - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert [b.kind for b in document.sections[0].blocks] == kinds assert [s.title for s in document.sections] == titles if "- \n" not in body: # an empty item is not written back (#46) - assert parse_document(serialize_document(document)) == document + assert parse(serialize_document(document)) == document @pytest.mark.parametrize(("body", "items", "after"), [ @@ -1312,7 +1312,7 @@ def test_where_a_list_does_not_continue_past_a_blank_line(body, kinds, titles): ("- a\n\n 2. b\n", ["a\n1. b"], []), ]) def test_what_an_items_later_content_is(body, items, after): - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) blocks = document.sections[0].blocks assert _texts(blocks[0]) == items and [b.kind for b in blocks[1:]] == after @@ -1323,10 +1323,10 @@ def test_what_an_items_later_content_is(body, items, after): ]) def test_a_line_short_of_every_open_items_content_is_code(body): """No item reaches it, so it is indented code (cmark-gfm).""" - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert [b.kind for b in document.sections[0].blocks] == ["ordered_list" if body[0] == "1" else "unordered_list", "code"] assert "ref-broken" not in _validate(body).rules() - assert parse_document(serialize_document(document)) == document + assert parse(serialize_document(document)) == document @pytest.mark.parametrize("body", [ @@ -1337,14 +1337,14 @@ def test_a_line_short_of_every_open_items_content_is_code(body): def test_a_block_after_a_wide_list_marker_round_trips(body): """The list is written back with its last item's content past the block's indentation, so the block stays after it.""" - document = parse_document(_FRONTMATTER + body) - assert parse_document(serialize_document(document)) == document + document = parse(_FRONTMATTER + body) + assert parse(serialize_document(document)) == document @pytest.mark.parametrize("text", ["
", "", ""]) def test_a_tag_is_split_only_where_both_lines_stay_paragraph_text(text): document = document_from_dict({"sections": [{"title": "A", "blocks": [{"kind": "paragraph", "text": text}]}]}) - blocks = parse_document(serialize_document(document)).sections[0].blocks + blocks = parse(serialize_document(document)).sections[0].blocks assert [b.kind for b in blocks] == ["paragraph"] @@ -1360,15 +1360,15 @@ def test_a_tag_is_split_only_where_both_lines_stay_paragraph_text(text): ("- a\n- \n\n code\n", ["unordered_list", "code"], ["Terms"]), ]) def test_what_follows_a_list_item_that_holds_no_open_paragraph(body, kinds, titles): - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert [b.kind for b in document.sections[0].blocks] == kinds assert [s.title for s in document.sections] == titles - assert parse_document(serialize_document(document)) == document + assert parse(serialize_document(document)) == document def test_wide_marker_spacing_sets_the_items_column_as_commonmark_does(): """Five spaces after the marker: the content column is one past it.""" - document = parse_document(_FRONTMATTER + "- foo\n\n bar\n") + document = parse(_FRONTMATTER + "- foo\n\n bar\n") # The rest of the line is indented code, and so is `bar`: " foo" and # "bar" (cmark-gfm). [block] = document.sections[0].blocks @@ -1380,12 +1380,12 @@ def test_an_unclosed_fence_after_a_list_is_closed_when_written(): {"title": "A", "blocks": [{"kind": "unordered_list", "items": ["a"]}, {"kind": "code", "text": "```\nx"}]}, {"title": "B", "blocks": []}, ]}) - assert [s.title for s in parse_document(serialize_document(document)).sections] == ["A", "B"] + assert [s.title for s in parse(serialize_document(document)).sections] == ["A", "B"] def test_an_empty_list_does_not_take_the_code_after_it(): - document = parse_document(_FRONTMATTER + "- a\n\n* \n\n code\n") - reparsed = parse_document(serialize_document(document)) + document = parse(_FRONTMATTER + "- a\n\n* \n\n code\n") + reparsed = parse(serialize_document(document)) assert [(b.kind, _texts(b), b.text) for b in reparsed.sections[0].blocks] == [ ("unordered_list", ["a"], ""), ("unordered_list", [""], ""), ("code", [], " code") ] @@ -1393,7 +1393,7 @@ def test_an_empty_list_does_not_take_the_code_after_it(): def _placeholders(body: str) -> int: - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) return sum( 1 for _s, _i, block in document.iter_blocks() for fragment in text_fragments(block) for directive in lex(fragment).directives if directive.name == "placeholder" and not directive.malformed @@ -1425,8 +1425,8 @@ def test_an_items_later_content_is_read_as_commonmark_reads_it(body, count): def test_an_items_later_content_round_trips(body): """A `#` line in it is written in far enough not to read as a heading; a fence left open in it ends with the item and is not closed.""" - document = parse_document(_FRONTMATTER + body) - assert parse_document(serialize_document(document)) == document + document = parse(_FRONTMATTER + body) + assert parse(serialize_document(document)) == document def test_a_table_cell_opening_like_a_fence_is_text(): @@ -1434,13 +1434,13 @@ def test_a_table_cell_opening_like_a_fence_is_text(): def test_frontmatter_with_cr_line_endings_is_frontmatter(): - document = parse_document("---\rtitle: X\r---\r\r# A\r") + document = parse("---\rtitle: X\r---\r\r# A\r") assert not document.metadata.frontmatter_absent assert document.metadata.title == "X" def test_a_quote_round_trips_a_character_that_is_not_a_line_ending(): - document = parse_document(_FRONTMATTER + "> One\x0ctwo.\n> Three.\n") + document = parse(_FRONTMATTER + "> One\x0ctwo.\n> Three.\n") assert serialize_document(document).endswith("> One\x0ctwo.\n> Three.\n") @@ -1449,7 +1449,7 @@ def test_a_quote_round_trips_a_character_that_is_not_a_line_ending(): ["Text.\n\n---\n\nMore.\n", "- item\n---\n", "> quoted\n---\n", "| a | b |\n|---|---|\n| c | d |\n---\n"], ) def test_dashes_not_under_a_paragraph_are_a_rule(body): - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert _outline(_FRONTMATTER + body) == [("Terms", 1, "terms")] assert "rule" in [b.kind for b in document.sections[0].blocks] @@ -1462,7 +1462,7 @@ def test_a_signature_block_heading_is_an_ordinary_section(hashes): f"{_BARE}# A\n\nSee {{{{ref: signature-block}}}}.\n\n" f"{hashes} Signature Block {{#signature-block}}\n\nSigned by the parties. {{{{ref: nowhere}}}}\n\n# After\n\nText.\n" ) - document = parse_document(source) + document = parse(source) assert [s.title for s in document.sections] == ["A", "Signature Block", "After"] result = validate_document(document) assert [d.message for d in result.diagnostics if d.rule == "ref-broken"] == [ @@ -1470,18 +1470,18 @@ def test_a_signature_block_heading_is_an_ordinary_section(hashes): ] serialized = serialize_document(document) assert "Signed by the parties." in serialized - assert parse_document(serialized) == document + assert parse(serialized) == document def test_a_second_signature_block_identifier_is_a_duplicate(): source = _BARE + "# Signature Block {#signature-block}\n\nA.\n\n# Signature Block {#signature-block}\n\nB.\n" - assert "anchor-duplicate" in validate_document(parse_document(source)).rules() + assert "anchor-duplicate" in validate_document(parse(source)).rules() def test_setext_heading_round_trips_as_an_atx_heading(): """Its generated identifier is not written: it stays generated.""" source = _BARE + "Scope\n=====\n\nText.\n" - assert "\n# Scope\n" in serialize_document(parse_document(source)) + assert "\n# Scope\n" in serialize_document(parse(source)) def test_fenced_code_is_one_literal_block(): @@ -1489,7 +1489,7 @@ def test_fenced_code_is_one_literal_block(): blank line inside it does not end it.""" fence = "```\n# not a heading\n\n{#z} {{ref: nope}} {{\n```" source = _FRONTMATTER + fence + "\n\n# Next {#next}\n" - document = parse_document(source) + document = parse(source) assert _outline(source) == [("Terms", 1, "terms"), ("Next", 1, "next")] assert [(b.kind, b.text) for b in document.sections[0].blocks] == [("code", fence)] assert validate_document(document).diagnostics == [] @@ -1516,13 +1516,13 @@ def test_inline_triple_backticks_do_not_open_a_fence(): def test_fence_interrupts_a_paragraph(): - document = parse_document(_FRONTMATTER + "Example:\n```\n{{ref: nope}}\n```\n") + document = parse(_FRONTMATTER + "Example:\n```\n{{ref: nope}}\n```\n") assert [b.kind for b in document.sections[0].blocks] == ["paragraph", "code"] assert validate_document(document).diagnostics == [] def _kinds(source: str) -> list[str]: - return [b.kind for b in parse_document(source).sections[0].blocks] + return [b.kind for b in parse(source).sections[0].blocks] @pytest.mark.parametrize( @@ -1549,7 +1549,7 @@ def test_dashes_after_a_lazy_continuation_are_a_rule(body): def test_fence_inside_a_list_item_does_not_swallow_the_section(): body = "- Example:\n ```\n code\n\n code2\n ```\n\nSee {{ref: nope}}.\n" - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert [b.kind for b in document.sections[0].blocks] == ["unordered_list", "ref"] assert "ref-broken" in validate_document(document).rules("error") @@ -1563,7 +1563,7 @@ def test_tab_indented_backticks_neither_open_nor_close_a_fence(): def test_indented_fence_round_trips_unchanged(): fence = " ```\n code\n ```" - assert fence in serialize_document(parse_document(_FRONTMATTER + "Para.\n\n" + fence + "\n")) + assert fence in serialize_document(parse(_FRONTMATTER + "Para.\n\n" + fence + "\n")) def test_paragraph_directly_above_dashes_is_a_setext_heading(): @@ -1576,7 +1576,7 @@ def test_paragraph_directly_above_dashes_is_a_setext_heading(): def test_fence_opening_on_a_list_marker_line_stays_in_the_item(): body = "- ```\n a\n\n b\n ```\n\n# Next {#next}\n\nSee {{ref: nope}}.\n" - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert _outline(_FRONTMATTER + body) == [("Terms", 1, "terms"), ("Next", 1, "next")] assert _texts(document.sections[0].blocks[0]) == ["```\na\n\nb\n```"] assert "ref-broken" in validate_document(document).rules("error") @@ -1584,7 +1584,7 @@ def test_fence_opening_on_a_list_marker_line_stays_in_the_item(): def test_fence_in_a_list_item_is_literal_and_round_trips(): body = "- Example:\n ~~~\n {{ref: nope}}\n\n {#zz}\n ~~~\n" - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert validate_document(document).diagnostics == [] assert body.strip() in serialize_document(document) @@ -1601,7 +1601,7 @@ def test_indented_dashes_are_code(): def test_a_single_pipe_line_is_a_paragraph_not_dropped(): - document = parse_document(_FRONTMATTER + "| lone {{ref: nope}}\n") + document = parse(_FRONTMATTER + "| lone {{ref: nope}}\n") assert _kinds(_FRONTMATTER + "| lone {{ref: nope}}\n") == ["ref"] assert "ref-broken" in validate_document(document).rules("error") @@ -1612,7 +1612,7 @@ def test_fence_in_a_block_quote_is_literal(): def test_code_block_without_a_fence_is_checked_as_text(): """Only a fenced block is literal; a hand-built one is not taken on trust.""" - document = parse_document(_FRONTMATTER + "Text.\n") + document = parse(_FRONTMATTER + "Text.\n") document.sections[0].blocks.append(Block(kind="code", text="See {{ref: nope}}.")) assert "ref-broken" in validate_document(document).rules("error") @@ -1634,21 +1634,21 @@ def test_list_or_quote_interrupts_a_paragraph(body, kinds): def test_item_text_after_its_closed_fence_is_checked(): - document = parse_document(_FRONTMATTER + "- ```\n code\n ```\n more {{ref: nowhere}}\n") + document = parse(_FRONTMATTER + "- ```\n code\n ```\n more {{ref: nowhere}}\n") assert _texts(document.sections[0].blocks[0]) == ["```\ncode\n```\nmore {{ref: nowhere}}"] assert "ref-broken" in validate_document(document).rules("error") - assert parse_document(serialize_document(document)).sections == document.sections + assert parse(serialize_document(document)).sections == document.sections def test_unclosed_fence_in_an_item_ends_at_the_next_item(): - document = parse_document(_FRONTMATTER + "1. ```\n x\n2. b\n3. c\n") + document = parse(_FRONTMATTER + "1. ```\n x\n2. b\n3. c\n") assert [(b.kind, _texts(b)) for b in document.sections[0].blocks] == [ ("ordered_list", ["```\nx", "b", "c"]) ] def test_tab_after_a_list_marker_counts_as_columns(): - document = parse_document(_FRONTMATTER + "-\t```\n\tx\n\t```\n") + document = parse(_FRONTMATTER + "-\t```\n\tx\n\t```\n") assert _texts(document.sections[0].blocks[0]) == ["```\nx\n```"] @@ -1665,23 +1665,23 @@ def test_rescan_after_a_directive_starts_at_its_line(): def test_code_block_text_after_its_closing_fence_is_checked(): - document = parse_document(_FRONTMATTER + "Text.\n") + document = parse(_FRONTMATTER + "Text.\n") document.sections[0].blocks.append(Block(kind="code", text="```\nx\n```\n{{ref: missing}}")) assert "ref-broken" in validate_document(document).rules("error") def test_serializer_closes_an_unclosed_fence(): - document = parse_document(_FRONTMATTER + "Text.\n\n# Next {#next}\n") + document = parse(_FRONTMATTER + "Text.\n\n# Next {#next}\n") document.sections[0].blocks.append(Block(kind="code", text="```\nx")) - reparsed = parse_document(serialize_document(document)) + reparsed = parse(serialize_document(document)) assert [s.identifier for s in reparsed.sections] == ["terms", "next"] def test_code_in_a_list_item_keeps_trailing_spaces(): source = _FRONTMATTER + "- ```\n keep \n ```\n" - item = parse_document(source).sections[0].blocks[0].items[0] + item = parse(source).sections[0].blocks[0].items[0] assert [(block.kind, block.text) for block in item.blocks] == [("code", "```\nkeep \n```")] - assert parse_document(serialize_document(parse_document(source))).sections[0].blocks[0].items[0] == item + assert parse(serialize_document(parse(source))).sections[0].blocks[0].items[0] == item @pytest.mark.parametrize( @@ -1699,31 +1699,31 @@ def test_backticks_in_directive_values_open_nothing(text, targets): @pytest.mark.parametrize("paragraph", [" ~~~\n x", " # not a heading"]) def test_paragraph_that_would_open_a_block_stays_indented(paragraph): source = _FRONTMATTER + paragraph + "\n\n# Next {#next}\n\nBody.\n" - reparsed = parse_document(serialize_document(parse_document(source))) - assert reparsed.sections == parse_document(source).sections + reparsed = parse(serialize_document(parse(source))) + assert reparsed.sections == parse(source).sections def test_tabs_inside_list_item_code_are_kept(): - document = parse_document(_FRONTMATTER + "- item\n ```\n a\tb\n ```\n") + document = parse(_FRONTMATTER + "- item\n ```\n a\tb\n ```\n") assert _texts(document.sections[0].blocks[0]) == ["item\n```\na\tb\n```"] def test_a_fence_left_open_in_a_list_item_ends_with_its_list(): # The next list takes another bullet: its marker ends the fence and the # list, so the fence is written as it is. - document = parse_document(_FRONTMATTER + "Text.\n") + document = parse(_FRONTMATTER + "Text.\n") document.sections[0].blocks += [ Block(kind="unordered_list", items=["x\n```\ncode"]), Block(kind="unordered_list", items=["y"]), ] - blocks = parse_document(serialize_document(document)).sections[0].blocks + blocks = parse(serialize_document(document)).sections[0].blocks assert [_texts(b) for b in blocks[1:]] == [["x\n```\ncode"], ["y"]] def test_indentation_after_the_quote_marker_is_content(): """Only one space after > is syntax; a fence line indented four columns inside quoted code is content, not a closing fence.""" - document = parse_document(_FRONTMATTER + "> ```\n> ```\n> {{ref: missing}}\n> ```\n") + document = parse(_FRONTMATTER + "> ```\n> ```\n> {{ref: missing}}\n> ```\n") assert validate_document(document).diagnostics == [] @@ -1745,8 +1745,8 @@ def test_only_some_list_items_interrupt_a_paragraph(second_line, kinds, headings def _blocks(body: str) -> list[tuple[str, str | list[str]]]: - document = parse_document(_FRONTMATTER + body) - assert parse_document(serialize_document(document)).sections == document.sections + document = parse(_FRONTMATTER + body) + assert parse(serialize_document(document)).sections == document.sections return [(b.kind, _texts(b) if b.kind.endswith("list") else b.text) for b in document.sections[0].blocks] @@ -1776,7 +1776,7 @@ def test_an_item_beginning_with_dashes_is_not_written_as_a_rule(body): def test_a_nested_item_of_another_type_stays_in_the_list(): - [block] = parse_document(_FRONTMATTER + "- parent\n * child\n 1) child\n- next\n").sections[0].blocks + [block] = parse(_FRONTMATTER + "- parent\n * child\n 1) child\n- next\n").sections[0].blocks assert _shape(block) == ("unordered_list", [ ["parent", ("unordered_list", [["child"]]), ("ordered_list", [["child"]])], ["next"] ]) @@ -1822,13 +1822,13 @@ def test_a_drafting_note_in_a_star_item_is_recognized(): def _table(body: str) -> Block: - [block] = parse_document(_FRONTMATTER + body).sections[0].blocks + [block] = parse(_FRONTMATTER + body).sections[0].blocks return block def _round_trips(body: str) -> None: - document = parse_document(_FRONTMATTER + body) - assert parse_document(serialize_document(document)).sections == document.sections + document = parse(_FRONTMATTER + body) + assert parse(serialize_document(document)).sections == document.sections def test_an_escaped_pipe_is_cell_text_and_alignment_is_kept(): @@ -1838,7 +1838,7 @@ def test_an_escaped_pipe_is_cell_text_and_alignment_is_kept(): assert block.align == ["left", "right", "center", ""] assert block.rows == [["1", "2", "3", "4"]] assert "| a \\| b | c | d | e |\n| :--- | ---: | :---: | --- |" in serialize_document( - parse_document(_FRONTMATTER + body) + parse(_FRONTMATTER + body) ) _round_trips(body) @@ -1899,7 +1899,7 @@ def test_table_cells_are_positional_in_the_model(): ]}]} ).sections[0].blocks[0] assert (block.headers, block.rows, block.align) == (["", "b"], [["", ""], ["1", ""], ["1", "2"]], ["right", ""]) - assert document_from_dict(document_to_dict(parse_document( + assert document_from_dict(document_to_dict(parse( _FRONTMATTER + "| a | b |\n|:--|--:|\n" ))).sections[0].blocks[0].align == ["left", "right"] assert document_from_dict({"sections": [{"blocks": [{"kind": "table"}]}]}).sections[0].blocks[0].rows == [["", ""]] @@ -1921,10 +1921,10 @@ def written(block: Block) -> list[str]: assert written({"kind": "table", "headers": [], "rows": [[]]}) == ["| |", "| --- |", "| |"] block = {"kind": "table", "headers": ["a\\|b", "c\\\\|"], "rows": [["\\", "|"]]} document = document_from_dict({"sections": [{"title": "A", "blocks": [block]}]}) - assert parse_document(serialize_document(document)).sections == document.sections + assert parse(serialize_document(document)).sections == document.sections # Nothing to take a width from: one empty column, still a table. document = document_from_dict({"sections": [{"title": "A", "blocks": [{"kind": "table", "headers": [], "rows": [[]]}]}]}) - assert parse_document(serialize_document(document)).sections[0].blocks[0].kind == "table" + assert parse(serialize_document(document)).sections[0].blocks[0].kind == "table" def test_default_headers_widen_to_the_widest_row(): @@ -1939,7 +1939,7 @@ def test_default_headers_widen_to_the_widest_row(): ) def test_a_table_interrupts_a_paragraph(body, kinds): """GFM: a paragraph's last line and a delimiter row under it start a table.""" - document = parse_document(_FRONTMATTER + body) + document = parse(_FRONTMATTER + body) assert [b.kind for b in document.sections[0].blocks] == kinds _round_trips(body) @@ -1950,8 +1950,8 @@ def test_a_table_interrupts_a_paragraph(body, kinds): def _html(body: str, *, bare: bool = False) -> tuple[list[str], list[tuple[str, str]]]: """Section titles, and the (kind, text) of every block, preamble first.""" source = (_BARE if bare else _FRONTMATTER) + body - document = parse_document(source) - assert parse_document(serialize_document(document)) == document + document = parse(source) + assert parse(serialize_document(document)) == document blocks = [(b.kind, b.text) for _section, _index, b in document.iter_blocks()] return [s.title for s in document.sections], blocks @@ -1994,14 +1994,14 @@ def test_a_whole_line_after_a_comment_is_raw_html(): """CommonMark: the line that ends a comment block belongs to it, so its text is not rendered and its directives are not recognized (§11.4).""" source = _FRONTMATTER + " The Buyer pays {{money: 5}} under {{ref: nope}}.\n" - result = validate_document(parse_document(source)) + result = validate_document(parse(source)) # Raw HTML, not a comment: rendered nowhere, and warned about (§8.7). assert result.rules() == {"raw-html"} and result.inline_money == [] def test_nothing_in_an_html_block_is_validated(): source = _FRONTMATTER + '
{{ref: nope}} {{bogus: x}} {{ "Fee" {{def: fee}}\n{#anchor}
\n\nSee {{ref: anchor}}.\n' - assert validate_document(parse_document(source)).rules() == {"ref-broken", "raw-html"} + assert validate_document(parse(source)).rules() == {"ref-broken", "raw-html"} @pytest.mark.parametrize( @@ -2037,7 +2037,7 @@ def test_an_html_block_in_the_model_keeps_its_indentation(): {"sections": [{"title": "A", "blocks": [{"kind": "html", "text": "\n\n
\n x\n
\n\n"}]}]} ) assert document.sections[0].blocks[0].text == "
\n x\n
" - assert parse_document(serialize_document(document)).sections == document.sections + assert parse(serialize_document(document)).sections == document.sections @pytest.mark.parametrize("prefix", ["
see", "# see", "```"]) @@ -2050,7 +2050,7 @@ def test_a_model_built_reference_that_would_open_a_block_is_escaped(prefix): {"kind": "paragraph", "text": f"{prefix} text"}, ]}]} ) - reparsed = parse_document(serialize_document(document)).sections[0].blocks + reparsed = parse(serialize_document(document)).sections[0].blocks assert [(b.kind, b.prefix or b.text) for b in reparsed] == [("ref", f"\\{prefix} "), ("paragraph", f"\\{prefix} text")] @@ -2058,14 +2058,14 @@ def test_a_model_built_reference_that_would_open_a_block_is_escaped(prefix): def _code_blocks(body: str) -> list[tuple[str, str]]: - document = parse_document(_FRONTMATTER + body) - assert parse_document(serialize_document(document)) == document + document = parse(_FRONTMATTER + body) + assert parse(serialize_document(document)) == document return [(b.kind, b.text) for b in document.sections[0].blocks] def test_directives_in_indented_code_are_literal(): body = "Text.\n\n {{ref: nope}} {#anchor} \"Fee\" {{def: fee}}\n {{bogus: x}}\n\nSee {{ref: anchor}}.\n" - result = validate_document(parse_document(_FRONTMATTER + body)) + result = validate_document(parse(_FRONTMATTER + body)) assert result.rules() == {"ref-broken"} assert [d.message for d in result.diagnostics] == ["Broken section reference: 'anchor'."] @@ -2096,14 +2096,14 @@ def test_where_indented_code_starts(body, kinds): def test_a_heading_indented_up_to_three_spaces_interrupts_a_paragraph(): - document = parse_document(_FRONTMATTER + "Text.\n ## Sub\n") + document = parse(_FRONTMATTER + "Text.\n ## Sub\n") assert [s.title for s in document.sections] == ["Terms", "Sub"] def test_a_document_can_open_with_indented_code(): - document = parse_document(" code\n\n# A\n") + document = parse(" code\n\n# A\n") assert [(b.kind, b.text) for b in document.preamble] == [("code", " code")] - assert parse_document(serialize_document(document)) == document + assert parse(serialize_document(document)) == document def test_indented_code_after_a_list_is_written_as_it_is(): @@ -2115,7 +2115,7 @@ def test_indented_code_after_a_list_is_written_as_it_is(): ]}]}) written = serialize_document(document) assert "- one\n\n x = `y`\n\n z" in written - assert parse_document(written) == document + assert parse(written) == document def test_every_indented_paragraph_after_a_list_is_the_items(): @@ -2124,7 +2124,7 @@ def test_every_indented_paragraph_after_a_list_is_the_items(): unindented paragraph ends the item (and the list).""" body = "1. Clause.\n\n Second.\n\n Pay {{placeholder: fee}}.\n\nThird.\n\n code {{ref: nope}}\n" assert [kind for kind, _text in _code_blocks(body)] == ["ordered_list", "paragraph", "code"] - result = validate_document(parse_document(_FRONTMATTER + body), final=True) + result = validate_document(parse(_FRONTMATTER + body), final=True) assert "placeholder-unfilled" in result.rules("error") and "ref-broken" not in result.rules() @@ -2134,8 +2134,8 @@ def test_an_indented_line_after_a_list_round_trips_whatever_it_begins_with(opene content: a block of the item's own.""" closer = f"\n {opener}" if opener in ("```", "~~~") else "" body = f"- a\n\n b\n\n {opener}{closer}\n\nd\n\n code\n" - document = parse_document(_FRONTMATTER + body) - assert parse_document(serialize_document(document)) == document + document = parse(_FRONTMATTER + body) + assert parse(serialize_document(document)) == document last = { "# foo": ("heading", "foo"), "
": ("html", "
"), "```": ("code", " ```\n ```"), "~~~": ("code", " ~~~\n ~~~"), "": ("html", " "), @@ -2163,7 +2163,7 @@ def test_a_table_with_an_indented_header_interrupts_a_paragraph(body, kinds): def test_an_indented_header_row_after_another_block_is_code(before): """Only an open paragraph can be interrupted by a table whose header row is indented; elsewhere that row is indented code (CommonMark).""" - document = parse_document(f"{_BARE}{before}\n | {{{{ref: nope}}}} |\n|---|\n") + document = parse(f"{_BARE}{before}\n | {{{{ref: nope}}}} |\n|---|\n") assert "table" not in [b.kind for _s, _i, b in document.iter_blocks()] assert "ref-broken" not in validate_document(document).rules() @@ -2174,8 +2174,8 @@ def test_code_directly_after_a_list_is_written_so_it_stays_code(): (CommonMark), so the code after it is ordinary indented code, not the item's; only code with no paragraph between it and the list needs fencing to stay code rather than become the item's own text.""" - document = parse_document(_FRONTMATTER + "- a\n\n\n\n code {{ref: nope}}\n") - reparsed = parse_document(serialize_document(document)) + document = parse(_FRONTMATTER + "- a\n\n\n\n code {{ref: nope}}\n") + reparsed = parse(serialize_document(document)) assert [(b.kind, b.text) for b in reparsed.sections[0].blocks] == [ ("unordered_list", ""), ("paragraph", ""), ("code", " code {{ref: nope}}") ] @@ -2183,12 +2183,12 @@ def test_code_directly_after_a_list_is_written_so_it_stays_code(): document = document_from_dict({"sections": [{"title": "A", "blocks": [ {"kind": "unordered_list", "items": ["a"]}, {"kind": "code", "text": " x"}, ]}]}) - blocks = parse_document(serialize_document(document)).sections[0].blocks + blocks = parse(serialize_document(document)).sections[0].blocks assert [(b.kind, b.text) for b in blocks] == [("unordered_list", ""), ("code", " x")] document = document_from_dict({"sections": [{"title": "A", "blocks": [ {"kind": "unordered_list", "items": ["a"]}, {"kind": "paragraph", "text": "# foo"}, {"kind": "code", "text": " x"}, ]}]}) - blocks = parse_document(serialize_document(document)).sections[0].blocks + blocks = parse(serialize_document(document)).sections[0].blocks assert [(b.kind, b.text) for b in blocks] == [("unordered_list", ""), ("paragraph", "\\# foo"), ("code", " x")] @@ -2197,7 +2197,7 @@ def test_text_after_a_list_that_would_open_a_block_is_escaped(): {"kind": "unordered_list", "items": ["a"]}, {"kind": "paragraph", "text": "> q"}, {"kind": "paragraph", "text": "# foo"}, ]}]}) - blocks = parse_document(serialize_document(document)).sections[0].blocks + blocks = parse(serialize_document(document)).sections[0].blocks assert (blocks[-1].kind, blocks[-1].text) == ("paragraph", "\\# foo") @@ -2217,7 +2217,7 @@ def test_text_after_a_list_that_would_open_a_block_is_escaped(): ) def test_table_rows_are_indented_at_most_three_columns_and_have_cells(body, kinds): assert [kind for kind, _text in _code_blocks(body)] == kinds - assert "ref-broken" not in validate_document(parse_document(_FRONTMATTER + body)).rules() + assert "ref-broken" not in validate_document(parse(_FRONTMATTER + body)).rules() @pytest.mark.parametrize("tag", ["
", "