From 2cb2963ee149c62725ed8f31d6d12c22896d87db Mon Sep 17 00:00:00 2001 From: Daniel Ellison Date: Thu, 6 Aug 2026 09:31:30 -0400 Subject: [PATCH] Verify Morris source reconstruction --- .github/workflows/validate.yml | 5 + README.md | 3 +- documents/reference/translation_process.md | 8 + project/README.md | 1 + project/development_log.md | 1 + project/near_term_development_plan.md | 10 +- project/source_reconstruction_manifest.json | 68 ++++ scripts/README.md | 11 + scripts/source_reconstruction.py | 374 ++++++++++++++++++++ scripts/test_source_reconstruction.py | 154 ++++++++ 10 files changed, 630 insertions(+), 5 deletions(-) create mode 100644 project/source_reconstruction_manifest.json create mode 100644 scripts/source_reconstruction.py create mode 100644 scripts/test_source_reconstruction.py diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml index 670a9f31..56f37317 100644 --- a/.github/workflows/validate.yml +++ b/.github/workflows/validate.yml @@ -43,6 +43,11 @@ jobs: python3 scripts/translation_process_status.py --check python3 scripts/test_translation_process_status.py + - name: Verify source reconstruction + run: | + python3 scripts/source_reconstruction.py + python3 scripts/test_source_reconstruction.py + - name: Check phonetic-neighbor baseline is in sync run: | python3 scripts/audit_phonetic_neighbors.py --output /tmp/phonetic_neighbors_baseline.txt diff --git a/README.md b/README.md index 2948cab3..ee75c261 100644 --- a/README.md +++ b/README.md @@ -46,9 +46,10 @@ python3 scripts/validate_examples.py python3 scripts/validate_sentences.py python3 scripts/test_sentence_validator.py python3 scripts/translation_process_status.py --check +python3 scripts/source_reconstruction.py ``` -Install the pinned dependency once for each Python environment. The lexical validator checks every entry against the [executable schema](vocabulary/schema.json) before applying Phi's sound, layout, and corpus rules. It also checks lexical use across active documents and compares supported citations with their stored sources. The sentence parser is a separate gate for complete examples in the lexicon and maintained teaching corpus. Machines can reject a particle in the wrong place; they cannot decide whether a translation has understood Morris. Literary translations therefore use the [isolated process](documents/reference/translation_process.md) and its generated [certification register](documents/evaluation/translation_process_status.md). +Install the pinned dependency once for each Python environment. The lexical validator checks every entry against the [executable schema](vocabulary/schema.json) before applying Phi's sound, layout, and corpus rules. It also checks lexical use across active documents. The sentence parser is a separate gate for complete examples in the lexicon and maintained teaching corpus. For the Morris chapters, the source-reconstruction checker selects each chapter from the stored novel and requires the ordered citations to reproduce it exactly. Machines can reject a particle in the wrong place or find a missing source phrase; they cannot decide whether a translation has understood Morris. Literary translations therefore use the [isolated process](documents/reference/translation_process.md) and its generated [certification register](documents/evaluation/translation_process_status.md). The [CI workflow](.github/workflows/validate.yml) contains the complete recipe. If you change the vocabulary, regenerate the derived reference with `python3 scripts/generate_reference.py`. Settled language decisions live in [canon.md](canon.md); the current sequence lives in the [near-term development plan](project/near_term_development_plan.md); broader work and evidence gates live in the [status roadmap](project/roadmap.md). The [development protocol](project/development_protocol.md) governs language work, while the [publishing strategy](project/publishing.md) holds the longer view. diff --git a/documents/reference/translation_process.md b/documents/reference/translation_process.md index 4134e7f2..76cc2a19 100644 --- a/documents/reference/translation_process.md +++ b/documents/reference/translation_process.md @@ -58,6 +58,14 @@ python3 scripts/translation_layers.py texts/north_wind_and_sun.md --phase source That view contains numbered Phi units and their decoded source citations. It omits glosses, derived English, notes, and limits. A new translation begins directly in the same numbered Phi-and-source form under `/tmp`; there is no English layer to extract. Before the freeze is accepted, the source citations must still reconstruct the chosen source passage exactly. +Where the repository holds an independent witness and a source manifest, check that reconstruction mechanically: + +```bash +python3 scripts/source_reconstruction.py +``` + +The manifest names the witness, citation label, selected boundaries, and normalization rule for each translation it covers. For the Project Gutenberg Morris source, normalization rejoins a hyphenated word split by a line wrap and collapses the remaining whitespace. Spelling, punctuation, and case stay exact. The checker then compares the complete selected passage with the citations in order; matching character counts alone cannot make it pass. + ## 4. Make compact anonymous bundles The frozen source-to-Phi view becomes the input to the second phase. For a work of ordinary length, make a directory of bounded packets: diff --git a/project/README.md b/project/README.md index ab200352..063b1c7f 100644 --- a/project/README.md +++ b/project/README.md @@ -11,6 +11,7 @@ This directory records how Phi is maintained. The index below gathers its workin | [Content vocabulary coverage](content_vocabulary_coverage.md) | Status-tracked semantic coverage audits and the review gate for content vocabulary batches. | | [Content vocabulary decisions](content_vocabulary_decisions.md) | Generated, readable view of every candidate carried forward from the content-vocabulary audit and its present decision state. | | [Translation certification register](translation_process_status.json) | Machine-readable status and checksum evidence for every translation document under the isolated process. | +| [Source reconstruction manifest](source_reconstruction_manifest.json) | Stored source, citation label, chapter boundaries, and normalization rule for each mechanically reconstructed translation. | | [Deferred questions](deferred_questions.md) | Parked questions and decisions, with the condition for returning to each one. | | [Model continuation handoff](handoff/README.md) | Dated transfer package with the exact work state, repository utilities, active vocabulary method, and standing maintainer instructions. | | [Publishing strategy](publishing.md) | Order of operations for public access and eventual print publication. | diff --git a/project/development_log.md b/project/development_log.md index 003fa3ae..9341e35c 100644 --- a/project/development_log.md +++ b/project/development_log.md @@ -131,6 +131,7 @@ Implementation status, dependencies, optional evaluation work, and the next deve | D113 | Accepted | Certify Gibran's *On Children* under the isolated translation process. | Eighteen aligned units reconstruct all 980 normalized Gibran characters. The source-only pass gives the framing exchange explicit participants, establishes `ne musatapha`, reserves reflexive `miso` for a true object, states thought possession and plurality directly, and unfolds the archery image through visible materials and roles. Anonymous derivation and independent source-blind review expose three invalid possessive structures, an incorrectly attached request topic, and ambiguous strength and gladness relations. Each affected English layer is discarded before the Phi is repaired and freshly derived. The final stream freezes at SHA-256 `028a864fb7bb333245159dcf9e2048b9021c3dab04a34f0e299506da9d059455`. No root, module membership, registered compound, function word, or grammar is added. | | D114 | Accepted | Give certified translations a public mark and explanation generated from the process register. | The website links every certified translation page to a public account of the two isolated directions, shows compact marks in the short-work and collection catalogues, and lists the current certified works from `project/translation_process_status.json`. The explanation states prominently that certified texts can still contain errors. The build validates the register before rendering and stops when a translation page, catalogue, or public list disagrees with it. Pending translations receive no mark. | | D115 | Accepted | Close the completed lexicon prose migration at the schema boundary. | Every lexicon entry must carry `description`, `articulatory_notes`, and structured `examples`. The schema rejects the retired `concept` and `grammatical_notes` properties instead of accepting them as fallbacks. Regression tests prove both requirements and both rejections, while the completed migration ledger remains available as evidence. No vocabulary entry or generated lexicon reference changes. | +| D116 | Accepted | Verify ordered source reconstruction against an independently stored witness. | A manifest selects each of the six current *News from Nowhere* chapters from the complete Morris source, names the `morris` citation stream, and applies a stated Project Gutenberg line-wrap normalization. The checker proves that 1,054 ordered citations reconstruct 75,243 normalized characters exactly and reports missing, duplicated, reordered, or altered spans directly. A required chapter pattern stops later chapters from bypassing the check, and synthetic regression tests exercise every failure class in CI. No translation layer changes. | ## Completed discretionary lexical migrations diff --git a/project/near_term_development_plan.md b/project/near_term_development_plan.md index 46460453..71ddf653 100644 --- a/project/near_term_development_plan.md +++ b/project/near_term_development_plan.md @@ -25,8 +25,8 @@ Community engagement is outside this plan. Phi can strengthen its language, corp | ID | Status | Work package | Completion gate | |---|---|---|---| | NTP-01 | **DONE** | Close the migrated vocabulary schema | Current entries pass a schema that requires the target prose fields and no longer accepts the retired fields. | -| NTP-02 | **NEXT** | Add independent source-reconstruction checking | At least the locally stored Morris source can be compared mechanically with each chapter's ordered citation stream. | -| NTP-03 | **READY** | Establish the *News from Nowhere* continuity record | Recurring source-to-Phi choices have one maintained home that is excluded from Phi-to-English derivation. | +| NTP-02 | **DONE** | Add independent source-reconstruction checking | At least the locally stored Morris source can be compared mechanically with each chapter's ordered citation stream. | +| NTP-03 | **NEXT** | Establish the *News from Nowhere* continuity record | Recurring source-to-Phi choices have one maintained home that is excluded from Phi-to-English derivation. | | NTP-04 | **READY** | Reconcile stale project records | The roadmap, development protocol, and deferred records describe the actual completed corpus and closed grammar boundary. | | NTP-05 | **READY** | Finish the short-work certification queue | The present Gibran, Taoist, and Buddhist short works have completed the isolated process and the generated register passes. | | NTP-06 | **PENDING** | Certify *News from Nowhere* chapter 1 as a pilot | The chapter has a current certification record and its findings have been classified as local or series-wide. | @@ -45,9 +45,11 @@ Completed under D115. The target fields are direct requirements, the retired pro ## NTP-02: Verify source reconstruction independently -The certification register verifies unit counts, digests, aligned layers, and recorded normalized source counts. It does not by itself compare an ordered citation stream with an independently stored source witness. That leaves a gap between a strong process and the strongest reading of the public reconstruction claim. +The certification register verified unit counts, digests, aligned layers, and recorded normalized source counts. It did not by itself compare an ordered citation stream with an independently stored source witness. That left a gap between a strong process and the strongest reading of the public reconstruction claim. -Begin with *News from Nowhere*, whose source witness is stored in the repository. A small manifest should identify the source file, citation label, selected chapter boundary, and normalization rule. A checker should join the chapter's decoded citations in order and compare the result character for character with the selected normalized source. It should report a missing span, duplicated span, reordering, or altered text directly. The design may later support other works with independently stored witnesses, but it should not pretend that every source has one when it does not. +The package begins with *News from Nowhere*, whose source witness is stored in the repository. A small manifest identifies the source file, citation label, selected chapter boundary, and normalization rule. A checker joins the chapter's decoded citations in order and compares the result character for character with the selected normalized source. It reports a missing span, duplicated span, reordering, or altered text directly. The design may later support other works with independently stored witnesses, but it does not pretend that every source has one when it does not. + +Completed under D116. All six current chapters reconstruct 75,243 normalized source characters across 1,054 citations. The manifest's required chapter pattern prevents a later Morris chapter from bypassing the check, and CI exercises the four named failure classes with synthetic mutations. ## NTP-03: Preserve Morris continuity diff --git a/project/source_reconstruction_manifest.json b/project/source_reconstruction_manifest.json new file mode 100644 index 00000000..d56cc258 --- /dev/null +++ b/project/source_reconstruction_manifest.json @@ -0,0 +1,68 @@ +{ + "format": "phi-source-reconstruction-v1", + "required_translation_globs": [ + "texts/news_from_nowhere/chapter_*.md" + ], + "documents": [ + { + "translation": "texts/news_from_nowhere/chapter_01.md", + "source": "texts/news_from_nowhere/source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER I: DISCUSSION AND BED", + "end_before": "CHAPTER II: A MORNING BATH" + }, + "normalization": "gutenberg-wrapped-prose-v1" + }, + { + "translation": "texts/news_from_nowhere/chapter_02.md", + "source": "texts/news_from_nowhere/source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER II: A MORNING BATH", + "end_before": "CHAPTER III: THE GUEST HOUSE AND BREAKFAST THEREIN" + }, + "normalization": "gutenberg-wrapped-prose-v1" + }, + { + "translation": "texts/news_from_nowhere/chapter_03.md", + "source": "texts/news_from_nowhere/source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER III: THE GUEST HOUSE AND BREAKFAST THEREIN", + "end_before": "CHAPTER IV: A MARKET BY THE WAY" + }, + "normalization": "gutenberg-wrapped-prose-v1" + }, + { + "translation": "texts/news_from_nowhere/chapter_04.md", + "source": "texts/news_from_nowhere/source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER IV: A MARKET BY THE WAY", + "end_before": "CHAPTER V: CHILDREN ON THE ROAD" + }, + "normalization": "gutenberg-wrapped-prose-v1" + }, + { + "translation": "texts/news_from_nowhere/chapter_05.md", + "source": "texts/news_from_nowhere/source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER V: CHILDREN ON THE ROAD", + "end_before": "CHAPTER VI: A LITTLE SHOPPING" + }, + "normalization": "gutenberg-wrapped-prose-v1" + }, + { + "translation": "texts/news_from_nowhere/chapter_06.md", + "source": "texts/news_from_nowhere/source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER VI: A LITTLE SHOPPING", + "end_before": "CHAPTER VII: TRAFALGAR SQUARE" + }, + "normalization": "gutenberg-wrapped-prose-v1" + } + ] +} diff --git a/scripts/README.md b/scripts/README.md index aeb1812d..ba723c02 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -136,6 +136,17 @@ python3 scripts/translation_layers.py texts/north_wind_and_sun.md --digest-only python3 scripts/test_translation_layers.py ``` +## source_reconstruction.py + +Checks the translation and source relationships declared in `project/source_reconstruction_manifest.json`. Each entry names the stored witness, citation label, source boundaries, and normalization rule. The current manifest selects the first six chapters of *News from Nowhere* from the complete Morris source and compares each chapter with its ordered `morris` citations. Its required path pattern also makes an unregistered new chapter an error. + +The `gutenberg-wrapped-prose-v1` rule rejoins a hyphenated word split by a Project Gutenberg line wrap, then collapses the remaining whitespace. It does not change spelling, punctuation, or case. A mismatch is reported as a missing span, duplicated span, reordered span, or alteration at the first citation where the streams diverge. + +```bash +python3 scripts/source_reconstruction.py +python3 scripts/test_source_reconstruction.py +``` + ## translation_process_status.py Validates the complete D102 certification queue and generates its readable ledger. Standalone translations and Gibran selections come from their catalogues; every catalogued book must declare whether the process applies, and the current *News from Nowhere* chapter glob supplies that book's documents. A certified row records both the frozen Phi digest and a digest over all four published layers, so later drift in Phi, gloss, derived English, or source citations stops CI. diff --git a/scripts/source_reconstruction.py b/scripts/source_reconstruction.py new file mode 100644 index 00000000..416527d5 --- /dev/null +++ b/scripts/source_reconstruction.py @@ -0,0 +1,374 @@ +#!/usr/bin/env python3 +"""Verify ordered translation citations against independent source witnesses.""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from dataclasses import dataclass +from pathlib import Path + +from translation_layers import parse_interlinear_units + + +ROOT = Path(__file__).resolve().parent.parent +MANIFEST_FILE = ROOT / "project" / "source_reconstruction_manifest.json" +FORMAT = "phi-source-reconstruction-v1" +NORMALIZATION = "gutenberg-wrapped-prose-v1" +LABEL_RE = re.compile(r"[a-z][a-z0-9-]*") +DOCUMENT_FIELDS = { + "translation", + "source", + "citation_label", + "selection", + "normalization", +} +SELECTION_FIELDS = {"start_after", "end_before"} +MANIFEST_FIELDS = {"format", "required_translation_globs", "documents"} + + +@dataclass(frozen=True) +class ReconstructionResult: + translation: str + citation_count: int + normalized_characters: int + + +def load_manifest(path: Path = MANIFEST_FILE): + return json.loads(path.read_text(encoding="utf-8")) + + +def normalize_gutenberg_prose(text: str) -> str: + """Undo Project Gutenberg line wrapping and collapse layout whitespace.""" + text = text.replace("\r\n", "\n").replace("\r", "\n") + text = re.sub(r"-\n(?=[A-Za-z])", "-", text) + return " ".join(text.split()) + + +def repository_file(root: Path, value, role: str) -> tuple[Path | None, str | None]: + if not isinstance(value, str) or not value: + return None, f"{role} must be a non-empty repository-relative path" + relative = Path(value) + if relative.is_absolute() or ".." in relative.parts: + return None, f"{role} must stay inside the repository: {value!r}" + root = root.resolve() + candidate = (root / relative).resolve() + try: + candidate.relative_to(root) + except ValueError: + return None, f"{role} must stay inside the repository: {value!r}" + if not candidate.is_file(): + return None, f"{role} does not exist: {value}" + return candidate, None + + +def marker_match(text: str, marker: str): + pattern = re.compile(rf"^{re.escape(marker)}$", re.M) + return list(pattern.finditer(text)) + + +def select_source( + text: str, selection: dict, context: str +) -> tuple[str | None, list[str]]: + errors: list[str] = [] + if not isinstance(selection, dict): + return None, [f"{context}: selection must be an object"] + missing = sorted(SELECTION_FIELDS - set(selection)) + extra = sorted(set(selection) - SELECTION_FIELDS) + if missing: + errors.append(f"{context}: selection is missing {', '.join(missing)}") + if extra: + errors.append(f"{context}: selection has unknown fields: {', '.join(extra)}") + if errors: + return None, errors + + start_marker = selection["start_after"] + end_marker = selection["end_before"] + if not isinstance(start_marker, str) or not start_marker: + errors.append(f"{context}: start_after must be a non-empty string") + if not isinstance(end_marker, str) or not end_marker: + errors.append(f"{context}: end_before must be a non-empty string") + if errors: + return None, errors + + starts = marker_match(text, start_marker) + ends = marker_match(text, end_marker) + if len(starts) != 1: + errors.append( + f"{context}: start marker must occur once, found {len(starts)}: {start_marker!r}" + ) + if len(ends) != 1: + errors.append( + f"{context}: end marker must occur once, found {len(ends)}: {end_marker!r}" + ) + if errors: + return None, errors + if starts[0].end() >= ends[0].start(): + return None, [f"{context}: source markers are reversed or overlap"] + return text[starts[0].end():ends[0].start()], [] + + +def excerpt(text: str, limit: int = 96) -> str: + text = text.strip() + if len(text) <= limit: + return text + return text[: limit - 3].rstrip() + "..." + + +def contains_span(stream: str, span: str) -> bool: + if not span: + return False + return ( + stream == span + or stream.startswith(span + " ") + or stream.endswith(" " + span) + or f" {span} " in stream + ) + + +def reconstruction_error( + expected: str, fragments: list[str], context: str +) -> str | None: + cursor = 0 + for index, fragment in enumerate(fragments): + separator = "" if index == 0 else " " + piece = separator + fragment + if expected.startswith(piece, cursor): + cursor += len(piece) + continue + + citation_number = index + 1 + later_citations = " ".join(fragments[index + 1:]) + if fragment in fragments[:index]: + return ( + f"{context}: duplicated source span at citation {citation_number}: " + f"{excerpt(fragment)!r}" + ) + + forward = expected.find(piece, cursor + 1) + if forward != -1: + skipped = expected[cursor:forward].strip() + if contains_span(later_citations, skipped): + return ( + f"{context}: reordered source spans at citation {citation_number}: " + f"expected {excerpt(skipped)!r} before {excerpt(fragment)!r}" + ) + return ( + f"{context}: missing source span before citation {citation_number}: " + f"{excerpt(skipped)!r}" + ) + + prior = expected.rfind(piece, 0, cursor) + if prior != -1: + return ( + f"{context}: reordered source span at citation {citation_number}: " + f"{excerpt(fragment)!r} belongs before the current position" + ) + + expected_near = expected[cursor:cursor + max(len(piece), 48)].strip() + return ( + f"{context}: altered source text at citation {citation_number}: " + f"expected near {excerpt(expected_near)!r}, found {excerpt(fragment)!r}" + ) + + if cursor < len(expected): + return f"{context}: missing trailing source span: {excerpt(expected[cursor:])!r}" + return None + + +def check_manifest(data, root: Path = ROOT) -> tuple[list[ReconstructionResult], list[str]]: + results: list[ReconstructionResult] = [] + errors: list[str] = [] + if not isinstance(data, dict): + return results, ["source reconstruction manifest must be an object"] + missing_manifest_fields = sorted(MANIFEST_FIELDS - set(data)) + extra_manifest_fields = sorted(set(data) - MANIFEST_FIELDS) + if missing_manifest_fields: + errors.append( + "manifest is missing fields: " + ", ".join(missing_manifest_fields) + ) + if extra_manifest_fields: + errors.append( + "manifest has unknown fields: " + ", ".join(extra_manifest_fields) + ) + if data.get("format") != FORMAT: + errors.append(f"format must be {FORMAT!r}") + required_globs = data.get("required_translation_globs") + required_translations: set[str] = set() + if not isinstance(required_globs, list) or not required_globs: + errors.append("required_translation_globs must be a non-empty list") + else: + for pattern in required_globs: + if not isinstance(pattern, str) or not pattern: + errors.append("every required translation glob must be a non-empty string") + continue + relative_pattern = Path(pattern) + if relative_pattern.is_absolute() or ".." in relative_pattern.parts: + errors.append( + f"required translation glob must stay inside the repository: {pattern!r}" + ) + continue + matches = sorted(path for path in root.glob(pattern) if path.is_file()) + if not matches: + errors.append(f"required translation glob matches no files: {pattern!r}") + continue + for match in matches: + resolved = match.resolve() + try: + relative = resolved.relative_to(root.resolve()).as_posix() + except ValueError: + errors.append( + f"required translation glob leaves the repository: {pattern!r}" + ) + continue + required_translations.add(relative) + documents = data.get("documents") + if not isinstance(documents, list) or not documents: + return results, [*errors, "documents must be a non-empty list"] + + translations: list[str] = [] + for index, document in enumerate(documents, start=1): + context = f"manifest document {index}" + if not isinstance(document, dict): + errors.append(f"{context}: entry must be an object") + continue + translation_value = document.get("translation") + if isinstance(translation_value, str) and translation_value: + context = translation_value + translations.append(translation_value) + + missing = sorted(DOCUMENT_FIELDS - set(document)) + extra = sorted(set(document) - DOCUMENT_FIELDS) + if missing: + errors.append(f"{context}: missing fields: {', '.join(missing)}") + if extra: + errors.append(f"{context}: unknown fields: {', '.join(extra)}") + if missing or extra: + continue + + label = document["citation_label"] + if not isinstance(label, str) or LABEL_RE.fullmatch(label) is None: + errors.append(f"{context}: citation_label is invalid: {label!r}") + continue + if document["normalization"] != NORMALIZATION: + errors.append( + f"{context}: unsupported normalization {document['normalization']!r}" + ) + continue + + translation_path, translation_error = repository_file( + root, document["translation"], "translation" + ) + source_path, source_error = repository_file(root, document["source"], "source") + if translation_error: + errors.append(f"{context}: {translation_error}") + if source_error: + errors.append(f"{context}: {source_error}") + if translation_path is None or source_path is None: + continue + + selected, selection_errors = select_source( + source_path.read_text(encoding="utf-8"), document["selection"], context + ) + errors.extend(selection_errors) + if selected is None: + continue + expected = normalize_gutenberg_prose(selected) + if not expected: + errors.append(f"{context}: selected source is empty after normalization") + continue + + try: + units = parse_interlinear_units( + translation_path.read_text(encoding="utf-8") + ) + except ValueError as exc: + errors.append(f"{context}: cannot decode citations: {exc}") + continue + if not units: + errors.append(f"{context}: no aligned translation units found") + continue + + fragments: list[str] = [] + missing_label_units: list[int] = [] + for unit_number, unit in enumerate(units, start=1): + witnesses = [ + witness + for source_label, witness in unit.sources + if source_label == label + ] + if not witnesses: + missing_label_units.append(unit_number) + fragments.extend(normalize_gutenberg_prose(witness) for witness in witnesses) + if missing_label_units: + listed = ", ".join(str(number) for number in missing_label_units[:8]) + suffix = "..." if len(missing_label_units) > 8 else "" + errors.append( + f"{context}: {len(missing_label_units)} translation unit(s) lack " + f"citation label {label!r}: {listed}{suffix}" + ) + continue + if any(not fragment for fragment in fragments): + errors.append(f"{context}: citation label {label!r} has an empty witness") + continue + + mismatch = reconstruction_error(expected, fragments, context) + if mismatch: + errors.append(mismatch) + continue + results.append( + ReconstructionResult( + translation=document["translation"], + citation_count=len(fragments), + normalized_characters=len(expected), + ) + ) + + duplicates = sorted( + translation + for translation in set(translations) + if translations.count(translation) > 1 + ) + for translation in duplicates: + errors.append(f"duplicate translation in manifest: {translation}") + for translation in sorted(required_translations - set(translations)): + errors.append(f"missing source reconstruction manifest entry: {translation}") + return results, errors + + +def parse_args(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--manifest", + type=Path, + default=MANIFEST_FILE, + help="manifest to check", + ) + return parser.parse_args(argv) + + +def main(argv=None) -> int: + args = parse_args(argv) + try: + data = load_manifest(args.manifest) + except (OSError, json.JSONDecodeError) as exc: + print(f"cannot read source reconstruction manifest: {exc}", file=sys.stderr) + return 1 + results, errors = check_manifest(data) + if errors: + for error in errors: + print(error, file=sys.stderr) + return 1 + print( + f"checked {len(results)} source reconstruction(s): " + f"{sum(result.citation_count for result in results)} citations, " + f"{sum(result.normalized_characters for result in results):,} " + "normalized source characters" + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/test_source_reconstruction.py b/scripts/test_source_reconstruction.py new file mode 100644 index 00000000..8959a5c7 --- /dev/null +++ b/scripts/test_source_reconstruction.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Tests for independent source-reconstruction checking.""" + +import copy +import json +import tempfile +import unittest +from pathlib import Path + +import source_reconstruction as reconstruction + + +class SourceReconstructionTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.repository_manifest = reconstruction.load_manifest() + + def fixture(self, citations): + temporary = tempfile.TemporaryDirectory() + root = Path(temporary.name) + source = root / "source.txt" + translation = root / "translation.md" + source.write_text( + "CHAPTER I: TEST\n\nAlpha one. Beta two. Gamma three.\n\n" + "CHAPTER II: NEXT\n\nDelta four.\n", + encoding="utf-8", + ) + blocks = [] + for index, citation in enumerate(citations, start=1): + blocks.append( + "```\n" + f"mia nai.\n1SG be.\n(I exist, unit {index}.)\n" + f"morris: {json.dumps(citation)}\n" + "```" + ) + translation.write_text("\n\n".join(blocks) + "\n", encoding="utf-8") + data = { + "format": reconstruction.FORMAT, + "required_translation_globs": ["translation.md"], + "documents": [ + { + "translation": "translation.md", + "source": "source.txt", + "citation_label": "morris", + "selection": { + "start_after": "CHAPTER I: TEST", + "end_before": "CHAPTER II: NEXT", + }, + "normalization": reconstruction.NORMALIZATION, + } + ], + } + return temporary, root, data + + def check_fixture(self, citations): + temporary, root, data = self.fixture(citations) + self.addCleanup(temporary.cleanup) + return reconstruction.check_manifest(data, root) + + def test_repository_manifest_reconstructs_six_morris_chapters(self): + results, errors = reconstruction.check_manifest(self.repository_manifest) + self.assertEqual(errors, []) + self.assertEqual(len(results), 6) + self.assertEqual(sum(result.citation_count for result in results), 1054) + self.assertEqual( + sum(result.normalized_characters for result in results), + 75243, + ) + self.assertEqual( + [result.translation for result in results], + [f"texts/news_from_nowhere/chapter_{number:02}.md" for number in range(1, 7)], + ) + + def test_exact_reconstruction_passes(self): + results, errors = self.check_fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.assertEqual(errors, []) + self.assertEqual(results[0].normalized_characters, 33) + + def test_gutenberg_normalization_repairs_layout_only(self): + source = "After-\nlecture Morris--Word\n\nStays." + self.assertEqual( + reconstruction.normalize_gutenberg_prose(source), + "After-lecture Morris--Word Stays.", + ) + + def test_missing_span_is_named(self): + _results, errors = self.check_fixture(["Alpha one.", "Gamma three."]) + self.assertTrue(any("missing source span" in error for error in errors)) + + def test_duplicated_span_is_named(self): + _results, errors = self.check_fixture( + ["Alpha one.", "Beta two.", "Beta two.", "Gamma three."] + ) + self.assertTrue(any("duplicated source span" in error for error in errors)) + + def test_reordered_spans_are_named(self): + _results, errors = self.check_fixture(["Alpha one.", "Gamma three.", "Beta two."]) + self.assertTrue(any("reordered source spans" in error for error in errors)) + + def test_altered_text_is_named(self): + _results, errors = self.check_fixture(["Alpha one.", "Beta too.", "Gamma three."]) + self.assertTrue(any("altered source text" in error for error in errors)) + + def test_unknown_normalization_is_rejected(self): + temporary, root, data = self.fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.addCleanup(temporary.cleanup) + data["documents"][0]["normalization"] = "unknown" + _results, errors = reconstruction.check_manifest(data, root) + self.assertTrue(any("unsupported normalization" in error for error in errors)) + + def test_missing_boundary_marker_is_rejected(self): + temporary, root, data = self.fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.addCleanup(temporary.cleanup) + data["documents"][0]["selection"]["start_after"] = "CHAPTER ZERO" + _results, errors = reconstruction.check_manifest(data, root) + self.assertTrue(any("start marker must occur once" in error for error in errors)) + + def test_repository_escape_is_rejected(self): + temporary, root, data = self.fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.addCleanup(temporary.cleanup) + data["documents"][0]["source"] = "../source.txt" + _results, errors = reconstruction.check_manifest(data, root) + self.assertTrue(any("must stay inside the repository" in error for error in errors)) + + def test_duplicate_translation_is_rejected(self): + temporary, root, data = self.fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.addCleanup(temporary.cleanup) + data["documents"].append(copy.deepcopy(data["documents"][0])) + _results, errors = reconstruction.check_manifest(data, root) + self.assertTrue(any("duplicate translation" in error for error in errors)) + + def test_required_translation_without_manifest_entry_is_rejected(self): + temporary, root, data = self.fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.addCleanup(temporary.cleanup) + extra = root / "translation_extra.md" + extra.write_text((root / "translation.md").read_text(encoding="utf-8"), encoding="utf-8") + data["required_translation_globs"] = ["translation*.md"] + _results, errors = reconstruction.check_manifest(data, root) + self.assertTrue( + any("missing source reconstruction manifest entry" in error for error in errors) + ) + + def test_required_translation_glob_cannot_escape_repository(self): + temporary, root, data = self.fixture(["Alpha one.", "Beta two.", "Gamma three."]) + self.addCleanup(temporary.cleanup) + data["required_translation_globs"] = ["../*.md"] + _results, errors = reconstruction.check_manifest(data, root) + self.assertTrue( + any("glob must stay inside the repository" in error for error in errors) + ) + + +if __name__ == "__main__": + unittest.main()