diff --git a/docs/DESIGN-RULES.md b/docs/DESIGN-RULES.md index 1bc4152..fe5312e 100644 --- a/docs/DESIGN-RULES.md +++ b/docs/DESIGN-RULES.md @@ -115,7 +115,8 @@ an explicit cue that thinking time is expected. A prompt without a pause is a li ## 3. Sentence-level style (`SENT-*`, `ORI-*`, `SIG-*`, `VOI-*`) **SENT-01** Median sentence length **≤ 20 words**; hard cap **35 words**; no sentence with more than -two subordinate clauses. Long source sentences are split, not compressed. *(KB §2.1; lint)* +two subordinate clauses. Long source sentences are split, not compressed. *(KB §2.1; lint; stage 6 +`split`, which is checked in both directions because a split is lossless and a summary is not)* **SENT-02** **No unresolved anaphora across a beat boundary.** "It", "this", "the former/latter", "the above" must be replaced by the referent whenever the antecedent is more than one sentence back diff --git a/docs/PLAN-ai.md b/docs/PLAN-ai.md index debe1e3..4dfad11 100644 --- a/docs/PLAN-ai.md +++ b/docs/PLAN-ai.md @@ -31,7 +31,7 @@ feature nobody can turn on. | Where | What the model does | Provider | Ever run? | |---|---|---|---| -| Stage 6, `elaborate` | gloss, anchor, analogy, why, compress | Anthropic only | **No** | +| Stage 6, `elaborate` | gloss, anchor, analogy, why, compress, split | Anthropic only | **No** | | Stage 6, figures | describe a rendered crop | Anthropic only | **No** | | Stage 6, gate | entailment check (`GRD-02`) | Anthropic only | **No** | | MCP assistant | *all of the above*, written in the conversation | none needed | **Yes** | @@ -60,7 +60,7 @@ should stop treating them as one thing. **Author.** The model produces text that reaches the listener. Highest risk — a fluent wrong sentence is the worst output this system can make — and every output must pass the grounding -gate. Currently: gloss, anchor, analogy, why, compress, figure. +gate. Currently: gloss, anchor, analogy, why, compress, figure, split. **Critic.** The model reads output the deterministic pipeline produced and says what is wrong with it. Low risk: its output is a report, not a script, and a wrong criticism costs a human diff --git a/docs/ai/index.md b/docs/ai/index.md index 5f2a196..327797c 100644 --- a/docs/ai/index.md +++ b/docs/ai/index.md @@ -8,7 +8,8 @@ icon: lucide/brain-circuit fluency, and the structure is deterministic: a document goes in and a listenable, checkable programme comes out with no model involved at all. What a model adds is the explaining — a one-line gloss, a concrete anchor, an analogy that says where it breaks down, a description of a -figure you cannot see. +figure you cannot see — and one thing that is not explaining at all: cutting the paper's longest +sentences into ones you can hold in your head. Start here: @@ -71,6 +72,23 @@ sentence must appear in the source sentences it was written from; a claim that r paper's direction is rejected before it reaches the programme (`GRD-03`). You can watch that happen — the elaboration report lists what was accepted, what was rejected and why. +**One task rewrites the paper's own words, and only one.** Rule `SENT-01` caps a sentence at +thirty-five spoken words, because a sentence you would re-read on the page is simply lost in +audio — and 465 sentences across a twelve-paper corpus are over it. The rule says what to do +about them: *long source sentences are split, not compressed.* + +That distinction is the whole of it. A summary of a sentence reads exactly like a split of it, +and the listener has no way to tell which they were given. So the check runs in **both** +directions: every number in the source must appear in the split, and every number in the split +must appear in the source. Nothing else a model writes here can fail the first of those, because +nothing else is supposed to be lossless. It is what makes this safe to do to a paper's sentences +at all. + +The paper is never edited. The split is stored beside the sentence it replaces, the span still +points at what was written, and `study.md` can show you one against the other. When a split +drops a value it is rejected and the long sentence is spoken as it stands — which happened twice +in ten on the first live run, and a long true sentence beats a short lossy one every time. + **When the model is absent, every task degrades along a documented path** and the manifest records *why* — there are four different reasons and they need four different actions from you: no provider configured, configured but unreachable, reached but refused the shape, or out of diff --git a/src/mimem/config.py b/src/mimem/config.py index 43396b1..cc4efa1 100644 --- a/src/mimem/config.py +++ b/src/mimem/config.py @@ -104,6 +104,15 @@ class ElaborationBudget(BaseModel): max_anchors: int = 4 anchor_min_abstractness: float = 0.50 + #: Sentences to split in one build (rule SENT-01), longest first. + #: + #: A cap on *calls*, unlike everything above it, and it has to be: the others scale with the + #: number of concepts a paper has and this one scales with the paper. A ninety-eight page + #: review offers well over a thousand sentences past the cap, and splitting all of them would + #: cost more than every other task in this file put together while the four analogies that + #: carry the programme went unwritten. + max_splits: int = 40 + #: Which implementation answers each task: ``off``, ``assist`` or ``prefer``. See #: :mod:`mimem.elaborate.reconcile`. #: @@ -121,7 +130,7 @@ class ElaborationBudget(BaseModel): #: keep a paid run away from it entirely. modes: dict[str, str] = Field( default_factory=lambda: dict.fromkeys( - ("gloss", "anchor", "analogy", "why", "figure", "compress"), "prefer" + ("gloss", "anchor", "analogy", "why", "figure", "compress", "split"), "prefer" ) ) diff --git a/src/mimem/elaborate/run.py b/src/mimem/elaborate/run.py index f100ec8..e04f845 100644 --- a/src/mimem/elaborate/run.py +++ b/src/mimem/elaborate/run.py @@ -31,9 +31,11 @@ from mimem.config import Listener, Profile from mimem.elaborate.reconcile import Absence, Deterministic, Mode, classify +from mimem.elaborate.sentences import LongSentence, over_long, verify_split from mimem.ir import ( Analogy, Anchor, + Block, Concept, ConceptRegistry, Document, @@ -44,7 +46,7 @@ from mimem.llm.cache import CacheStats from mimem.llm.client import Client, LLMRefusedError, LLMUnavailableError, NullClient, Request from mimem.llm.cost import BudgetExceededError, Ledger, Plan, estimate -from mimem.llm.schemas import AnalogyOut, AnchorOut, FigureOut, GlossOut, WhyOut +from mimem.llm.schemas import AnalogyOut, AnchorOut, FigureOut, GlossOut, SplitOut, WhyOut # The supporting-sentence index is shared between stage 6 and stage 7. It lives with the # planner, which is its heavier user; importing it here is deliberate rather than a layering @@ -57,8 +59,14 @@ #: need to: the task carries the sentences that matter, and the document is context. MAX_DOCUMENT_CHARS = 60_000 +#: The cap the split is asked to get under, and it is *below* the linter's thirty-five. A +#: sentence that lands exactly on the limit written is over it spoken, because the numbers in it +#: have not been said yet -- see :func:`~mimem.elaborate.sentences.over_long`. Asking for +#: twenty-five leaves room for the verbalizer. +SPLIT_CAP = 25 + #: Every task that can be reconciled, so a mode exists for each. -TASKS = ("gloss", "anchor", "analogy", "why", "figure", "compress") +TASKS = ("gloss", "anchor", "analogy", "why", "figure", "compress", "split") @dataclass(frozen=True) @@ -217,6 +225,12 @@ def _requests( for task in figure_tasks(doc): yield tasks.figure(task.caption, task.references, document) + # Sentence splits, which are the one task whose count scales with the length of the paper + # rather than with the number of concepts -- so a dry run that did not price them was + # quoting for the wrong build on anything longer than a letter. + for sentence in over_long(doc, profile, listener)[: profile.elaboration.max_splits]: + yield tasks.split(sentence.text, document, SPLIT_CAP) + def elaborate( doc: Document, @@ -275,7 +289,7 @@ def elaborate( report, client, tasks.gloss(concept, support, document, listener), - concept, + concept.canonical, partial(_apply_gloss, concept), source=source, fallback="the source's own definitional sentence", @@ -289,7 +303,7 @@ def elaborate( report, client, tasks.anchor(concept, support, document, listener), - concept, + concept.canonical, partial(_apply_anchor, concept), source=source, kinds=GROUNDING_KINDS["anchor"], @@ -303,7 +317,7 @@ def elaborate( report, client, tasks.why(concept, support, document), - concept, + concept.canonical, partial(_apply_why, concept, spans), source=source, fallback="no why-explanation", @@ -320,7 +334,7 @@ def elaborate( report, client, tasks.analogy(concept, support, document, listener), - concept, + concept.canonical, partial(_apply_analogy, concept), source="\n".join(support), kinds=GROUNDING_KINDS["analogy"], @@ -331,9 +345,69 @@ def elaborate( ) _describe_figures(report, client, doc, document, out_dir, model=model, progress=progress) + _split_sentences( + report, + client, + doc, + document, + profile, + listener, + mode=modes["split"], + model=model, + progress=progress, + ) return report +def _split_sentences( + report: ElaborationReport, + client: Client, + doc: Document, + document: str, + profile: Profile, + listener: Listener | None, + *, + mode: Mode, + model: str | None, + progress: Callable[[str, str], None] | None, +) -> None: + """Cut the sentences a listener cannot hold in one piece (rule SENT-01). + + Last, and deliberately. Every other task competes for the elaboration budget against the + *concepts* rule DIF-02 ranks; this one competes against the length of the paper, and a + hundred splits would starve the four analogies that carry the programme. Running it after + the others means the budget answers the question in the right order. + """ + if mode is Mode.OFF: + return + for sentence in over_long(doc, profile, listener)[: profile.elaboration.max_splits]: + block = doc.block(sentence.block_id) + _run( + report, + client, + tasks.split(sentence.text, document, SPLIT_CAP), + _shorten(sentence.text), + partial(_apply_split, block, sentence), + source=sentence.text, + inspect=lambda data, source: verify_split(source, list(data.sentences), cap=SPLIT_CAP), + fallback="the source sentence, unchanged", + model=model, + progress=progress, + mode=mode, + ) + + +def _shorten(text: str, words: int = 6) -> str: + """A sentence named by its opening, for the progress line and the degradation report.""" + head = text.split()[:words] + return " ".join(head) + ("..." if len(text.split()) > words else "") + + +def _apply_split(block: Block, sentence: LongSentence, out: SplitOut) -> None: + """Store the split beside the sentence it replaces, leaving the source alone.""" + block.rewrites[sentence.key] = " ".join(s.strip() for s in out.sentences) + + def _describe_figures( report: ElaborationReport, client: Client, @@ -401,12 +475,13 @@ def _run( report: ElaborationReport, client: Client, request: Request, - concept: Concept, + subject: str, apply: Callable[[Any], None], *, source: str, fallback: str, kinds: tuple[str, ...] = ("number", "year", "name", "direction"), + inspect: Callable[[Any, str], list[Finding]] | None = None, on_degrade: Deterministic | None = None, mode: Mode = Mode.PREFER, model: str | None = None, @@ -418,6 +493,13 @@ def _run( failure -- and under ``assist`` it runs *first*, which is the whole point of the mode: the rules answer what they are good at, and the model is asked only about the rest. It returns whether it produced anything, because nothing else can tell the caller that. + + ``inspect`` replaces the grounding check for a task whose contract is different. Everything + here writes *new* text and can only be asked whether it invented something; a sentence split + is lossless, so it is also asked whether it lost something, which no other task can fail. + + ``subject`` is a name for the thing being worked on, for the progress line and the + degradation report. It was a whole :class:`Concept` until the split arrived, which has none. """ deterministic = on_degrade @@ -436,7 +518,7 @@ def _run( if model: request = replace(request, model=model) if progress is not None: - progress(request.task, concept.canonical) + progress(request.task, subject) report._count(report.attempted, request.task) try: @@ -445,19 +527,23 @@ def _run( except BudgetExceededError: raise except (LLMUnavailableError, LLMRefusedError) as exc: - _degrade(report, request.task, concept.canonical, str(exc), fallback, classify(exc)) + _degrade(report, request.task, subject, str(exc), fallback, classify(exc)) if deterministic is not None and deterministic(): report._count(report.deterministic, request.task) return report.ledger.record(response) - findings = ground(response.data, source, kinds=kinds) + findings = ( + inspect(response.data, source) + if inspect is not None + else ground(response.data, source, kinds=kinds) + ) if findings: report.rejected.extend(findings) _degrade( report, request.task, - concept.canonical, + subject, "; ".join(str(f) for f in findings[:3]), fallback, Absence.REJECTED, diff --git a/src/mimem/elaborate/sentences.py b/src/mimem/elaborate/sentences.py new file mode 100644 index 0000000..8e8c7d1 --- /dev/null +++ b/src/mimem/elaborate/sentences.py @@ -0,0 +1,163 @@ +"""Splitting the source's long sentences (rule SENT-01). + +The design rule is one line and the whole module follows from it: *long source sentences are +split, not compressed*. A sentence a reader parses by going back a line is simply lost in audio, +and 465 of the stress corpus's sentences are over the thirty-five-word cap -- the largest single +family of lint findings in the project, and the one the rule's own docstring says belongs to a +rewriting stage. + +**Why a split is the one rewrite that can be checked.** Everything else stage 6 produces is +*new* text -- a gloss, an anchor, an analogy -- and the check can only ask whether it invented +something. A split invents nothing by definition, so the check runs both ways: every number and +name in the source must appear in the split, and every number and name in the split must appear +in the source. A summary passes the first test and fails the second; that asymmetry is what +makes the difference between splitting and compressing mechanically detectable, and it is the +only reason this is safe to do to a paper's own words. + +**The source is never edited.** The split is stored beside the sentence it replaces and the span +still points at what the paper wrote, so ``study.md`` can show one against the other and the +groundedness check has something real to check against. What changes is only what is spoken. + +**Selection is measured on the spoken form, not the written one.** "298 K" is two words on the +page and four in the ear, so a sentence that is comfortably inside the cap when read can be well +over it when heard -- which is what the linter sees and what the listener gets. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +from mimem.config import Listener, Profile +from mimem.ir import Block, BlockKind, Document, TriageAction +from mimem.verify import Finding, Severity, numbers_in + +#: Kinds whose text is not read out as prose. A table announced by its caption has no sentences +#: to split, and an equation's words are not the thing it means. +NOT_PROSE = frozenset( + {BlockKind.TABLE, BlockKind.FIGURE, BlockKind.EQUATION, BlockKind.CODE, BlockKind.CAPTION} +) + +#: How much longer the split may be than the sentence it replaces. Splitting costs words -- +#: a repeated subject, a connective made explicit -- and a little growth is the point. Half as +#: much again is not a split, it is an expansion, and expansion is where facts get added. +MAX_GROWTH = 1.5 + +#: Sentences to split in one build, most over the cap first. A cap on calls, because this is the +#: one task whose candidate count scales with the length of the document rather than with the +#: number of concepts: a ninety-eight page review offered 1,400 of them. +DEFAULT_MAX_SPLITS = 40 + + +@dataclass(frozen=True) +class LongSentence: + """A sentence over the cap, and where it is.""" + + block_id: str + start: int + end: int + text: str + spoken_words: int + + @property + def key(self) -> str: + return f"{self.start}:{self.end}" + + +def _narrated(doc: Document) -> list[Block]: + """Blocks whose sentences the programme will read out one by one.""" + return [ + block + for block in doc.blocks + if block.text.strip() + and block.kind not in NOT_PROSE + and not (block.triage is not None and block.triage.action is not TriageAction.KEEP) + ] + + +def over_long( + doc: Document, profile: Profile, listener: Listener | None = None, *, cap: int | None = None +) -> list[LongSentence]: + """Every sentence the listener would meet over the cap, longest first. + + Counted on the verbalized text, because that is the sentence the listener hears and the one + the linter measures. Reading the written form instead misses the sentences that are long + *because* of what they state: "a capacity of 3867.3 mAhg-1 at 0.1 C" is seven words written + and nineteen spoken. + """ + from mimem.lint.rules import SentenceLength + from mimem.render.narrate import uses_superscript_citations + from mimem.verbalize import verbalize_text + + limit = cap if cap is not None else SentenceLength.HARD_CAP + superscripts = uses_superscript_citations(doc) + + out: list[LongSentence] = [] + for block in _narrated(doc): + for start, end in block.sentences or [(0, len(block.text))]: + written = block.text[start:end].strip() + if not written: + continue + spoken = verbalize_text(written, profile, listener, strip_superscripts=superscripts) + words = len(spoken.split()) + if words > limit: + out.append(LongSentence(block.id, start, end, written, words)) + out.sort(key=lambda s: -s.spoken_words) + return out + + +def verify_split(original: str, sentences: list[str], *, cap: int) -> list[Finding]: + """Everything wrong with a proposed split, or nothing (rules SENT-01, GRD-03). + + Six checks, and the first two are the ones that matter. A split is *lossless*, so the numbers + have to match in both directions -- a dropped value is a fact the listener will never hear + and an added one is a fact the paper never stated. Neither is visible in the prose. + """ + out: list[Finding] = [] + joined = " ".join(sentences).strip() + + # The direction rule GRD-03 cannot check, because everything else stage 6 writes is allowed + # to leave things out and a split is not. A summary of a sentence passes every check in + # `check` and fails this one, which is the whole basis for doing this to a paper's own words. + for value in sorted(set(numbers_in(original)) - set(numbers_in(joined))): + out.append( + Finding( + kind="number", + value=value, + message="dropped by the split; a split may not lose a value", + ) + ) + + # ...and the direction it can: invented numbers, invented names, and a flipped comparison. + # "higher" for "lower" is the worst thing this system can produce and it reads perfectly. + from mimem.verify import check + + out.extend(check(joined, original)) + + if len(sentences) < 2: + out.append( + Finding( + kind="split", + value=str(len(sentences)), + message="one sentence back; nothing was split", + severity=Severity.WARNING, + ) + ) + over = [s for s in sentences if len(s.split()) > cap] + if over: + out.append( + Finding( + kind="split", + value=f"{len(over)} of {len(sentences)}", + message=f"still over the {cap}-word cap, so the split did not do its job", + severity=Severity.WARNING, + ) + ) + if len(joined.split()) > MAX_GROWTH * len(original.split()): + out.append( + Finding( + kind="split", + value=f"{len(joined.split())} words from {len(original.split())}", + message="longer than a split should be; this is an expansion", + ) + ) + return out diff --git a/src/mimem/ir/models.py b/src/mimem/ir/models.py index 8d01910..c2f16db 100644 --- a/src/mimem/ir/models.py +++ b/src/mimem/ir/models.py @@ -175,6 +175,12 @@ class Block(Base): triage: TriageDecision | None = None # filled by stage 3 attrs: dict[str, Any] = Field(default_factory=dict) + #: Sentences stage 6 split, keyed ``"start:end"`` by the offsets of the sentence they + #: replace (rule SENT-01). The *source* is never edited: a span still points at what the + #: paper wrote, which is what makes the rewrite checkable and what rule GRD-02 asks for. + #: What changes is only what gets spoken. + rewrites: dict[str, str] = Field(default_factory=dict) + @property def is_empty(self) -> bool: return not self.text.strip() @@ -186,6 +192,10 @@ def span(self) -> Span: def sentence_texts(self) -> list[str]: return [self.text[a:b] for a, b in self.sentences] + def spoken_text(self, start: int, end: int) -> str: + """What the programme says for this span: the split, where there is one.""" + return self.rewrites.get(f"{start}:{end}") or self.text[start:end] + class DiagnosticLevel(StrEnum): INFO = "info" diff --git a/src/mimem/llm/schemas.py b/src/mimem/llm/schemas.py index 9ee6faa..e938a42 100644 --- a/src/mimem/llm/schemas.py +++ b/src/mimem/llm/schemas.py @@ -83,6 +83,17 @@ class CompressOut(Out): text: str = Field(max_length=1200) +class SplitOut(Out): + """One long source sentence, cut into short ones (rule SENT-01). + + A list rather than a paragraph, because the *number* of sentences is the thing being asked + for and a single string would let the model return the original with a comma moved. Two at + least: a "split" that returns one sentence has not split anything. + """ + + sentences: list[str] = Field(min_length=2, max_length=6) + + class FigureOut(Out): """A figure description in the accessibility template order (rule FIG-01). diff --git a/src/mimem/llm/tasks.py b/src/mimem/llm/tasks.py index 0fc793f..db17c9f 100644 --- a/src/mimem/llm/tasks.py +++ b/src/mimem/llm/tasks.py @@ -30,6 +30,7 @@ CompressOut, FigureOut, GlossOut, + SplitOut, VerifyOut, WhyOut, ) @@ -227,6 +228,51 @@ def compress(text: str, document: str, seconds: float) -> Request: ) +def split(sentence: str, document: str, cap: int) -> Request: + """Cut one long source sentence into short ones (rule SENT-01). + + The design rule says **split, not compress**, and the instruction says so four ways, because + this is the one task where a fluent wrong answer is indistinguishable from a right one. A + summary of a sentence reads exactly like a split of it, and the listener has no way to know + which they were given. So the verification is bidirectional -- every number and name has to + survive in *both* directions -- and the instruction is written to make that verification + pass rather than to make the prose pretty. + + The subject is repeated rather than pronominalised. "X, which does Y" splits naturally into + "X. It does Y", and that is a rule ``SENT-02`` violation manufactured by the fix for + ``SENT-01``: the listener meets "it" at the start of a sentence with the referent now behind + a full stop. Repeating the noun costs two words and is the whole reason a split is safe. + + *Degrades to:* the source sentence, unchanged. Long, and the linter goes on saying so. + """ + instruction = f"""\ +Split this sentence into two or more shorter sentences, each under {cap} words. + +This is a split, not a summary. Every fact, number, unit, name and qualifier in the original +must appear in your sentences, and you must not add any that are not there. Do not shorten by +leaving something out: if a clause cannot be carried over, return the sentence unchanged as a +single-element list and it will be used as it is. + +Start each sentence with a noun, not with "it", "this", "they" or "these" — repeat the subject +instead. These sentences are heard, not read, so a pronoun at the start of one points at +something the listener can no longer see. + +Keep the paper's own wording wherever it fits. You are moving clauses apart, not rephrasing them. + +Sentence: +{sentence} +""" + return Request( + task="split", + system=SYSTEM, + document=document, + instruction=instruction, + schema=SplitOut, + effort=EFFORT_LOW, + max_tokens=900, + ) + + def figure( caption: str, references: list[str], document: str, image: bytes | None = None ) -> Request: diff --git a/src/mimem/plan/beats.py b/src/mimem/plan/beats.py index 06cca88..94d5c65 100644 --- a/src/mimem/plan/beats.py +++ b/src/mimem/plan/beats.py @@ -187,7 +187,11 @@ def flush() -> None: if not batch: return start, end = batch[0][0], batch[-1][1] - raw = block.text[start:end].strip() + # Per sentence, not over the whole batch: a beat covers several, and stage 6 splits one + # at a time (rule SENT-01). Where it has, the split is what gets spoken; the span still + # covers the source, which is what rule GRD-02 checks the beat against and what + # ``study.md`` shows a reader who wants to know what the paper actually said. + raw = " ".join(block.spoken_text(a, b).strip() for a, b in batch).strip() spoken = factory.speak(raw) if spoken: out.append( diff --git a/tests/unit/test_sentence_splits.py b/tests/unit/test_sentence_splits.py new file mode 100644 index 0000000..ff9d8b4 --- /dev/null +++ b/tests/unit/test_sentence_splits.py @@ -0,0 +1,221 @@ +"""Stage 6 splitting the source's long sentences (rule SENT-01). + +The design rule is *long source sentences are split, not compressed*, and these tests are +mostly about the difference. A summary of a sentence reads exactly like a split of it, so the +prose cannot tell them apart and the check has to: a split is lossless, so every number has to +survive in **both** directions, and that asymmetry is the only reason this is safe to do to a +paper's own words. + +The answers are scripted rather than recorded, for the reason ``test_elaborate`` gives: these +check the plumbing and the gate, neither of which should depend on what a model says today. +""" + +from __future__ import annotations + +import pytest + +from mimem.config import Listener, load_profile +from mimem.elaborate import elaborate +from mimem.elaborate.sentences import MAX_GROWTH, over_long, verify_split +from mimem.ir import Block, BlockKind, BlockRole, Document, SourceMeta, TriageAction +from mimem.ir.models import TriageDecision +from mimem.llm import NullClient, ScriptedClient + +LONG = ( + "The reversible capacity of the electrodes cycled in the fluoroethylene carbonate " + "electrolyte was 3867.3 mAh/g after the first cycle, which is higher than the 252 mAh/g " + "measured for the electrodes cycled in the ethylene carbonate electrolyte under otherwise " + "identical conditions at a rate of 0.1 C." +) + +SPLIT = [ + "The electrodes cycled in the fluoroethylene carbonate electrolyte had a reversible " + "capacity of 3867.3 mAh/g after the first cycle.", + "That is higher than the 252 mAh/g measured for the electrodes cycled in the ethylene " + "carbonate electrolyte at a rate of 0.1 C.", +] + + +def _doc(text: str = LONG) -> Document: + from mimem.clean import sentence_spans + + block = Block( + id="b1", + kind=BlockKind.PARAGRAPH, + role=BlockRole.BODY, + text=text, + page=1, + order=0, + sentences=sentence_spans(text), + triage=TriageDecision(action=TriageAction.KEEP, rule="COH-04", reason="carries it"), + ) + return Document(id="d", source=SourceMeta(format="pdf"), blocks=[block]) + + +# -- what counts as too long -------------------------------------------------------------------- + + +def test_length_is_counted_on_what_the_listener_hears() -> None: + """ "298 K" is two words written and four spoken, and the listener gets the spoken one. + + Selecting on the written form misses the sentences that are long *because of what they + state*, which are the ones a listener most needs split. + """ + profile = load_profile("study") + written = "The cell was held at 298 K and 3867.3 mAh/g was measured at 0.1 C afterwards." + assert len(written.split()) < 20 + found = over_long(_doc(written), profile, Listener(), cap=20) + assert found and found[0].spoken_words > 20 + + +def test_a_short_sentence_is_left_alone() -> None: + found = over_long(_doc("The cell failed early."), load_profile("study"), Listener()) + assert found == [] + + +def test_a_table_has_no_sentences_to_split() -> None: + """Its caption speaks for it, and its words are not prose.""" + doc = _doc() + doc.blocks[0].kind = BlockKind.TABLE + assert over_long(doc, load_profile("study"), Listener()) == [] + + +def test_a_dropped_block_is_not_worth_splitting() -> None: + doc = _doc() + doc.blocks[0].triage = TriageDecision( + action=TriageAction.DROP, rule="COH-01", reason="back matter" + ) + assert over_long(doc, load_profile("study"), Listener()) == [] + + +# -- the check that separates a split from a summary --------------------------------------------- + + +def test_a_faithful_split_passes() -> None: + assert verify_split(LONG, SPLIT, cap=25) == [] + + +def test_a_summary_is_caught_by_the_numbers_it_drops() -> None: + """The check no other task in stage 6 can fail, and the reason this one is safe. + + Everything else here writes new text and can only be asked whether it invented something. A + summary invents nothing -- it just quietly stops saying four of the paper's measurements. + """ + summary = [ + "The capacity was much higher in the fluoroethylene carbonate electrolyte.", + "The ethylene carbonate electrolyte performed worse.", + ] + dropped = {f.value for f in verify_split(LONG, summary, cap=25) if f.kind == "number"} + assert {"3867.3", "252", "0.1"} <= dropped + + +def test_a_flipped_comparison_is_caught() -> None: + """ "higher" for "lower" is the worst thing this system can produce, and it reads perfectly.""" + flipped = [SPLIT[0], SPLIT[1].replace("higher", "lower")] + findings = verify_split(LONG, flipped, cap=25) + assert any(f.kind == "direction" for f in findings) + + +def test_an_invented_number_is_caught() -> None: + invented = [SPLIT[0], SPLIT[1] + " A third cell reached 99 mAh/g."] + assert any(f.value == "99" for f in verify_split(LONG, invented, cap=25)) + + +def test_one_sentence_back_is_not_a_split() -> None: + assert any(f.kind == "split" for f in verify_split(LONG, [LONG], cap=25)) + + +def test_sentences_still_over_the_cap_have_not_done_the_job() -> None: + findings = verify_split(LONG, [LONG[:200], LONG[200:]], cap=5) + assert any("cap" in f.message for f in findings) + + +def test_an_expansion_is_not_a_split() -> None: + """Growth is expected -- a repeated subject costs words -- and half as much again is not.""" + padded = [SPLIT[0], SPLIT[1], " ".join(["and the cells were then rested"] * 12)] + findings = verify_split(LONG, padded, cap=60) + assert any("expansion" in f.message for f in findings) + assert MAX_GROWTH < 2 + + +# -- the stage ------------------------------------------------------------------------------------ + + +def _elaborate(doc: Document, answers: dict[str, list[dict[str, object]]]): + registry_profile = load_profile("study") + from mimem.concepts import build as build_registry + + return elaborate( + doc, + build_registry(doc, Listener()), + registry_profile, + Listener(), + ScriptedClient(answers), + ) + + +def test_the_split_is_stored_beside_the_sentence_and_the_source_is_not_touched() -> None: + doc = _doc() + before = doc.blocks[0].text + report = _elaborate(doc, {"split": [{"sentences": SPLIT}]}) + + block = doc.blocks[0] + assert block.text == before, "the paper's own words were edited" + assert block.rewrites, "the split was not stored" + start, end = block.sentences[0] + assert block.spoken_text(start, end) == " ".join(SPLIT) + assert report.succeeded.get("split") == 1 + + +def test_a_summary_is_rejected_and_the_source_sentence_survives() -> None: + doc = _doc() + summary = ["The capacity was higher with the additive.", "Both cells were tested."] + report = _elaborate(doc, {"split": [{"sentences": summary}]}) + + assert doc.blocks[0].rewrites == {} + assert report.rejected, "a summary passed the gate" + assert any(d.task == "split" for d in report.degraded) + + +def test_no_model_means_the_sentence_is_left_long() -> None: + """The control path. The linter goes on reporting it, which is the honest outcome.""" + doc = _doc() + from mimem.concepts import build as build_registry + + report = elaborate( + doc, build_registry(doc, Listener()), load_profile("study"), Listener(), NullClient() + ) + assert doc.blocks[0].rewrites == {} + assert report.no_provider + + +def test_the_planner_speaks_the_split_and_still_points_at_the_source() -> None: + """Rule GRD-02 checks a beat against its spans, so the span has to stay on what was written.""" + from mimem.concepts import build as build_registry + from mimem.ir import BeatType + from mimem.plan import plan + + doc = _doc() + _elaborate(doc, {"split": [{"sentences": SPLIT}]}) + script = plan(doc, build_registry(doc, Listener()), load_profile("study"), Listener()) + + exposition = [b for b in script.beats() if b.type is BeatType.EXPOSITION] + assert exposition, "the fixture produced no exposition" + said = " ".join(b.text for b in exposition) + assert "That is higher than" in said, "the split was not spoken" + for beat in exposition: + for span in beat.spans: + assert doc.text_of(span) in doc.blocks[0].text + + +@pytest.mark.parametrize("cap", [25]) +def test_the_split_cap_leaves_room_for_the_verbalizer(cap: int) -> None: + """Asked for twenty-five, checked at thirty-five, because numbers are not spoken yet. + + A sentence split to exactly the linter's cap in *written* words is over it by the time its + numbers have been said, which would spend a model call and change nothing. + """ + from mimem.elaborate.run import SPLIT_CAP + from mimem.lint.rules import SentenceLength + + assert SPLIT_CAP == cap < SentenceLength.HARD_CAP