diff --git a/src/mimem/plan/planner.py b/src/mimem/plan/planner.py index 63821aa..e632eb3 100644 --- a/src/mimem/plan/planner.py +++ b/src/mimem/plan/planner.py @@ -733,6 +733,22 @@ def _close_prequestions(ctx: _Context, script: Script, prequestion_ids: set[str] ) +#: How many supports to pass over before giving up on a segment prompt. A concept with three +#: sentences about it and two of them already spent as answers elsewhere should still get asked; +#: a concept with nothing unshared left should not be asked twice with one fact. +MAX_SUPPORT_TRIES = 4 + + +def _answered(script: Script) -> set[tuple[str, int, int]]: + """The source sentences already serving as some card's answer (rule RET-06). + + Compared by span rather than by text: two cards sharing a sentence share its location, and + the answers themselves differ by construction -- each opens on a lead carrying its own + subject's name, which is why comparing whole answers reports nothing. + """ + return {(s.block_id, s.char_start, s.char_end) for card in script.cards for s in card.spans} + + def _add_segment_prompts(ctx: _Context, script: Script) -> None: """Rule RET-03: up to one retrieval prompt per segment, where there is one to ask. @@ -750,6 +766,7 @@ def _add_segment_prompts(ctx: _Context, script: Script) -> None: budget = script.budget_seconds running = script.est_seconds + answered = _answered(script) for section in script.sections: for segment in section.segments: if segment.has(BeatType.PROMPT) or running >= budget: @@ -761,9 +778,25 @@ def _add_segment_prompts(ctx: _Context, script: Script) -> None: concept = ctx.registry.concepts.get(concept_id) if concept is None: continue - support = ctx.support.take(concept_id) + # Rule RET-06: not a sentence some other card is already answering with. + # ``SupportPool`` keys its ledger per concept, which is right for exposition -- + # one sentence can reasonably serve two ideas -- and wrong for a question. On + # the corpus this was forty-two warnings: two prompts, one fact, and retrieval + # practice spent without being had. + support = None + for _ in range(MAX_SUPPORT_TRIES): + candidate = ctx.support.take(concept_id) + if candidate is None: + break + span = candidate.span + if (span.block_id, span.char_start, span.char_end) not in answered: + support = candidate + break if support is None: continue + answered.add( + (support.span.block_id, support.span.char_start, support.span.char_end) + ) card = make.make_card(concept, support, segment.section_id, ctx.factory) beats = make.prompt_beats(ctx.factory, card) cost = sum(b.total_seconds for b in beats) diff --git a/tests/unit/test_plan.py b/tests/unit/test_plan.py index 20ba478..7365cc4 100644 --- a/tests/unit/test_plan.py +++ b/tests/unit/test_plan.py @@ -346,3 +346,32 @@ def test_a_fallback_transition_ends_on_a_sentence_boundary(study_profile: Profil factory = BeatFactory(study_profile) for title in ("Available online at www.sciencedirect.com", "Results and discussion", ""): assert section_fallback_transition(factory, title).endswith(".") + + +def test_no_two_cards_answer_with_the_same_sentence(interphase_script: Script) -> None: + """Rule RET-06, and the half of it the planner was not keeping. + + The section-closing cards already avoided sharing, through ``_unshared_support``. The segment + prompts drew through ``SupportPool.take``, whose ledger is keyed per *concept* -- right for + exposition, where one sentence can reasonably serve two ideas, and wrong for a question. So a + sentence already answering one was free to answer another: forty-two warnings across the + twelve-paper stress corpus, each one two prompts and a single fact between them. + """ + spans = [ + (s.block_id, s.char_start, s.char_end) + for card in interphase_script.cards + for s in card.spans + ] + assert len(spans) == len(set(spans)), "two cards are answering with the same sentence" + + +def test_a_segment_prompt_is_dropped_rather_than_answered_with_a_borrowed_sentence( + interphase_script: Script, +) -> None: + """The alternative was to ask anyway. A question whose answer the listener has already been + given as the answer to a different question is retrieval practice spent without being had. + """ + prompts = [b for b in interphase_script.beats() if b.type is BeatType.PROMPT] + assert prompts, "the fixture should still place prompts" + answers = [b for b in interphase_script.beats() if b.type is BeatType.ANSWER] + assert len(answers) == len(prompts) diff --git a/tests/unit/test_script_lint.py b/tests/unit/test_script_lint.py index b237720..04dabd2 100644 --- a/tests/unit/test_script_lint.py +++ b/tests/unit/test_script_lint.py @@ -378,17 +378,19 @@ def _announce_the_wrong_thing(script: Script) -> None: #: This exists so that "known defect" is tracked rather than hidden. The clean-plan test asserts #: these rules *do* fire, so the day the planner is fixed the entry fails and has to be removed. #: A skip would have rotted silently. -KNOWN_FINDINGS: dict[str, str] = { - "RET-06": ( - "a section whose concept has only one usable sentence still shares it; the planner " - "prefers an unshared sentence and falls back rather than losing the section's question" - ), -} +KNOWN_FINDINGS: dict[str, str] = {} #: ``STR-09`` was here and is gone, which is the mechanism working. It was replaced by #: ``SEG-04`` when the evidence turned out to be about broken promises rather than thin text, #: and the planner was then fixed so that ``SEG-04`` passes on the clean plan. An entry that #: stops firing fails this test and has to be removed, which is how it left. +#: +#: ``RET-06`` left the same way. It was here because "a section whose concept has only one +#: usable sentence still shares it; the planner prefers an unshared sentence and falls back +#: rather than losing the section's question" -- which was true of the section-closing cards and +#: said nothing about the segment prompts, where the sharing actually came from. Those drew +#: through ``SupportPool.take``, whose ledger is keyed per concept, so a sentence already +#: answering one question was free to answer another. Forty-two warnings on the stress corpus. @pytest.mark.parametrize(("rule_id", "mutate"), CASES, ids=IDS)