From 1d7f1af30ea9aee5c11e86f335b2fbe561e52181 Mon Sep 17 00:00:00 2001 From: jepegit Date: Wed, 9 Sep 2026 03:37:27 +0200 Subject: [PATCH] Three last things the corpus was still catching MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A marker after more than one closing bracket. Parentheticals nest -- "5 wt% of Super P carbon (Tim-cal) in N-Methylpyrrolidone (Aldrich)).19" closes two of them between the sentence's last letter and its full stop -- and matching a single one was still one short. The brackets are consumed and put back now rather than looked behind, because a lookbehind in Python has to be a fixed width and a run of them is not. A designation that begins where a number ends. A hydrate is written with the ether welded to the stoichiometry, "AlH3·0.25Et2O", and the pattern refused to start on a token with a digit in front of it -- so the plain pass said "zero point two five" and left the ether spelled out as a number. The leftmost match still wins, so a coefficient in front is claimed with its token as before. And the Norwegian company suffix, which is not the English word "as". Written a/?s under re.I alongside "university" and "institute", it matched every front-matter block containing that word -- including one paper's title, "Aluminum hydride as a hydrogen and energy storage material". Triage dropped the title as content-free while the orientation went on saying it, which is what rule STR-01 asks for and what COH-01 then reported as a leak. Two hints either side of it had never matched anything at all: the pattern ends in a word boundary and there is none between "Dept." and the space after it, nor after "Inc." at the end of a phrase. Corpus: two errors across the twelve papers, and ten of them build clean. Co-Authored-By: Claude Opus 5 --- src/mimem/clean/sections.py | 12 ++++++++++- src/mimem/verbalize/citations.py | 15 ++++++++----- src/mimem/verbalize/numbers.py | 8 ++++++- tests/unit/test_clean_text.py | 27 +++++++++++++++++++++++ tests/unit/test_corpus_defects.py | 36 +++++++++++++++++++++++++++++++ 5 files changed, 91 insertions(+), 7 deletions(-) diff --git a/src/mimem/clean/sections.py b/src/mimem/clean/sections.py index 4269287..e238d98 100644 --- a/src/mimem/clean/sections.py +++ b/src/mimem/clean/sections.py @@ -80,9 +80,19 @@ _LEADING_NUMBER = re.compile(r"^\s*(?:\d+(?:\.\d+)*|[IVXLC]+)[.)]?\s+") _EMAIL = re.compile(r"[\w.+-]+@[\w-]+\.[\w.]+") +#: ``A/S`` is the Norwegian limited-company suffix, and it is spelled in capitals. Written +#: ``a/?s`` under ``re.I`` alongside everything else it matched the English word **as**, so any +#: front-matter block containing it was an affiliation -- including one paper's title, "Aluminum +#: hydride *as* a hydrogen and energy storage material". Triage then dropped the title as +#: content-free while the orientation went on saying it, which is what rule ``STR-01`` asks for +#: and what ``COH-01`` then reported as a leak. _AFFILIATION_HINT = re.compile( r"\b(university|universitet|institute|institutt|department|dept\.|laborator|college|" - r"school of|centre|center|academy|hospital|gmbh|inc\.|ltd|a/?s|norway|sweden|denmark)\b", + r"school of|centre|center|academy|hospital|gmbh|ltd|(?-i:A/?S)|" + # No full stop on these two: the pattern ends in a word boundary, and there is none + # between "Dept." and the space after it, so both alternatives never matched anything. + r"dept|inc|" + r"norway|sweden|denmark)\b", re.I, ) #: A reference entry needs *both* a list marker and a bibliographic signal. Requiring only the diff --git a/src/mimem/verbalize/citations.py b/src/mimem/verbalize/citations.py index 9a94785..3cb9a40 100644 --- a/src/mimem/verbalize/citations.py +++ b/src/mimem/verbalize/citations.py @@ -201,11 +201,16 @@ def _attribute(match: re.Match[str]) -> str: #: between them** is not a shape prose has. "Fig. 3" and "Ref. 12" have the space and never #: match; a decimal has a digit before its point and never matches either. #: -#: The optional bracket, comma or quotation mark is for a marker that follows a parenthetical -#: or a quotation: "(LEDC).31", "carbonates),.17" and a simplified "falling cards model".25 all -#: put something between the last letter and the stop. +#: The brackets, commas and quotation marks are for a marker that follows a parenthetical or a +#: quotation: "(LEDC).31", "carbonates),.17" and a simplified "falling cards model".25 all put +#: something between the last letter and the stop. It is a *run*, because parentheticals nest: +#: "in N-Methylpyrrolidone (Aldrich)).19" closes two of them at once, and matching a single one +#: was still one short. +#: +#: They are consumed and put back rather than looked behind, because a lookbehind in Python has +#: to be a fixed width and a run of them is not. _SUPERSCRIPT_AFTER_STOP = re.compile( - rf"(?:(?<=[A-Za-z][.!?])|(?<=[A-Za-z][)\],\"'”’][.!?]))({_RUN})(?![A-Za-z])" + rf"(?P[A-Za-z][)\],\"'”’]*[.!?])(?P{_RUN})(?![A-Za-z])" ) #: Below this many markers, the pattern is more likely to be data than a citation style. @@ -263,7 +268,7 @@ def strip_superscript_citations(text: str) -> str: :data:`MIN_SUPERSCRIPT_EVIDENCE`. A single stray digit after a word is far more likely to be a typo or a variable than a citation, and deleting it would be silent corruption. """ - text = _SUPERSCRIPT_AFTER_STOP.sub("", text) + text = _SUPERSCRIPT_AFTER_STOP.sub(r"\g", text) return _SUPERSCRIPT_AFTER_WORD.sub( lambda m: ( m.group(0) if _is_unit_exponent(text, m) or _is_formula_subscript(text, m) else "" diff --git a/src/mimem/verbalize/numbers.py b/src/mimem/verbalize/numbers.py index 3b5efda..adcfe30 100644 --- a/src/mimem/verbalize/numbers.py +++ b/src/mimem/verbalize/numbers.py @@ -258,7 +258,13 @@ def __init__( # at all and "1.5" reached the audio as a bare number. The plain pass could not # take it either: it will not start on a digit that follows a letter. re.compile( - r"(?(?=\w*[A-Za-z])(?=\w*\d)[A-Za-z0-9]+(?:\.\d+)?)" + # ...or straight after a number, when a letter starts the token. A hydrate + # is written that way -- "AlH3\u00b70.25Et2O" -- and the lookbehind refused + # to begin on "Et2O" because a digit was in front of it, so the plain pass + # said "zero point two five" and left the ether spelled out as a number. + # The leftmost match still wins, so "4AlH3" is claimed whole as before. + r"(?:(?(?=\w*[A-Za-z])(?=\w*\d)[A-Za-z0-9]+(?:\.\d+)?)" r"(?!\w)(?!\.\d)" ), "designation", diff --git a/tests/unit/test_clean_text.py b/tests/unit/test_clean_text.py index b29e267..a708d2c 100644 --- a/tests/unit/test_clean_text.py +++ b/tests/unit/test_clean_text.py @@ -220,3 +220,30 @@ def test_only_headings_are_split() -> None: ], ) assert len(split_stacked_headings(doc).blocks) == 1 + + +def test_the_norwegian_company_suffix_is_not_the_english_word_as() -> None: + """``a/?s`` under ``re.I``, alongside "university" and "institute", matched **as**. + + Any front-matter block containing that word was called an affiliation -- including one + paper's title, "Aluminum hydride *as* a hydrogen and energy storage material". Triage then + dropped the title as content-free while the orientation went on saying it, which is what + rule STR-01 asks for and what COH-01 then reported as a leak. + """ + from mimem.clean.sections import _AFFILIATION_HINT + + assert not _AFFILIATION_HINT.search("Aluminum hydride as a hydrogen storage material") + assert not _AFFILIATION_HINT.search("the capacity as measured after cycling") + assert _AFFILIATION_HINT.search("Elkem AS, Kristiansand") + assert _AFFILIATION_HINT.search("Institutt for energiteknikk, Kjeller") + + +def test_two_hints_that_a_full_stop_had_made_unreachable() -> None: + """The pattern ends in a word boundary, and there is none between "Dept." and the space.""" + from mimem.clean.sections import _AFFILIATION_HINT + + assert _AFFILIATION_HINT.search("Dept. of Chemistry, Uppsala") + assert _AFFILIATION_HINT.search("Elkem Inc., Pittsburgh") + # ...without swallowing the ordinary words they are prefixes of. + assert not _AFFILIATION_HINT.search("we incorporated the binder") + assert not _AFFILIATION_HINT.search("the incident light was filtered") diff --git a/tests/unit/test_corpus_defects.py b/tests/unit/test_corpus_defects.py index edb96de..313045d 100644 --- a/tests/unit/test_corpus_defects.py +++ b/tests/unit/test_corpus_defects.py @@ -506,3 +506,39 @@ def test_a_marker_after_a_quotation_mark() -> None: ) def test_a_designation_may_have_a_decimal_point_in_its_name(written: str, spoken: str) -> None: assert _spoken(written) == spoken + + +# -- the last three ---------------------------------------------------------------------------- + + +def test_a_marker_after_more_than_one_closing_bracket() -> None: + """Parentheticals nest, and matching a single closing bracket was still one short. + + "...5 wt% of Super P carbon (Tim-cal) in N-Methylpyrrolidone (Aldrich)).19 Electrochemical + tests were performed..." -- two brackets close at once between the sentence's last letter + and its full stop. + """ + written = "in N-Methylpyrrolidone (Aldrich)).19 Electrochemical tests were performed" + assert strip_superscript_citations(written) == ( + "in N-Methylpyrrolidone (Aldrich)). Electrochemical tests were performed" + ) + + +@pytest.mark.parametrize( + ("written", "spoken"), + [ + # A hydrate is written with the water or the ether welded to the stoichiometry, and the + # designation pattern refused to begin on a token with a digit in front of it -- so the + # plain pass said "zero point two five" and left the ether spelled out as a number. + ( + "a composition close to AlH3·0.25Et2O and although it is amorphous", + "a composition close to AlH three times zero point two five Et two O " + "and although it is amorphous", + ), + # The leftmost match still wins, so a coefficient in front is claimed with the token. + ("4AlH3 was distilled off", "four AlH three was distilled off"), + ("3LiH and AlCl3 react", "three LiH and AlCl three react"), + ], +) +def test_a_designation_may_begin_where_a_number_ends(written: str, spoken: str) -> None: + assert _spoken(written) == spoken