Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion src/mimem/clean/sections.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,9 +80,19 @@

_LEADING_NUMBER = re.compile(r"^\s*(?:\d+(?:\.\d+)*|[IVXLC]+)[.)]?\s+")
_EMAIL = re.compile(r"[\w.+-]+@[\w-]+\.[\w.]+")
#: ``A/S`` is the Norwegian limited-company suffix, and it is spelled in capitals. Written
#: ``a/?s`` under ``re.I`` alongside everything else it matched the English word **as**, so any
#: front-matter block containing it was an affiliation -- including one paper's title, "Aluminum
#: hydride *as* a hydrogen and energy storage material". Triage then dropped the title as
#: content-free while the orientation went on saying it, which is what rule ``STR-01`` asks for
#: and what ``COH-01`` then reported as a leak.
_AFFILIATION_HINT = re.compile(
r"\b(university|universitet|institute|institutt|department|dept\.|laborator|college|"
r"school of|centre|center|academy|hospital|gmbh|inc\.|ltd|a/?s|norway|sweden|denmark)\b",
r"school of|centre|center|academy|hospital|gmbh|ltd|(?-i:A/?S)|"
# No full stop on these two: the pattern ends in a word boundary, and there is none
# between "Dept." and the space after it, so both alternatives never matched anything.
r"dept|inc|"
r"norway|sweden|denmark)\b",
re.I,
)
#: A reference entry needs *both* a list marker and a bibliographic signal. Requiring only the
Expand Down
15 changes: 10 additions & 5 deletions src/mimem/verbalize/citations.py
Original file line number Diff line number Diff line change
Expand Up @@ -201,11 +201,16 @@ def _attribute(match: re.Match[str]) -> str:
#: between them** is not a shape prose has. "Fig. 3" and "Ref. 12" have the space and never
#: match; a decimal has a digit before its point and never matches either.
#:
#: The optional bracket, comma or quotation mark is for a marker that follows a parenthetical
#: or a quotation: "(LEDC).31", "carbonates),.17" and a simplified "falling cards model".25 all
#: put something between the last letter and the stop.
#: The brackets, commas and quotation marks are for a marker that follows a parenthetical or a
#: quotation: "(LEDC).31", "carbonates),.17" and a simplified "falling cards model".25 all put
#: something between the last letter and the stop. It is a *run*, because parentheticals nest:
#: "in N-Methylpyrrolidone (Aldrich)).19" closes two of them at once, and matching a single one
#: was still one short.
#:
#: They are consumed and put back rather than looked behind, because a lookbehind in Python has
#: to be a fixed width and a run of them is not.
_SUPERSCRIPT_AFTER_STOP = re.compile(
rf"(?:(?<=[A-Za-z][.!?])|(?<=[A-Za-z][)\],\"'”’][.!?]))({_RUN})(?![A-Za-z])"
rf"(?P<before>[A-Za-z][)\],\"'”’]*[.!?])(?P<run>{_RUN})(?![A-Za-z])"
)

#: Below this many markers, the pattern is more likely to be data than a citation style.
Expand Down Expand Up @@ -263,7 +268,7 @@ def strip_superscript_citations(text: str) -> str:
:data:`MIN_SUPERSCRIPT_EVIDENCE`. A single stray digit after a word is far more likely to be
a typo or a variable than a citation, and deleting it would be silent corruption.
"""
text = _SUPERSCRIPT_AFTER_STOP.sub("", text)
text = _SUPERSCRIPT_AFTER_STOP.sub(r"\g<before>", text)
return _SUPERSCRIPT_AFTER_WORD.sub(
lambda m: (
m.group(0) if _is_unit_exponent(text, m) or _is_formula_subscript(text, m) else ""
Expand Down
8 changes: 7 additions & 1 deletion src/mimem/verbalize/numbers.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,7 +258,13 @@ def __init__(
# at all and "1.5" reached the audio as a bare number. The plain pass could not
# take it either: it will not start on a digit that follows a letter.
re.compile(
r"(?<![\w.])(?P<token>(?=\w*[A-Za-z])(?=\w*\d)[A-Za-z0-9]+(?:\.\d+)?)"
# ...or straight after a number, when a letter starts the token. A hydrate
# is written that way -- "AlH3\u00b70.25Et2O" -- and the lookbehind refused
# to begin on "Et2O" because a digit was in front of it, so the plain pass
# said "zero point two five" and left the ether spelled out as a number.
# The leftmost match still wins, so "4AlH3" is claimed whole as before.
r"(?:(?<![\w.])|(?<=\d)(?=[A-Za-z]))"
r"(?P<token>(?=\w*[A-Za-z])(?=\w*\d)[A-Za-z0-9]+(?:\.\d+)?)"
r"(?!\w)(?!\.\d)"
),
"designation",
Expand Down
27 changes: 27 additions & 0 deletions tests/unit/test_clean_text.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,3 +220,30 @@ def test_only_headings_are_split() -> None:
],
)
assert len(split_stacked_headings(doc).blocks) == 1


def test_the_norwegian_company_suffix_is_not_the_english_word_as() -> None:
"""``a/?s`` under ``re.I``, alongside "university" and "institute", matched **as**.

Any front-matter block containing that word was called an affiliation -- including one
paper's title, "Aluminum hydride *as* a hydrogen and energy storage material". Triage then
dropped the title as content-free while the orientation went on saying it, which is what
rule STR-01 asks for and what COH-01 then reported as a leak.
"""
from mimem.clean.sections import _AFFILIATION_HINT

assert not _AFFILIATION_HINT.search("Aluminum hydride as a hydrogen storage material")
assert not _AFFILIATION_HINT.search("the capacity as measured after cycling")
assert _AFFILIATION_HINT.search("Elkem AS, Kristiansand")
assert _AFFILIATION_HINT.search("Institutt for energiteknikk, Kjeller")


def test_two_hints_that_a_full_stop_had_made_unreachable() -> None:
"""The pattern ends in a word boundary, and there is none between "Dept." and the space."""
from mimem.clean.sections import _AFFILIATION_HINT

assert _AFFILIATION_HINT.search("Dept. of Chemistry, Uppsala")
assert _AFFILIATION_HINT.search("Elkem Inc., Pittsburgh")
# ...without swallowing the ordinary words they are prefixes of.
assert not _AFFILIATION_HINT.search("we incorporated the binder")
assert not _AFFILIATION_HINT.search("the incident light was filtered")
36 changes: 36 additions & 0 deletions tests/unit/test_corpus_defects.py
Original file line number Diff line number Diff line change
Expand Up @@ -506,3 +506,39 @@ def test_a_marker_after_a_quotation_mark() -> None:
)
def test_a_designation_may_have_a_decimal_point_in_its_name(written: str, spoken: str) -> None:
assert _spoken(written) == spoken


# -- the last three ----------------------------------------------------------------------------


def test_a_marker_after_more_than_one_closing_bracket() -> None:
"""Parentheticals nest, and matching a single closing bracket was still one short.

"...5 wt% of Super P carbon (Tim-cal) in N-Methylpyrrolidone (Aldrich)).19 Electrochemical
tests were performed..." -- two brackets close at once between the sentence's last letter
and its full stop.
"""
written = "in N-Methylpyrrolidone (Aldrich)).19 Electrochemical tests were performed"
assert strip_superscript_citations(written) == (
"in N-Methylpyrrolidone (Aldrich)). Electrochemical tests were performed"
)


@pytest.mark.parametrize(
("written", "spoken"),
[
# A hydrate is written with the water or the ether welded to the stoichiometry, and the
# designation pattern refused to begin on a token with a digit in front of it -- so the
# plain pass said "zero point two five" and left the ether spelled out as a number.
(
"a composition close to AlH3·0.25Et2O and although it is amorphous",
"a composition close to AlH three times zero point two five Et two O "
"and although it is amorphous",
),
# The leftmost match still wins, so a coefficient in front is claimed with the token.
("4AlH3 was distilled off", "four AlH three was distilled off"),
("3LiH and AlCl3 react", "three LiH and AlCl three react"),
],
)
def test_a_designation_may_begin_where_a_number_ends(written: str, spoken: str) -> None:
assert _spoken(written) == spoken
Loading