From def75111b63dc749a7e0946c2e95c52fbed31238 Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:25:53 +0000 Subject: [PATCH 1/2] Convert USFM versification before updating from rows Port sillsdev/machine#472 and the follow-up fixes from sillsdev/machine#521. ParatextProjectTextUpdaterBase.update_usfm now converts the source USFM to the rows' versification when it differs from the project's, using the new ConvertUsfmVersificationHandler. The test project settings now default to English versification, matching ScriptureRef, so existing update tests do not trigger a conversion. Closes #369 Closes #382 Co-Authored-By: Claude Sonnet 5.5 Co-authored-by: Damien Daspit <3261883+ddaspit@users.noreply.github.com> --- machine/corpora/__init__.py | 2 + .../convert_usfm_versification_handler.py | 225 ++++++ .../paratext_project_text_updater_base.py | 13 +- machine/corpora/update_usfm_parser_handler.py | 13 +- ...test_convert_usfm_versification_handler.py | 763 ++++++++++++++++++ .../test_update_usfm_parser_handler.py | 31 + .../memory_paratext_project_file_handler.py | 4 +- 7 files changed, 1043 insertions(+), 8 deletions(-) create mode 100644 machine/corpora/convert_usfm_versification_handler.py create mode 100644 tests/corpora/test_convert_usfm_versification_handler.py diff --git a/machine/corpora/__init__.py b/machine/corpora/__init__.py index 20f3a1da..86372492 100644 --- a/machine/corpora/__init__.py +++ b/machine/corpora/__init__.py @@ -2,6 +2,7 @@ from .alignment_collection import AlignmentCollection from .alignment_corpus import AlignmentCorpus from .alignment_row import AlignmentRow +from .convert_usfm_versification_handler import ConvertUsfmVersificationHandler from .corpora_utils import batch from .corpus import Corpus from .dbl_bundle_text_corpus import DblBundleTextCorpus @@ -99,6 +100,7 @@ "AlignmentCorpus", "AlignmentRow", "batch", + "ConvertUsfmVersificationHandler", "Corpus", "create_versification_ref_corpus", "TextRowContentType", diff --git a/machine/corpora/convert_usfm_versification_handler.py b/machine/corpora/convert_usfm_versification_handler.py new file mode 100644 index 00000000..a09b0613 --- /dev/null +++ b/machine/corpora/convert_usfm_versification_handler.py @@ -0,0 +1,225 @@ +from typing import Dict, List, Optional, Sequence, Tuple + +import regex as re + +from ..scripture.verse_ref import VerseRef, Versification +from .scripture_ref_usfm_parser_handler_base import ScriptureRefUsfmParserHandlerBase +from .usfm_parser_state import UsfmParserState +from .usfm_stylesheet import UsfmStylesheet +from .usfm_token import UsfmToken, UsfmTokenType +from .usfm_tokenizer import UsfmTokenizer + +_TRAILING_PARAGRAPH_MARKER_PATTERNS = re.compile(r"^(?:mte?\d*|ms\d*|sd?\d*|mr|sr|sp|d|r)$") + + +def _change_versification(verse_ref: VerseRef, versification: Versification) -> VerseRef: + new_verse_ref = verse_ref.copy() + new_verse_ref.change_versification(versification) + return new_verse_ref + + +def _new_nb_token() -> UsfmToken: + return UsfmToken(UsfmTokenType.PARAGRAPH, "nb", "", "", "") + + +class ConvertUsfmVersificationHandler(ScriptureRefUsfmParserHandlerBase): + def __init__(self, target_versification: Versification) -> None: + super().__init__() + self._tokens: List[UsfmToken] = [] + self._trailing_verse_tokens: List[Tuple[int, UsfmToken]] = [] + self._prev_verse_ref = VerseRef() + self._verse_boundary = 0 + self._target_versification = target_versification + self._insert_chapter_index = -1 + self._skip = False + + @property + def tokens(self) -> Sequence[UsfmToken]: + return self._tokens + + def chapter( + self, + state: UsfmParserState, + number: str, + marker: str, + alt_number: Optional[str], + pub_number: Optional[str], + ) -> None: + super().chapter(state, number, marker, alt_number, pub_number) + self._process_tokens(state) + vr = state.verse_ref.copy() + # The versification of verse 0 cannot properly be changed + vr.verse = "1" + if not self._prev_verse_ref.is_default and ( + _change_versification(vr, self._target_versification).book != self._prev_verse_ref.book + or vr.chapter_num == -1 + ): + self._skip = True + self._insert_chapter_index = len(self._tokens) + + def verse( + self, + state: UsfmParserState, + number: str, + marker: str, + alt_number: Optional[str], + pub_number: Optional[str], + ) -> None: + super().verse(state, number, marker, alt_number, pub_number) + + verse_ref = state.verse_ref + + self._process_tokens(state) + + verse_refs = [_change_versification(vr, self._target_versification) for vr in state.verse_ref.all_verses()] + + if ( + self._prev_verse_ref.is_default + or ( + verse_refs[0].book_num == self._prev_verse_ref.book_num + and verse_refs[0].chapter_num != self._prev_verse_ref.chapter_num + ) + ) and verse_refs[0].chapter_num != -1: + new_chapter_token = UsfmToken(UsfmTokenType.CHAPTER, "c", "", "", verse_refs[0].chapter) + + if self._insert_chapter_index == -1: + chapter_index = len(self._tokens) + self._tokens.append(new_chapter_token) + trailing_at_chapter = [t for i, t in self._trailing_verse_tokens if i == chapter_index] + if len(trailing_at_chapter) == 0: + # The chapter break falls mid-paragraph, so the paragraph continues across it. + self._tokens.append(_new_nb_token()) + else: + # The trailing markers follow the new chapter and break the paragraph. If they do not open a + # paragraph of their own, the verse still needs one. + last_paragraph = next( + (t for t in reversed(trailing_at_chapter) if t.type == UsfmTokenType.PARAGRAPH), None + ) + if last_paragraph is None or _TRAILING_PARAGRAPH_MARKER_PATTERNS.match(last_paragraph.marker or ""): + self._trailing_verse_tokens.append((chapter_index, _new_nb_token())) + self._trailing_verse_tokens = [ + (i + 1 if i == chapter_index else i, t) for i, t in self._trailing_verse_tokens + ] + else: + self._tokens.insert(self._insert_chapter_index, new_chapter_token) + self._trailing_verse_tokens = [ + (i + 1 if i >= self._insert_chapter_index else i, t) for i, t in self._trailing_verse_tokens + ] + + added_verse_text = False + duplicate_verse = False + + start: Optional[str] = None + for vr in verse_refs: + if (not self._prev_verse_ref.is_default and vr.book != self._prev_verse_ref.book) or vr.chapter_num == -1: + continue + if start is not None: + end = "-" + self._prev_verse_ref.verse if start != self._prev_verse_ref.verse else "" + if self._prev_verse_ref.book_num == vr.book_num and self._prev_verse_ref.chapter_num != vr.chapter_num: + self._add_trailing_tokens() + if not duplicate_verse: + self._tokens.append(UsfmToken(UsfmTokenType.VERSE, "v", "", "", start + end)) + added_verse_text = self._add_next_text_token(state, added_verse_text) + self._tokens.append(UsfmToken(UsfmTokenType.CHAPTER, "c", "", "", vr.chapter)) + self._tokens.append(_new_nb_token()) + start = vr.verse + duplicate_verse = False + self._prev_verse_ref = vr + elif self._prev_verse_ref.verse_num + 1 != vr.verse_num: + self._add_trailing_tokens() + if not duplicate_verse: + self._tokens.append(UsfmToken(UsfmTokenType.VERSE, "v", "", "", start + end)) + added_verse_text = self._add_next_text_token(state, added_verse_text) + start = vr.verse + duplicate_verse = False + self._prev_verse_ref = vr + else: + # The duplicated verse was already written, so the range starts after it. + if duplicate_verse: + start = vr.verse + duplicate_verse = False + self._prev_verse_ref = vr + else: + start = vr.verse + duplicate_verse = vr == self._prev_verse_ref + self._prev_verse_ref = vr + verse_ref = vr + + if start is not None: + self._add_trailing_tokens() + end = "-" + self._prev_verse_ref.verse if start != self._prev_verse_ref.verse else "" + if not duplicate_verse: + self._tokens.append(UsfmToken(UsfmTokenType.VERSE, "v", "", "", start + end)) + self._skip = False + self._insert_chapter_index = -1 + self._prev_verse_ref = verse_ref + else: + self._skip = True + # Markers that introduce a dropped verse would otherwise be flushed at the next kept verse. + self._trailing_verse_tokens.clear() + + def end_usfm(self, state: UsfmParserState) -> None: + super().end_usfm(state) + self._process_tokens(state) + token = state.token + if ( + not self._skip + and token is not None + and not (token.type == UsfmTokenType.CHAPTER or token.type == UsfmTokenType.VERSE) + ): + self._tokens.append(token) + + def get_usfm(self, stylesheet: UsfmStylesheet) -> str: + tokenizer = UsfmTokenizer(stylesheet) + return tokenizer.detokenize(self._tokens) + + def _add_next_text_token(self, state: UsfmParserState, added_verse_text: bool) -> bool: + if not added_verse_text and state.index + 1 < len(state.tokens): + next_token = state.tokens[state.index + 1] + if next_token.type == UsfmTokenType.TEXT: + self._tokens.append(next_token) + self._verse_boundary += 1 + return True + return added_verse_text + + def _process_tokens(self, state: UsfmParserState) -> None: + offset = 0 + in_preserved_paragraph = False + while self._verse_boundary + offset < state.index: + token = state.tokens[self._verse_boundary + offset] + if _is_preserved_trailing_paragraph_marker(state.tokens, self._verse_boundary + offset): + in_preserved_paragraph = True + elif in_preserved_paragraph: + in_preserved_paragraph = token.type != UsfmTokenType.PARAGRAPH + else: + in_preserved_paragraph = False + + if in_preserved_paragraph: + self._trailing_verse_tokens.append((len(self._tokens), token)) + elif not self._skip: + self._tokens.append(token) + + offset += 1 + self._verse_boundary = state.index + 1 + + def _add_trailing_tokens(self) -> None: + grouped: Dict[int, List[UsfmToken]] = {} + for index, token in self._trailing_verse_tokens: + grouped.setdefault(index, []).append(token) + for index in sorted(grouped, reverse=True): + self._tokens[index:index] = grouped[index] + self._trailing_verse_tokens.clear() + + +def _is_preserved_trailing_paragraph_marker(tokens: Sequence[UsfmToken], index: int) -> bool: + token = tokens[index] + next_token = tokens[index + 1] if index + 1 < len(tokens) else None + return ( + token.type == UsfmTokenType.PARAGRAPH + and next_token is not None + and (next_token.type == UsfmTokenType.VERSE or _is_preserved_trailing_paragraph_marker(tokens, index + 1)) + ) or ( + token.type == UsfmTokenType.PARAGRAPH + and token.marker is not None + and _TRAILING_PARAGRAPH_MARKER_PATTERNS.match(token.marker) is not None + ) diff --git a/machine/corpora/paratext_project_text_updater_base.py b/machine/corpora/paratext_project_text_updater_base.py index 188f31c7..175ea544 100644 --- a/machine/corpora/paratext_project_text_updater_base.py +++ b/machine/corpora/paratext_project_text_updater_base.py @@ -2,11 +2,12 @@ from typing import Callable, Iterable, List, Optional, Sequence, Tuple, Union from ..utils.string_utils import parse_integer +from .convert_usfm_versification_handler import ConvertUsfmVersificationHandler from .paratext_project_file_handler import ParatextProjectFileHandler from .paratext_project_settings import ParatextProjectSettings from .paratext_project_settings_parser_base import ParatextProjectSettingsParserBase from .update_usfm_behavior import UpdateUsfmMarkerBehavior, UpdateUsfmTextBehavior -from .update_usfm_parser_handler import UpdateUsfmParserHandler, UpdateUsfmRow +from .update_usfm_parser_handler import UpdateUsfmParserHandler, UpdateUsfmRow, get_rows_versification from .usfm_parser import parse_usfm from .usfm_token import UsfmTokenType from .usfm_tokenizer import UsfmToken, UsfmTokenizer @@ -63,7 +64,15 @@ def update_usfm( tokenizer = UsfmTokenizer(self._settings.stylesheet) tokens = tokenizer.tokenize(usfm) tokens = filter_tokens_by_chapter(tokens, chapters) - parse_usfm(tokens, handler, self._settings.stylesheet, self._settings.versification) + + rows_versification = get_rows_versification(rows) + parse_versification = self._settings.versification + if rows_versification != self._settings.versification: + converter = ConvertUsfmVersificationHandler(rows_versification) + parse_usfm(tokens, converter, self._settings.stylesheet, self._settings.versification) + tokens = converter.tokens + parse_versification = rows_versification + parse_usfm(tokens, handler, self._settings.stylesheet, parse_versification) return handler.get_usfm(self._settings.stylesheet) except Exception as e: error_message = ( diff --git a/machine/corpora/update_usfm_parser_handler.py b/machine/corpora/update_usfm_parser_handler.py index 87689d8a..b3642825 100644 --- a/machine/corpora/update_usfm_parser_handler.py +++ b/machine/corpora/update_usfm_parser_handler.py @@ -31,6 +31,14 @@ def _sanitize_verse_data(verse_data: str) -> str: return verse_data.replace("\u200F", "") +def get_rows_versification(rows: Optional[Sequence[UpdateUsfmRow]]) -> Versification: + if rows is not None: + for row in rows: + if len(row.refs) > 0: + return row.refs[0].versification + return Versification.get_builtin("English") + + class UpdateUsfmParserHandler(ScriptureRefUsfmParserHandlerBase): def __init__( self, @@ -52,10 +60,7 @@ def __init__( self._verse_row_index = 0 self._verse_rows_map: Dict[VerseRef, List[_RowInfo]] = {} self._verse_rows_ref = VerseRef() - if len(self._rows) > 0: - self._update_rows_versification: Versification = self._rows[0].refs[0].versification - else: - self._update_rows_versification = Versification.get_builtin("English") + self._update_rows_versification: Versification = get_rows_versification(self._rows) self._tokens: List[UsfmToken] = [] self._updated_text: List[UsfmToken] = [] self._update_block_stack: list[UsfmUpdateBlock] = [] diff --git a/tests/corpora/test_convert_usfm_versification_handler.py b/tests/corpora/test_convert_usfm_versification_handler.py new file mode 100644 index 00000000..60903ab0 --- /dev/null +++ b/tests/corpora/test_convert_usfm_versification_handler.py @@ -0,0 +1,763 @@ +from testutils.memory_paratext_project_file_handler import DefaultParatextProjectSettings + +from machine.corpora import ConvertUsfmVersificationHandler, UsfmTokenizer, parse_usfm +from machine.scripture import ( + ENGLISH_VERSIFICATION, + ORIGINAL_VERSIFICATION, + RUSSIAN_ORTHODOX_VERSIFICATION, + Versification, +) + + +def test_one_fewer_chapter() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\p +\s1 Section +\v 1-5 +\v 6 Text +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\p +\s1 Section +\v 19-23 +\v 24 Text +""" + assert_usfm_equals(target, result) + + +def test_one_more_chapter() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-23 +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_one_fewer_book() -> None: + # Russian Orthodox vs. Original + # PSA 151:1-7 = PS2 1:1-7 + usfm = r"""\id PSA - Test +\h Psalms +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +\c 151 +\p +\v 1-7 More lines +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id PSA - Test +\h Psalms +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +""" + assert_usfm_equals(target, result) + + +def test_one_more_book() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\h Daniel +\c 3 +\p +\v 1-23 Text 1 +\v 24-90 Text 2 +\p More text 2 +\v 91-100 Text 3 +\c 4 +\p +\v 1 Text 4 +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\h Daniel +\c 3 +\p +\v 1-23 Text 1 +\v 24-33 Text 3 +\c 4 +\p +\v 1 Text 4 +""" + assert_usfm_equals(target, result) + + +def test_back_one_verse_to_previous_chapter() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22 +\v 23 +\c 9 +\p +\v 1 +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\nb +\v 1 +\p +\v 2 +""" + assert_usfm_equals(target, result) + + +def test_forward_one_verse_to_next_chapter() -> None: + # Original vs. English + # ISA 8:23 = ISA 9:1 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\p +\v 1 +\v 2 +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\p +\v 23 +\c 9 +\nb +\v 1 +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22-23 +\c 9 +\p +\v 1 +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\nb +\v 1 +\p +\v 2 +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range_cross_book() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-22 +\v 23-89 +\v 90-100 +\c 4 +\p +\v 1 +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-22 +\v 23 +\v 24-33 +\c 4 +\p +\v 1 +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range_cross_book_within_single_range() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-100 +\c 4 +\p +\v 1 +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-33 +\c 4 +\p +\v 1 +""" + assert_usfm_equals(target, result) + + +def test_heading_introducing_kept_verse_is_preserved() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-90 Dropped text +\s1 \nd Section\nd* +\p +\v 91-100 More text +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\s1 \nd Section\nd* +\p +\v 24-33 More text +""" + assert_usfm_equals(target, result) + + +def test_heading_introducing_dropped_verse_is_dropped() -> None: + # Russian Orthodox vs. Original + # PSA 151:1-7 = PS2 1:1-7 + usfm = r"""\id PSA - Test +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +\c 151 +\s1 \nd Section\nd* +\p +\v 1-7 More lines +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id PSA - Test +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +""" + assert_usfm_equals(target, result) + + +def test_paragraph_introducing_dropped_verse_is_dropped() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-50 Dropped text +\q1 +\v 51-90 More dropped text +\p +\v 91-100 More text +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\p +\v 24-33 More text +""" + assert_usfm_equals(target, result) + + +def test_drop_verse_text() -> None: + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-90 Dropped text +\v 91-100 More text +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-33 More text +""" + assert_usfm_equals(target, result) + + +def test_chapter_marker_is_followed_by_paragraph_marker() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL - Test +\c 3 +\p +\v 1-18 Text +\v 19-23 More text +\v 24 Last text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL - Test +\c 3 +\p +\v 1-18 Text +\c 4 +\nb +\v 1-5 More text +\v 6 Last text +""" + assert_usfm_equals(target, result) + + +def test_chapter_marker_is_followed_by_paragraph_marker_cross_chapter_verse_range() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22-23 +\c 9 +\p +\v 1 +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\nb +\v 1 +\p +\v 2 +""" + assert_usfm_equals(target, result) + + +def test_chapter_marker_is_followed_by_paragraph_marker_heading_opens_paragraph() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL - Test +\c 3 +\p +\v 18 Text +\s1 Section +\p +\v 19-23 More text +\v 24 Last text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL - Test +\c 3 +\p +\v 18 Text +\c 4 +\s1 Section +\p +\v 1-5 More text +\v 6 Last text +""" + assert_usfm_equals(target, result) + + +def test_heading_after_chapter_label_keeps_marker_content() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22 Text +\c 9 +\cl Chapter Nine +\s1 Section +\p +\v 1 Nine one +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 Text +\c 9 +\cl Chapter Nine +\s1 Section +\p +\v 2 Nine one +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range_text_stays_with_first_verse() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22-23 Verse twenty-two and twenty-three text +\c 9 +\p +\v 1 Chapter nine verse one text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 Verse twenty-two and twenty-three text +\c 9 +\nb +\v 1 +\p +\v 2 Chapter nine verse one text +""" + assert_usfm_equals(target, result) + + +def test_ignore_invalid_chapter() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2@ +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-23 +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + + # Strip out invalid chapters since we can't reliably convert them + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_ignore_invalid_verse() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1@ Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-23 +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + + # Just pass invalid verses through to target + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1@ Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_missing_verse_in_range() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-21,23 Text +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-3 Text +\v 5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_same_source_and_target_versification() -> None: + usfm = r"""\id MAT - Test +\h Matthew +\mt Matthew +\ip An introduction to Matthew\fe + \ft This is an endnote.\fe* +\p \rq MAT 1\rq* Here is another paragraph. +\p and with a \w keyword|a special concept\w* in it. +\p and a \weirdtaglookingthing that is not an actual tag. +\c 1 +\s Chapter One +\v 1 Chapter \pn one\+pro WON\+pro*\pn*, verse one.\f + \fr 1:1: \ft This is a footnote for v1.\f* +\li1 +\v 2 \bd C\bd*hapter one, +\li2 verse\f + \fr 1:2: \ft This is a footnote for v2.\f* two. +\v 3 Chapter one \w*, +\li2 verse three. +\v 4 Chapter one with odd whitespace, +\li2 verse four, +\v 5 Chapter one, +\li2 verse \fig Figure 1|src="image1.png" size="col" ref="1:5"\fig* five. +\v 6 Verse 6 content. +\v 7 +\v 8 +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ENGLISH_VERSIFICATION) + assert_usfm_equals(target, usfm) + + +def test_preceding_headings_not_moved() -> None: + # English vs. Original + # JOL 2:27-28 = JOL 2:27-3:1 + usfm = r"""\id JOL +\c 2 +\v 27 Then you will know that I am present in Israel +\q2 and that I am the LORD your God, +\q2 and there is no other. +\q1 My people will never again +\q2 be put to shame. +\s1 I Will Pour Out My Spirit +\r (Acts 2:14–36) +\q1 +\v 28 And afterward, I will pour out My Spirit on all people. +\q2 Your sons and daughters will prophesy, +\q1 your old men will dream dreams, +\q2 your young men will see visions. +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id JOL +\c 2 +\v 27 Then you will know that I am present in Israel +\q2 and that I am the LORD your God, +\q2 and there is no other. +\q1 My people will never again +\q2 be put to shame. +\c 3 +\s1 I Will Pour Out My Spirit +\r (Acts 2:14–36) +\q1 +\v 1 And afterward, I will pour out My Spirit on all people. +\q2 Your sons and daughters will prophesy, +\q1 your old men will dream dreams, +\q2 your young men will see visions. +""" + assert_usfm_equals(target, result) + + +def test_merged_verses() -> None: + # Original vs. English + # PSA 51:1-3 = PSA 51:0-1 + usfm = r"""\id PSA +\c 51 +\s1 Create in Me a Clean Heart, O God +\r (2 Samuel 12:1–12) +\p +\v 1 For the choirmaster. A Psalm of David. +\v 2 When Nathan the prophet came to him after his adultery with Bathsheba. +\b +\q1 +\v 3 Have mercy on me, O God, +\q2 according to Your loving devotion; +\q1 according to Your great compassion, +\q2 blot out my transgressions. +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id PSA +\c 51 +\s1 Create in Me a Clean Heart, O God +\r (2 Samuel 12:1–12) +\p +\v 0 For the choirmaster. A Psalm of David. When Nathan the prophet came to him after his adultery with Bathsheba. +\b +\q1 +\v 1 Have mercy on me, O God, +\q2 according to Your loving devotion; +\q1 according to Your great compassion, +\q2 blot out my transgressions. +""" + assert_usfm_equals(target, result) + + +def test_merged_verses_range_extends_past_merged_verse() -> None: + # Original vs. Russian Orthodox + # LEV 14:55-56 = LEV 14:55 + usfm = r"""\id LEV +\c 14 +\p +\v 55 for leprosy in a garment or in a house, +\v 56-57 for a swelling, a rash, or a spot, to determine when something is clean or unclean. +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, RUSSIAN_ORTHODOX_VERSIFICATION) + result = r"""\id LEV +\c 14 +\p +\v 55 for leprosy in a garment or in a house, +\v 56 for a swelling, a rash, or a spot, to determine when something is clean or unclean. +""" + assert_usfm_equals(target, result) + + +def convert_usfm(source: str, source_versification: Versification, target_versification: Versification) -> str: + source = source.strip().replace("\r\n", "\n") + "\r\n" + settings = DefaultParatextProjectSettings( + versification=source_versification, file_name_form="MAT", file_name_suffix="" + ) + handler = ConvertUsfmVersificationHandler(target_versification) + tokenizer = UsfmTokenizer(settings.stylesheet) + tokens = tokenizer.tokenize(source) + parse_usfm(tokens, handler, settings.stylesheet, settings.versification) + return handler.get_usfm(settings.stylesheet) + + +def assert_usfm_equals(target: str, truth: str) -> None: + assert target is not None + target_lines = target.split("\n") + truth_lines = truth.split("\n") + for i, truth_line in enumerate(truth_lines): + assert target_lines[i].strip() == truth_line.strip(), f"Line {i}" diff --git a/tests/corpora/test_update_usfm_parser_handler.py b/tests/corpora/test_update_usfm_parser_handler.py index 33e54994..53453dd1 100644 --- a/tests/corpora/test_update_usfm_parser_handler.py +++ b/tests/corpora/test_update_usfm_parser_handler.py @@ -16,6 +16,7 @@ UsfmUpdateBlockElementType, UsfmUpdateBlockHandler, ) +from machine.scripture import ORIGINAL_VERSIFICATION def test_get_usfm_verse_char_style() -> None: @@ -1756,6 +1757,36 @@ def test_unclosed_style_marker_does_not_consume_next_paragraph_marker() -> None: assert_usfm_equals(target, result) +def test_update_usfm_converts_to_rows_versification() -> None: + # Original vs. English + # MAL 4:1 = MAL 3:19 + rows = [ + UpdateUsfmRow(scr_ref("MAL 3:18"), "New 18"), + UpdateUsfmRow(scr_ref("MAL 4:1"), "New 1"), + ] + source = r"""\id MAL - Test +\c 3 +\p +\v 18 Old 18 +\v 19 Old 19 +""" + settings = DefaultParatextProjectSettings( + versification=ORIGINAL_VERSIFICATION, file_name_form="MAT", file_name_suffix="" + ) + updater = MemoryParatextProjectTextUpdater({"MAL": source.replace("\n", "\r\n")}, settings) + target = updater.update_usfm("MAL", rows, text_behavior=UpdateUsfmTextBehavior.PREFER_NEW) + + result = r"""\id MAL - Test +\c 3 +\p +\v 18 New 18 +\c 4 +\nb +\v 1 New 1 +""" + assert_usfm_equals(target, result) + + def scr_ref(*refs: str) -> List[ScriptureRef]: return [ScriptureRef.parse(ref) for ref in refs] diff --git a/tests/testutils/memory_paratext_project_file_handler.py b/tests/testutils/memory_paratext_project_file_handler.py index be6692a5..91aafd13 100644 --- a/tests/testutils/memory_paratext_project_file_handler.py +++ b/tests/testutils/memory_paratext_project_file_handler.py @@ -3,7 +3,7 @@ from machine.corpora import ParatextProjectFileHandler, ParatextProjectSettings, UsfmStylesheet from machine.corpora.paratext_project_text_updater_base import ParatextProjectTextUpdaterBase -from machine.scripture import ORIGINAL_VERSIFICATION, Versification +from machine.scripture import ENGLISH_VERSIFICATION, Versification class MemoryParatextProjectFileHandler(ParatextProjectFileHandler): @@ -61,7 +61,7 @@ def __init__( name, full_name, encoding if encoding is not None else "utf-8", - versification if versification is not None else ORIGINAL_VERSIFICATION, + versification if versification is not None else ENGLISH_VERSIFICATION, stylesheet if stylesheet is not None else UsfmStylesheet("usfm.sty"), file_name_prefix, file_name_form, From aa0d660acf1095e52106f8bb6b2741493330a8a0 Mon Sep 17 00:00:00 2001 From: "claude[bot]" <41898282+claude[bot]@users.noreply.github.com> Date: Thu, 1 Oct 2026 17:37:17 +0000 Subject: [PATCH 2/2] Allow optional chapter numbers in the scripture ref handler base Co-Authored-By: Claude Sonnet 5.5 Co-authored-by: Damien Daspit <3261883+ddaspit@users.noreply.github.com> --- machine/corpora/scripture_ref_usfm_parser_handler_base.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/machine/corpora/scripture_ref_usfm_parser_handler_base.py b/machine/corpora/scripture_ref_usfm_parser_handler_base.py index 30000595..4f8d58fd 100644 --- a/machine/corpora/scripture_ref_usfm_parser_handler_base.py +++ b/machine/corpora/scripture_ref_usfm_parser_handler_base.py @@ -43,7 +43,9 @@ def _current_text_type(self) -> ScriptureTextType: def end_usfm(self, state: UsfmParserState) -> None: self._end_verse_text_wrapper(state) - def chapter(self, state: UsfmParserState, number: str, marker: str, alt_number: str, pub_number: str) -> None: + def chapter( + self, state: UsfmParserState, number: str, marker: str, alt_number: Optional[str], pub_number: Optional[str] + ) -> None: self._end_verse_text_wrapper(state) self._update_verse_ref(state.verse_ref, marker)