diff --git a/machine/corpora/__init__.py b/machine/corpora/__init__.py index 20f3a1da..86372492 100644 --- a/machine/corpora/__init__.py +++ b/machine/corpora/__init__.py @@ -2,6 +2,7 @@ from .alignment_collection import AlignmentCollection from .alignment_corpus import AlignmentCorpus from .alignment_row import AlignmentRow +from .convert_usfm_versification_handler import ConvertUsfmVersificationHandler from .corpora_utils import batch from .corpus import Corpus from .dbl_bundle_text_corpus import DblBundleTextCorpus @@ -99,6 +100,7 @@ "AlignmentCorpus", "AlignmentRow", "batch", + "ConvertUsfmVersificationHandler", "Corpus", "create_versification_ref_corpus", "TextRowContentType", diff --git a/machine/corpora/convert_usfm_versification_handler.py b/machine/corpora/convert_usfm_versification_handler.py new file mode 100644 index 00000000..a09b0613 --- /dev/null +++ b/machine/corpora/convert_usfm_versification_handler.py @@ -0,0 +1,225 @@ +from typing import Dict, List, Optional, Sequence, Tuple + +import regex as re + +from ..scripture.verse_ref import VerseRef, Versification +from .scripture_ref_usfm_parser_handler_base import ScriptureRefUsfmParserHandlerBase +from .usfm_parser_state import UsfmParserState +from .usfm_stylesheet import UsfmStylesheet +from .usfm_token import UsfmToken, UsfmTokenType +from .usfm_tokenizer import UsfmTokenizer + +_TRAILING_PARAGRAPH_MARKER_PATTERNS = re.compile(r"^(?:mte?\d*|ms\d*|sd?\d*|mr|sr|sp|d|r)$") + + +def _change_versification(verse_ref: VerseRef, versification: Versification) -> VerseRef: + new_verse_ref = verse_ref.copy() + new_verse_ref.change_versification(versification) + return new_verse_ref + + +def _new_nb_token() -> UsfmToken: + return UsfmToken(UsfmTokenType.PARAGRAPH, "nb", "", "", "") + + +class ConvertUsfmVersificationHandler(ScriptureRefUsfmParserHandlerBase): + def __init__(self, target_versification: Versification) -> None: + super().__init__() + self._tokens: List[UsfmToken] = [] + self._trailing_verse_tokens: List[Tuple[int, UsfmToken]] = [] + self._prev_verse_ref = VerseRef() + self._verse_boundary = 0 + self._target_versification = target_versification + self._insert_chapter_index = -1 + self._skip = False + + @property + def tokens(self) -> Sequence[UsfmToken]: + return self._tokens + + def chapter( + self, + state: UsfmParserState, + number: str, + marker: str, + alt_number: Optional[str], + pub_number: Optional[str], + ) -> None: + super().chapter(state, number, marker, alt_number, pub_number) + self._process_tokens(state) + vr = state.verse_ref.copy() + # The versification of verse 0 cannot properly be changed + vr.verse = "1" + if not self._prev_verse_ref.is_default and ( + _change_versification(vr, self._target_versification).book != self._prev_verse_ref.book + or vr.chapter_num == -1 + ): + self._skip = True + self._insert_chapter_index = len(self._tokens) + + def verse( + self, + state: UsfmParserState, + number: str, + marker: str, + alt_number: Optional[str], + pub_number: Optional[str], + ) -> None: + super().verse(state, number, marker, alt_number, pub_number) + + verse_ref = state.verse_ref + + self._process_tokens(state) + + verse_refs = [_change_versification(vr, self._target_versification) for vr in state.verse_ref.all_verses()] + + if ( + self._prev_verse_ref.is_default + or ( + verse_refs[0].book_num == self._prev_verse_ref.book_num + and verse_refs[0].chapter_num != self._prev_verse_ref.chapter_num + ) + ) and verse_refs[0].chapter_num != -1: + new_chapter_token = UsfmToken(UsfmTokenType.CHAPTER, "c", "", "", verse_refs[0].chapter) + + if self._insert_chapter_index == -1: + chapter_index = len(self._tokens) + self._tokens.append(new_chapter_token) + trailing_at_chapter = [t for i, t in self._trailing_verse_tokens if i == chapter_index] + if len(trailing_at_chapter) == 0: + # The chapter break falls mid-paragraph, so the paragraph continues across it. + self._tokens.append(_new_nb_token()) + else: + # The trailing markers follow the new chapter and break the paragraph. If they do not open a + # paragraph of their own, the verse still needs one. + last_paragraph = next( + (t for t in reversed(trailing_at_chapter) if t.type == UsfmTokenType.PARAGRAPH), None + ) + if last_paragraph is None or _TRAILING_PARAGRAPH_MARKER_PATTERNS.match(last_paragraph.marker or ""): + self._trailing_verse_tokens.append((chapter_index, _new_nb_token())) + self._trailing_verse_tokens = [ + (i + 1 if i == chapter_index else i, t) for i, t in self._trailing_verse_tokens + ] + else: + self._tokens.insert(self._insert_chapter_index, new_chapter_token) + self._trailing_verse_tokens = [ + (i + 1 if i >= self._insert_chapter_index else i, t) for i, t in self._trailing_verse_tokens + ] + + added_verse_text = False + duplicate_verse = False + + start: Optional[str] = None + for vr in verse_refs: + if (not self._prev_verse_ref.is_default and vr.book != self._prev_verse_ref.book) or vr.chapter_num == -1: + continue + if start is not None: + end = "-" + self._prev_verse_ref.verse if start != self._prev_verse_ref.verse else "" + if self._prev_verse_ref.book_num == vr.book_num and self._prev_verse_ref.chapter_num != vr.chapter_num: + self._add_trailing_tokens() + if not duplicate_verse: + self._tokens.append(UsfmToken(UsfmTokenType.VERSE, "v", "", "", start + end)) + added_verse_text = self._add_next_text_token(state, added_verse_text) + self._tokens.append(UsfmToken(UsfmTokenType.CHAPTER, "c", "", "", vr.chapter)) + self._tokens.append(_new_nb_token()) + start = vr.verse + duplicate_verse = False + self._prev_verse_ref = vr + elif self._prev_verse_ref.verse_num + 1 != vr.verse_num: + self._add_trailing_tokens() + if not duplicate_verse: + self._tokens.append(UsfmToken(UsfmTokenType.VERSE, "v", "", "", start + end)) + added_verse_text = self._add_next_text_token(state, added_verse_text) + start = vr.verse + duplicate_verse = False + self._prev_verse_ref = vr + else: + # The duplicated verse was already written, so the range starts after it. + if duplicate_verse: + start = vr.verse + duplicate_verse = False + self._prev_verse_ref = vr + else: + start = vr.verse + duplicate_verse = vr == self._prev_verse_ref + self._prev_verse_ref = vr + verse_ref = vr + + if start is not None: + self._add_trailing_tokens() + end = "-" + self._prev_verse_ref.verse if start != self._prev_verse_ref.verse else "" + if not duplicate_verse: + self._tokens.append(UsfmToken(UsfmTokenType.VERSE, "v", "", "", start + end)) + self._skip = False + self._insert_chapter_index = -1 + self._prev_verse_ref = verse_ref + else: + self._skip = True + # Markers that introduce a dropped verse would otherwise be flushed at the next kept verse. + self._trailing_verse_tokens.clear() + + def end_usfm(self, state: UsfmParserState) -> None: + super().end_usfm(state) + self._process_tokens(state) + token = state.token + if ( + not self._skip + and token is not None + and not (token.type == UsfmTokenType.CHAPTER or token.type == UsfmTokenType.VERSE) + ): + self._tokens.append(token) + + def get_usfm(self, stylesheet: UsfmStylesheet) -> str: + tokenizer = UsfmTokenizer(stylesheet) + return tokenizer.detokenize(self._tokens) + + def _add_next_text_token(self, state: UsfmParserState, added_verse_text: bool) -> bool: + if not added_verse_text and state.index + 1 < len(state.tokens): + next_token = state.tokens[state.index + 1] + if next_token.type == UsfmTokenType.TEXT: + self._tokens.append(next_token) + self._verse_boundary += 1 + return True + return added_verse_text + + def _process_tokens(self, state: UsfmParserState) -> None: + offset = 0 + in_preserved_paragraph = False + while self._verse_boundary + offset < state.index: + token = state.tokens[self._verse_boundary + offset] + if _is_preserved_trailing_paragraph_marker(state.tokens, self._verse_boundary + offset): + in_preserved_paragraph = True + elif in_preserved_paragraph: + in_preserved_paragraph = token.type != UsfmTokenType.PARAGRAPH + else: + in_preserved_paragraph = False + + if in_preserved_paragraph: + self._trailing_verse_tokens.append((len(self._tokens), token)) + elif not self._skip: + self._tokens.append(token) + + offset += 1 + self._verse_boundary = state.index + 1 + + def _add_trailing_tokens(self) -> None: + grouped: Dict[int, List[UsfmToken]] = {} + for index, token in self._trailing_verse_tokens: + grouped.setdefault(index, []).append(token) + for index in sorted(grouped, reverse=True): + self._tokens[index:index] = grouped[index] + self._trailing_verse_tokens.clear() + + +def _is_preserved_trailing_paragraph_marker(tokens: Sequence[UsfmToken], index: int) -> bool: + token = tokens[index] + next_token = tokens[index + 1] if index + 1 < len(tokens) else None + return ( + token.type == UsfmTokenType.PARAGRAPH + and next_token is not None + and (next_token.type == UsfmTokenType.VERSE or _is_preserved_trailing_paragraph_marker(tokens, index + 1)) + ) or ( + token.type == UsfmTokenType.PARAGRAPH + and token.marker is not None + and _TRAILING_PARAGRAPH_MARKER_PATTERNS.match(token.marker) is not None + ) diff --git a/machine/corpora/paratext_project_text_updater_base.py b/machine/corpora/paratext_project_text_updater_base.py index 188f31c7..175ea544 100644 --- a/machine/corpora/paratext_project_text_updater_base.py +++ b/machine/corpora/paratext_project_text_updater_base.py @@ -2,11 +2,12 @@ from typing import Callable, Iterable, List, Optional, Sequence, Tuple, Union from ..utils.string_utils import parse_integer +from .convert_usfm_versification_handler import ConvertUsfmVersificationHandler from .paratext_project_file_handler import ParatextProjectFileHandler from .paratext_project_settings import ParatextProjectSettings from .paratext_project_settings_parser_base import ParatextProjectSettingsParserBase from .update_usfm_behavior import UpdateUsfmMarkerBehavior, UpdateUsfmTextBehavior -from .update_usfm_parser_handler import UpdateUsfmParserHandler, UpdateUsfmRow +from .update_usfm_parser_handler import UpdateUsfmParserHandler, UpdateUsfmRow, get_rows_versification from .usfm_parser import parse_usfm from .usfm_token import UsfmTokenType from .usfm_tokenizer import UsfmToken, UsfmTokenizer @@ -63,7 +64,15 @@ def update_usfm( tokenizer = UsfmTokenizer(self._settings.stylesheet) tokens = tokenizer.tokenize(usfm) tokens = filter_tokens_by_chapter(tokens, chapters) - parse_usfm(tokens, handler, self._settings.stylesheet, self._settings.versification) + + rows_versification = get_rows_versification(rows) + parse_versification = self._settings.versification + if rows_versification != self._settings.versification: + converter = ConvertUsfmVersificationHandler(rows_versification) + parse_usfm(tokens, converter, self._settings.stylesheet, self._settings.versification) + tokens = converter.tokens + parse_versification = rows_versification + parse_usfm(tokens, handler, self._settings.stylesheet, parse_versification) return handler.get_usfm(self._settings.stylesheet) except Exception as e: error_message = ( diff --git a/machine/corpora/scripture_ref_usfm_parser_handler_base.py b/machine/corpora/scripture_ref_usfm_parser_handler_base.py index 30000595..4f8d58fd 100644 --- a/machine/corpora/scripture_ref_usfm_parser_handler_base.py +++ b/machine/corpora/scripture_ref_usfm_parser_handler_base.py @@ -43,7 +43,9 @@ def _current_text_type(self) -> ScriptureTextType: def end_usfm(self, state: UsfmParserState) -> None: self._end_verse_text_wrapper(state) - def chapter(self, state: UsfmParserState, number: str, marker: str, alt_number: str, pub_number: str) -> None: + def chapter( + self, state: UsfmParserState, number: str, marker: str, alt_number: Optional[str], pub_number: Optional[str] + ) -> None: self._end_verse_text_wrapper(state) self._update_verse_ref(state.verse_ref, marker) diff --git a/machine/corpora/update_usfm_parser_handler.py b/machine/corpora/update_usfm_parser_handler.py index 87689d8a..b3642825 100644 --- a/machine/corpora/update_usfm_parser_handler.py +++ b/machine/corpora/update_usfm_parser_handler.py @@ -31,6 +31,14 @@ def _sanitize_verse_data(verse_data: str) -> str: return verse_data.replace("\u200F", "") +def get_rows_versification(rows: Optional[Sequence[UpdateUsfmRow]]) -> Versification: + if rows is not None: + for row in rows: + if len(row.refs) > 0: + return row.refs[0].versification + return Versification.get_builtin("English") + + class UpdateUsfmParserHandler(ScriptureRefUsfmParserHandlerBase): def __init__( self, @@ -52,10 +60,7 @@ def __init__( self._verse_row_index = 0 self._verse_rows_map: Dict[VerseRef, List[_RowInfo]] = {} self._verse_rows_ref = VerseRef() - if len(self._rows) > 0: - self._update_rows_versification: Versification = self._rows[0].refs[0].versification - else: - self._update_rows_versification = Versification.get_builtin("English") + self._update_rows_versification: Versification = get_rows_versification(self._rows) self._tokens: List[UsfmToken] = [] self._updated_text: List[UsfmToken] = [] self._update_block_stack: list[UsfmUpdateBlock] = [] diff --git a/tests/corpora/test_convert_usfm_versification_handler.py b/tests/corpora/test_convert_usfm_versification_handler.py new file mode 100644 index 00000000..60903ab0 --- /dev/null +++ b/tests/corpora/test_convert_usfm_versification_handler.py @@ -0,0 +1,763 @@ +from testutils.memory_paratext_project_file_handler import DefaultParatextProjectSettings + +from machine.corpora import ConvertUsfmVersificationHandler, UsfmTokenizer, parse_usfm +from machine.scripture import ( + ENGLISH_VERSIFICATION, + ORIGINAL_VERSIFICATION, + RUSSIAN_ORTHODOX_VERSIFICATION, + Versification, +) + + +def test_one_fewer_chapter() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\p +\s1 Section +\v 1-5 +\v 6 Text +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\p +\s1 Section +\v 19-23 +\v 24 Text +""" + assert_usfm_equals(target, result) + + +def test_one_more_chapter() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-23 +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_one_fewer_book() -> None: + # Russian Orthodox vs. Original + # PSA 151:1-7 = PS2 1:1-7 + usfm = r"""\id PSA - Test +\h Psalms +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +\c 151 +\p +\v 1-7 More lines +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id PSA - Test +\h Psalms +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +""" + assert_usfm_equals(target, result) + + +def test_one_more_book() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\h Daniel +\c 3 +\p +\v 1-23 Text 1 +\v 24-90 Text 2 +\p More text 2 +\v 91-100 Text 3 +\c 4 +\p +\v 1 Text 4 +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\h Daniel +\c 3 +\p +\v 1-23 Text 1 +\v 24-33 Text 3 +\c 4 +\p +\v 1 Text 4 +""" + assert_usfm_equals(target, result) + + +def test_back_one_verse_to_previous_chapter() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22 +\v 23 +\c 9 +\p +\v 1 +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\nb +\v 1 +\p +\v 2 +""" + assert_usfm_equals(target, result) + + +def test_forward_one_verse_to_next_chapter() -> None: + # Original vs. English + # ISA 8:23 = ISA 9:1 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\p +\v 1 +\v 2 +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\p +\v 23 +\c 9 +\nb +\v 1 +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22-23 +\c 9 +\p +\v 1 +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\nb +\v 1 +\p +\v 2 +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range_cross_book() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-22 +\v 23-89 +\v 90-100 +\c 4 +\p +\v 1 +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-22 +\v 23 +\v 24-33 +\c 4 +\p +\v 1 +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range_cross_book_within_single_range() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-100 +\c 4 +\p +\v 1 +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-33 +\c 4 +\p +\v 1 +""" + assert_usfm_equals(target, result) + + +def test_heading_introducing_kept_verse_is_preserved() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-90 Dropped text +\s1 \nd Section\nd* +\p +\v 91-100 More text +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\s1 \nd Section\nd* +\p +\v 24-33 More text +""" + assert_usfm_equals(target, result) + + +def test_heading_introducing_dropped_verse_is_dropped() -> None: + # Russian Orthodox vs. Original + # PSA 151:1-7 = PS2 1:1-7 + usfm = r"""\id PSA - Test +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +\c 151 +\s1 \nd Section\nd* +\p +\v 1-7 More lines +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id PSA - Test +\c 150 +\p +\v 1-5 Lines +\v 6 Line +\q Another line +""" + assert_usfm_equals(target, result) + + +def test_paragraph_introducing_dropped_verse_is_dropped() -> None: + # Russian Orthodox vs. Original + # DAN 3:24-90 = DAG 3:24-90 + # DAN 3:91-100 = DAN 3:24-33 + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-50 Dropped text +\q1 +\v 51-90 More dropped text +\p +\v 91-100 More text +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\p +\v 24-33 More text +""" + assert_usfm_equals(target, result) + + +def test_drop_verse_text() -> None: + usfm = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-90 Dropped text +\v 91-100 More text +""" + target = convert_usfm(usfm, RUSSIAN_ORTHODOX_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id DAN - Test +\c 3 +\p +\v 1-23 Text +\v 24-33 More text +""" + assert_usfm_equals(target, result) + + +def test_chapter_marker_is_followed_by_paragraph_marker() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL - Test +\c 3 +\p +\v 1-18 Text +\v 19-23 More text +\v 24 Last text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL - Test +\c 3 +\p +\v 1-18 Text +\c 4 +\nb +\v 1-5 More text +\v 6 Last text +""" + assert_usfm_equals(target, result) + + +def test_chapter_marker_is_followed_by_paragraph_marker_cross_chapter_verse_range() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22-23 +\c 9 +\p +\v 1 +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 +\c 9 +\nb +\v 1 +\p +\v 2 +""" + assert_usfm_equals(target, result) + + +def test_chapter_marker_is_followed_by_paragraph_marker_heading_opens_paragraph() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL - Test +\c 3 +\p +\v 18 Text +\s1 Section +\p +\v 19-23 More text +\v 24 Last text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL - Test +\c 3 +\p +\v 18 Text +\c 4 +\s1 Section +\p +\v 1-5 More text +\v 6 Last text +""" + assert_usfm_equals(target, result) + + +def test_heading_after_chapter_label_keeps_marker_content() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22 Text +\c 9 +\cl Chapter Nine +\s1 Section +\p +\v 1 Nine one +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 Text +\c 9 +\cl Chapter Nine +\s1 Section +\p +\v 2 Nine one +""" + assert_usfm_equals(target, result) + + +def test_cross_chapter_verse_range_text_stays_with_first_verse() -> None: + # English vs. Original + # ISA 9:1 = ISA 8:23 + usfm = r"""\id ISA - Test +\c 8 +\p +\v 22-23 Verse twenty-two and twenty-three text +\c 9 +\p +\v 1 Chapter nine verse one text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id ISA - Test +\c 8 +\p +\v 22 Verse twenty-two and twenty-three text +\c 9 +\nb +\v 1 +\p +\v 2 Chapter nine verse one text +""" + assert_usfm_equals(target, result) + + +def test_ignore_invalid_chapter() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2@ +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-23 +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + + # Strip out invalid chapters since we can't reliably convert them + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_ignore_invalid_verse() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1@ Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-23 +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + + # Just pass invalid verses through to target + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1@ Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_missing_verse_in_range() -> None: + # English vs. Original + # MAL 4:1-6 = MAL 3:19-24 + usfm = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\v 19-21,23 Text +\v 24 Text +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id MAL +\h Malachi +\c 1 +\s1 Section +\p +\v 1 Text +\v 2-14 +\c 2 +\v 1-17 +\c 3 +\p +\v 1-17 +\v 18 Text \f More text \f* +\c 4 +\nb +\v 1-3 Text +\v 5 +\v 6 Text +""" + assert_usfm_equals(target, result) + + +def test_same_source_and_target_versification() -> None: + usfm = r"""\id MAT - Test +\h Matthew +\mt Matthew +\ip An introduction to Matthew\fe + \ft This is an endnote.\fe* +\p \rq MAT 1\rq* Here is another paragraph. +\p and with a \w keyword|a special concept\w* in it. +\p and a \weirdtaglookingthing that is not an actual tag. +\c 1 +\s Chapter One +\v 1 Chapter \pn one\+pro WON\+pro*\pn*, verse one.\f + \fr 1:1: \ft This is a footnote for v1.\f* +\li1 +\v 2 \bd C\bd*hapter one, +\li2 verse\f + \fr 1:2: \ft This is a footnote for v2.\f* two. +\v 3 Chapter one \w*, +\li2 verse three. +\v 4 Chapter one with odd whitespace, +\li2 verse four, +\v 5 Chapter one, +\li2 verse \fig Figure 1|src="image1.png" size="col" ref="1:5"\fig* five. +\v 6 Verse 6 content. +\v 7 +\v 8 +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ENGLISH_VERSIFICATION) + assert_usfm_equals(target, usfm) + + +def test_preceding_headings_not_moved() -> None: + # English vs. Original + # JOL 2:27-28 = JOL 2:27-3:1 + usfm = r"""\id JOL +\c 2 +\v 27 Then you will know that I am present in Israel +\q2 and that I am the LORD your God, +\q2 and there is no other. +\q1 My people will never again +\q2 be put to shame. +\s1 I Will Pour Out My Spirit +\r (Acts 2:14–36) +\q1 +\v 28 And afterward, I will pour out My Spirit on all people. +\q2 Your sons and daughters will prophesy, +\q1 your old men will dream dreams, +\q2 your young men will see visions. +""" + target = convert_usfm(usfm, ENGLISH_VERSIFICATION, ORIGINAL_VERSIFICATION) + result = r"""\id JOL +\c 2 +\v 27 Then you will know that I am present in Israel +\q2 and that I am the LORD your God, +\q2 and there is no other. +\q1 My people will never again +\q2 be put to shame. +\c 3 +\s1 I Will Pour Out My Spirit +\r (Acts 2:14–36) +\q1 +\v 1 And afterward, I will pour out My Spirit on all people. +\q2 Your sons and daughters will prophesy, +\q1 your old men will dream dreams, +\q2 your young men will see visions. +""" + assert_usfm_equals(target, result) + + +def test_merged_verses() -> None: + # Original vs. English + # PSA 51:1-3 = PSA 51:0-1 + usfm = r"""\id PSA +\c 51 +\s1 Create in Me a Clean Heart, O God +\r (2 Samuel 12:1–12) +\p +\v 1 For the choirmaster. A Psalm of David. +\v 2 When Nathan the prophet came to him after his adultery with Bathsheba. +\b +\q1 +\v 3 Have mercy on me, O God, +\q2 according to Your loving devotion; +\q1 according to Your great compassion, +\q2 blot out my transgressions. +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, ENGLISH_VERSIFICATION) + result = r"""\id PSA +\c 51 +\s1 Create in Me a Clean Heart, O God +\r (2 Samuel 12:1–12) +\p +\v 0 For the choirmaster. A Psalm of David. When Nathan the prophet came to him after his adultery with Bathsheba. +\b +\q1 +\v 1 Have mercy on me, O God, +\q2 according to Your loving devotion; +\q1 according to Your great compassion, +\q2 blot out my transgressions. +""" + assert_usfm_equals(target, result) + + +def test_merged_verses_range_extends_past_merged_verse() -> None: + # Original vs. Russian Orthodox + # LEV 14:55-56 = LEV 14:55 + usfm = r"""\id LEV +\c 14 +\p +\v 55 for leprosy in a garment or in a house, +\v 56-57 for a swelling, a rash, or a spot, to determine when something is clean or unclean. +""" + target = convert_usfm(usfm, ORIGINAL_VERSIFICATION, RUSSIAN_ORTHODOX_VERSIFICATION) + result = r"""\id LEV +\c 14 +\p +\v 55 for leprosy in a garment or in a house, +\v 56 for a swelling, a rash, or a spot, to determine when something is clean or unclean. +""" + assert_usfm_equals(target, result) + + +def convert_usfm(source: str, source_versification: Versification, target_versification: Versification) -> str: + source = source.strip().replace("\r\n", "\n") + "\r\n" + settings = DefaultParatextProjectSettings( + versification=source_versification, file_name_form="MAT", file_name_suffix="" + ) + handler = ConvertUsfmVersificationHandler(target_versification) + tokenizer = UsfmTokenizer(settings.stylesheet) + tokens = tokenizer.tokenize(source) + parse_usfm(tokens, handler, settings.stylesheet, settings.versification) + return handler.get_usfm(settings.stylesheet) + + +def assert_usfm_equals(target: str, truth: str) -> None: + assert target is not None + target_lines = target.split("\n") + truth_lines = truth.split("\n") + for i, truth_line in enumerate(truth_lines): + assert target_lines[i].strip() == truth_line.strip(), f"Line {i}" diff --git a/tests/corpora/test_update_usfm_parser_handler.py b/tests/corpora/test_update_usfm_parser_handler.py index 33e54994..53453dd1 100644 --- a/tests/corpora/test_update_usfm_parser_handler.py +++ b/tests/corpora/test_update_usfm_parser_handler.py @@ -16,6 +16,7 @@ UsfmUpdateBlockElementType, UsfmUpdateBlockHandler, ) +from machine.scripture import ORIGINAL_VERSIFICATION def test_get_usfm_verse_char_style() -> None: @@ -1756,6 +1757,36 @@ def test_unclosed_style_marker_does_not_consume_next_paragraph_marker() -> None: assert_usfm_equals(target, result) +def test_update_usfm_converts_to_rows_versification() -> None: + # Original vs. English + # MAL 4:1 = MAL 3:19 + rows = [ + UpdateUsfmRow(scr_ref("MAL 3:18"), "New 18"), + UpdateUsfmRow(scr_ref("MAL 4:1"), "New 1"), + ] + source = r"""\id MAL - Test +\c 3 +\p +\v 18 Old 18 +\v 19 Old 19 +""" + settings = DefaultParatextProjectSettings( + versification=ORIGINAL_VERSIFICATION, file_name_form="MAT", file_name_suffix="" + ) + updater = MemoryParatextProjectTextUpdater({"MAL": source.replace("\n", "\r\n")}, settings) + target = updater.update_usfm("MAL", rows, text_behavior=UpdateUsfmTextBehavior.PREFER_NEW) + + result = r"""\id MAL - Test +\c 3 +\p +\v 18 New 18 +\c 4 +\nb +\v 1 New 1 +""" + assert_usfm_equals(target, result) + + def scr_ref(*refs: str) -> List[ScriptureRef]: return [ScriptureRef.parse(ref) for ref in refs] diff --git a/tests/testutils/memory_paratext_project_file_handler.py b/tests/testutils/memory_paratext_project_file_handler.py index be6692a5..91aafd13 100644 --- a/tests/testutils/memory_paratext_project_file_handler.py +++ b/tests/testutils/memory_paratext_project_file_handler.py @@ -3,7 +3,7 @@ from machine.corpora import ParatextProjectFileHandler, ParatextProjectSettings, UsfmStylesheet from machine.corpora.paratext_project_text_updater_base import ParatextProjectTextUpdaterBase -from machine.scripture import ORIGINAL_VERSIFICATION, Versification +from machine.scripture import ENGLISH_VERSIFICATION, Versification class MemoryParatextProjectFileHandler(ParatextProjectFileHandler): @@ -61,7 +61,7 @@ def __init__( name, full_name, encoding if encoding is not None else "utf-8", - versification if versification is not None else ORIGINAL_VERSIFICATION, + versification if versification is not None else ENGLISH_VERSIFICATION, stylesheet if stylesheet is not None else UsfmStylesheet("usfm.sty"), file_name_prefix, file_name_form,