From ea81e9a864b7b267d9ad13e081cc86a899967a84 Mon Sep 17 00:00:00 2001 From: Ciprian-LocalPulse Date: Tue, 22 Sep 2026 14:50:24 +0300 Subject: [PATCH] fix: classify publication origins explicitly --- CHANGELOG.md | 6 +- ROADMAP.md | 2 +- docs/API.md | 8 +- docs/audits/v0.3.0-baseline.md | 14 ++- scripts/seed_test_db.py | 2 + src/openlongevity/api.py | 32 +++++- src/openlongevity/origins.py | 27 +++++ src/openlongevity/providers/base.py | 3 + src/openlongevity/providers/pubmed.py | 6 +- src/openlongevity/repository.py | 28 +++-- tests/test_providers.py | 63 +++++++++++ tests/test_publication_origin.py | 144 ++++++++++++++++++++++++++ 12 files changed, 316 insertions(+), 19 deletions(-) create mode 100644 src/openlongevity/origins.py create mode 100644 tests/test_publication_origin.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 7f1ad6c..625f179 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,10 @@ ## Unreleased — documentation and integrity audit +Persisted publication responses now expose an explicit `origin` contract with `unknown`, `manual`, `provider`, and `synthetic` states plus a tri-state `synthetic` interpretation. Synthetic seeds and `SYN-*` or `SEED-*` identifiers remain synthetic even when their payload has provider-shaped metadata. Legacy rows without a documented origin are normalized conservatively as `unknown` rather than being presented as real provider retrievals. Revision history carries the normalized origin fields in each payload. + +PubMed parsing remains conservative: direct XML parsing and injected transports do not certify provider origin, while the ordinary built-in PubMed search path marks records as provider-derived before persistence. Tests now cover origin normalization, PubMed path classification, provider-shaped synthetic payloads, revision changes from unknown to manual origin, API list/detail/history responses, and the CI seed record's synthetic label. + Evidence list, detail, and citation export now reconstruct current review metadata from the latest persisted event by ID. Historical events remain intact, synthetic origin remains authoritative, and configured storage failures return an error instead of stale fixture review metadata. Integration coverage exercises successive review states, deliberately backdated timestamps, and recovery through a fresh application instance. Review-event history now uses bounded cursor pagination with a default page size of fifty and a maximum of one hundred. Clients follow `next_after_id` to retrieve subsequent pages. PostgreSQL integration coverage verifies persisted review payloads, event-ID ordering, record isolation, and continued exclusion of reviewed fixtures from citation export. @@ -16,7 +20,7 @@ Documentation tooling now counts prose separately from fenced code and diagrams, The abbreviated license file has been replaced with the official Apache License 2.0 text. Project attribution is recorded separately in NOTICE, and citation metadata identifies the author as an independent Romanian researcher. The existing release tag is preserved. The citation's release version is not silently advanced to an unpublished development version merely because the Python package declares that version. -Known implementation issues remain visible: synthetic-origin serialization, incomplete pathway adjustment, duplicate multi-omics handling, and the absence of an operational scientific review service. The frontend still needs evidence of a production build and interaction testing beyond compiler-only scripts. No new deployment, clinical validation, or independent benchmark is asserted by this documentation entry. +Known implementation issues remain visible: incomplete pathway adjustment, duplicate multi-omics handling, and the absence of an operational scientific review service. The frontend still needs evidence of a production build and interaction testing beyond compiler-only scripts. Publication origin classification is improved, but provider origin remains a retrieval-boundary label rather than scientific validation. No new deployment, clinical validation, or independent benchmark is asserted by this documentation entry. ## [0.2.0] - 2026-09-14 diff --git a/ROADMAP.md b/ROADMAP.md index e104c4d..1166532 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -34,7 +34,7 @@ The current baseline is a research prototype with partially integrated infrastru ## Immediate integrity and documentation priorities -Correct the publication-origin model so that synthetic, manually entered, and genuinely retrieved records are classified through explicit provenance rather than a fixed false synthetic flag. The acceptance evidence should include the existing test seed, ordinary provider ingestion, and a deliberately provider-shaped synthetic payload. Classification must survive persistence, API responses, and presentation. A field-name change alone is insufficient if the underlying decision remains unverified. +Publication-origin classification has moved from a fixed false synthetic flag to an explicit persistence contract. Synthetic seeds, manually documented records, ordinary PubMed provider-path records, and legacy unknown records now have different API semantics. Acceptance coverage includes the CI seed, ordinary PubMed search classification, a deliberately provider-shaped synthetic payload, persistence, list/detail responses, and revision history. The remaining work is presentation: the frontend must display these distinctions clearly rather than flattening them into a single visual style. Complete the documentation expansion with distinct, source-backed explanations rather than repeated filler. Every tracked Markdown file has a minimum prose target, but accuracy and implementation alignment remain separate requirements. The automated inventory should continue to expose short documents, missing attribution, and structural problems. Code examples need execution checks, and diagrams need to identify proposed components clearly. A passing word-count gate is not a substitute for editorial review. diff --git a/docs/API.md b/docs/API.md index 0d24b5f..0ceeb91 100644 --- a/docs/API.md +++ b/docs/API.md @@ -57,6 +57,12 @@ The page response includes items, total, page, page size, persisted mode, and th Publication detail uses the local identifier in the path. Encode identifiers correctly as URL path data and do not infer that a local identifier is the exact string expected by a provider API. Missing records return a structured not-found error. The history route beneath a publication identifier returns stored revision payloads. A revision is a local content-history object; it should not be presented as an independent scientific review or a source correction notice unless that relationship is explicitly established. +Publication responses include two related interpretation fields: `origin` and `synthetic`. `origin` can be `unknown`, `manual`, `provider`, or `synthetic`. The `synthetic` field is tri-state: `true` means the record is a fixture or seed record, `false` means the record entered through the implemented standard provider path, and `null` means the platform is preserving uncertainty rather than asserting a real provider retrieval. Legacy records without explicit origin metadata are treated conservatively as `unknown` unless their identifier or stored payload marks them as synthetic. This is intentional backward-compatible behavior for JSON rows already present in PostgreSQL. + +`provider` is an ingestion-boundary label, not a scientific validation label. In the current implementation, ordinary PubMed ingestion through the built-in provider can produce `origin: provider`; PubMed XML parsed directly in a fixture and records retrieved through an injected test transport remain `unknown` until a higher-level path documents the retrieval boundary. A provider-shaped payload does not override explicit synthetic status, and seed identifiers remain synthetic even if their provenance object resembles provider metadata. + +Publication history returns the normalized origin fields inside each revision payload. The revision content hash continues to identify the stored revision content at the time it was written; the response may also annotate legacy payloads with conservative origin fields for client clarity. Clients should therefore compare revision numbers and hashes for local history, and use `origin` and `synthetic` for interpretation. A change from `unknown` to `manual`, `provider`, or `synthetic` is a content-contract change that can create a new revision when saved through the repository. + ## Ingestion request and transaction semantics The ingestion body requires a nonempty query of at most two hundred characters and a limit between one and twenty-five, defaulting to five. Additional body fields are rejected by the request model. Whitespace-only queries are rejected when constructing the provider search query. The request should be sent by an authorized operator, and the server must have both the ingestion key and a configured publication repository before useful work can proceed. @@ -81,7 +87,7 @@ The citation export route should not be described as a complete publication expo Evidence grades and scores require their methodological labels. The A–G mapping is a project taxonomy, and the numerical navigation score uses heuristic constants. Neither is a calibrated scientific certainty estimate. The score also depends on execution time when a publication date is present. The API's ability to serialize a number does not justify describing it as a treatment effect, probability of truth, or measure of human longevity benefit. -Publication responses currently enforce a false synthetic flag without deriving authenticity from a verified origin model. This is a known limitation for test-seeded or otherwise manually inserted records. Clients and operators must not use that flag alone as proof that an item came from a real provider request. Correcting this requires code and schema decisions plus a regression test; documenting the limitation does not repair it. +Publication origin classification is now explicit for persisted publications, including synthetic seed rows and records saved through the PubMed ingestion path. Clients and operators must still avoid treating `synthetic: false` as proof of scientific reliability. It means the record was not classified as synthetic by the storage contract and entered through a provider-boundary path; it does not mean the publication is complete, unretracted, clinically relevant, or human reviewed. ### Bounded review history diff --git a/docs/audits/v0.3.0-baseline.md b/docs/audits/v0.3.0-baseline.md index f274bba..91f2a7a 100644 --- a/docs/audits/v0.3.0-baseline.md +++ b/docs/audits/v0.3.0-baseline.md @@ -61,7 +61,7 @@ The whitepaper previously used constructor fields and a study-type name absent f The API reference previously described query-parameter ingestion and a clinical-trials ingestion route not supported by the current application. It now documents the JSON request body, operator key, bounds, and local persistence behavior of PubMed ingestion. Publication title search is distinguished from full-text or federated search. Synthetic evidence and graph demonstrations are distinguished from persisted publication resources. The removed wiki reference has been replaced with an existing provider-provenance document. -Several implementation limitations remain unresolved by these editorial changes. Publication serialization assigns a false synthetic flag even when a row was inserted as test material. Multi-omics integration overwrites duplicate sample-layer entries and does not enforce participant consistency. Pathway adjustment lacks parts of standard Benjamini–Hochberg processing. Review-state fields do not establish an operational authorization and adjudication service. The revised documents identify these limits directly instead of claiming that prose changes repaired the code. +Several implementation limitations remain unresolved by these editorial changes. At that checkpoint, publication serialization assigned a false synthetic flag even when a row was inserted as test material. Multi-omics integration overwrote duplicate sample-layer entries and did not enforce participant consistency. Pathway adjustment lacked parts of standard Benjamini–Hochberg processing. Review-state fields did not establish an operational authorization and adjudication service. The revised documents identified these limits directly instead of claiming that prose changes repaired the code. ## Executed verification and its limits @@ -75,7 +75,17 @@ The Python documentation auditor verifies attribution, prose thresholds, local i The product remains a research prototype with partially integrated infrastructure. The publication path has more substance than a static mockup, but the existence of a repository and API does not establish an end-to-end scientific workflow. The frontend package still uses compiler-only scripts rather than a demonstrated production build and interaction-test suite. No verified Vercel deployment or new release gate completion is established by this follow-up. -The next documentation work is to expand the remaining outlines with distinct methods, assumptions, examples, and evidence rather than repeat common paragraphs. The next implementation work is to address the documented origin, statistical, and integration limitations with targeted tests. Release readiness needs its own review against the mandatory gates in the project request. Neither a higher word count nor a more polished diagram can substitute for those execution results. +The next documentation work is to expand the remaining outlines with distinct methods, assumptions, examples, and evidence rather than repeat common paragraphs. The next implementation work is to address the documented statistical and integration limitations with targeted tests, and to keep refining origin presentation in the frontend. Release readiness needs its own review against the mandatory gates in the project request. Neither a higher word count nor a more polished diagram can substitute for those execution results. + +## Origin classification follow-up: 22 September 2026 + +The publication-origin limitation recorded above has a targeted implementation correction. Persisted publication payloads now carry an explicit `origin` value and a tri-state `synthetic` field. The origin vocabulary distinguishes `unknown`, `manual`, `provider`, and `synthetic`. Legacy records without an explicit origin are normalized conservatively as `unknown`; `SYN-*` and `SEED-*` identifiers, or payloads already marked synthetic, remain synthetic even if the surrounding provenance object resembles a provider record. This prevents CI seed data or manually shaped rows from being displayed as ordinary provider-derived publications. + +The PubMed adapter now marks records as provider-derived only through the ordinary built-in search path. Parser-only records and records produced through injected transports remain `unknown`, because XML parsing or test transport execution is not the same evidence as a live provider retrieval boundary. The repository includes the normalized origin fields in saved payloads, list responses, detail responses, and revision-history payloads. Changing origin from unknown to a documented state is treated as a content-contract change and can create a new revision. + +The verification added for this correction covers origin normalization rules, PubMed parser behavior, PubMed search-path classification, provider-shaped synthetic payloads, revision history, API list/detail/history responses, and the seeded CI publication. Local verification reported `ruff check` passing and the local Python suite passing with PostgreSQL integration cases skipped when `TEST_DATABASE_URL` is absent. The GitHub PostgreSQL workflow remains the authoritative check for database integration under CI. + +This correction improves source interpretation, but it does not complete the scientific evidence workflow. `origin: provider` means the record crossed an implemented retrieval boundary; it does not mean the paper is correct, complete, unretracted, clinically relevant, or human reviewed. The product remains an alpha research prototype until frontend rendering, provider-backed evidence extraction, review operations, export manifests, and release gates are demonstrated together. This audit records project responsibility under Ciprian Ștefan Pleșca, an independent Romanian researcher. It does not claim an external audit institution, university affiliation, scientific peer review, or security certification. Its contribution is a traceable account of what was inspected, what changed, what was actually checked, and what remains unresolved. Future updates should preserve that separation so that readers can evaluate progress without having to infer completion from presentation quality. diff --git a/scripts/seed_test_db.py b/scripts/seed_test_db.py index 451a3c6..bb91da4 100644 --- a/scripts/seed_test_db.py +++ b/scripts/seed_test_db.py @@ -17,6 +17,8 @@ RETRIEVED_AT = "2026-01-01T00:00:00Z" SEED_PAYLOAD = { + "origin": "synthetic", + "synthetic": True, "identifier": SEED_IDENTIFIER, "title": SEED_TITLE, "abstract": "Synthetic seed record used only for CI contract tests.", diff --git a/src/openlongevity/api.py b/src/openlongevity/api.py index dc0c280..3019f8a 100644 --- a/src/openlongevity/api.py +++ b/src/openlongevity/api.py @@ -21,8 +21,9 @@ from .exports import build_citation_export, evidence_record_payload from .gaps import ResearchGapDetector from .models import EvidenceRecord, ReviewStatus, StudyType +from .origins import PublicationOrigin from .providers import PubMedProvider, SearchQuery -from .providers.base import ProviderError, Publication +from .providers.base import ProviderError from .review import apply_human_review @@ -53,11 +54,30 @@ class PublicationResponse(BaseModel): retraction_status: str = "unknown" corrections: list[dict[str, str]] = Field(default_factory=list) revision: int - synthetic: Literal[False] = False + origin: PublicationOrigin + synthetic: bool | None first_retrieved_at: str last_retrieved_at: str +class PublicationRevisionPayload(BaseModel): + identifier: str + title: str + abstract: str = "" + authors: list[str] = Field(default_factory=list) + journal: str | None = None + publication_date: str | None = None + doi: str | None = None + publication_types: list[str] = Field(default_factory=list) + mesh_terms: list[str] = Field(default_factory=list) + citation_count: int | None = None + provenance: ProvenanceResponse + retraction_status: str = "unknown" + corrections: list[dict[str, str]] = Field(default_factory=list) + origin: PublicationOrigin + synthetic: bool | None + + class PublicationPage(BaseModel): items: list[PublicationResponse] total: int @@ -83,7 +103,7 @@ class EvidenceReviewRequest(BaseModel): class RevisionResponse(BaseModel): revision: int - payload: Publication + payload: PublicationRevisionPayload content_hash: str retrieved_at: str @@ -208,8 +228,10 @@ async def publication(identifier: str) -> PublicationResponse: @app.get("/api/v1/publications/{identifier}/history", response_model=list[RevisionResponse]) async def history(identifier: str) -> list[RevisionResponse]: await publication(identifier) - return [RevisionResponse.model_validate(row) - for row in await require_repository().history(identifier)] + return [ + RevisionResponse.model_validate(row) + for row in await require_repository().history(identifier) + ] engine = EvidenceEngine() fixtures = [EvidenceRecord( diff --git a/src/openlongevity/origins.py b/src/openlongevity/origins.py new file mode 100644 index 0000000..d64869f --- /dev/null +++ b/src/openlongevity/origins.py @@ -0,0 +1,27 @@ +"""Explicit data-origin labels; provider metadata alone does not prove retrieval.""" + +from collections.abc import Mapping +from enum import StrEnum +from typing import Any + + +class PublicationOrigin(StrEnum): + UNKNOWN = "unknown" + MANUAL = "manual" + PROVIDER = "provider" + SYNTHETIC = "synthetic" + + +def publication_origin_fields(payload: Mapping[str, Any]) -> dict[str, Any]: + """Normalize legacy records conservatively without rewriting stored history.""" + try: + origin = PublicationOrigin(payload.get("origin", "unknown")) + except (ValueError, TypeError): + origin = PublicationOrigin.UNKNOWN + identifier = str(payload.get("identifier", "")).upper() + if payload.get("synthetic") is True or identifier.startswith(("SYN-", "SEED-")): + origin = PublicationOrigin.SYNTHETIC + synthetic = True if origin == PublicationOrigin.SYNTHETIC else ( + False if origin == PublicationOrigin.PROVIDER else None + ) + return {"origin": origin, "synthetic": synthetic} diff --git a/src/openlongevity/providers/base.py b/src/openlongevity/providers/base.py index 2205e21..0331004 100644 --- a/src/openlongevity/providers/base.py +++ b/src/openlongevity/providers/base.py @@ -5,6 +5,8 @@ from datetime import UTC, datetime from typing import Protocol +from ..origins import PublicationOrigin + class ProviderError(RuntimeError): """A recoverable upstream provider or parsing failure.""" @@ -58,6 +60,7 @@ class Publication: provenance: Provenance | None = None retraction_status: str = "unknown" corrections: tuple[dict[str, str], ...] = () + origin: PublicationOrigin = PublicationOrigin.UNKNOWN @dataclass(frozen=True) diff --git a/src/openlongevity/providers/pubmed.py b/src/openlongevity/providers/pubmed.py index b57753d..26b45ea 100644 --- a/src/openlongevity/providers/pubmed.py +++ b/src/openlongevity/providers/pubmed.py @@ -1,6 +1,7 @@ """Bounded NCBI E-utilities ingestion with injectable HTTP transport.""" import asyncio import hashlib +from dataclasses import replace from datetime import UTC, datetime from os import getenv from xml.etree import ElementTree @@ -9,6 +10,7 @@ from defusedxml.common import DefusedXmlException from defusedxml.ElementTree import fromstring +from ..origins import PublicationOrigin from .base import Provenance, ProviderError, Publication, SearchQuery @@ -77,7 +79,9 @@ async def search(self, query: SearchQuery) -> list[Publication]: fetched = await self._request(client, "efetch.fcgi", { "db": "pubmed", "id": ",".join(ids[:query.limit]), "retmode": "xml" }) - return self._parse(fetched.text) + origin = (PublicationOrigin.PROVIDER if self.transport is None + else PublicationOrigin.UNKNOWN) + return [replace(record, origin=origin) for record in self._parse(fetched.text)] async def get_by_id(self, external_id: str) -> Publication | None: pmid = external_id.removeprefix("PMID:") diff --git a/src/openlongevity/repository.py b/src/openlongevity/repository.py index 2162d25..65426c1 100644 --- a/src/openlongevity/repository.py +++ b/src/openlongevity/repository.py @@ -10,6 +10,7 @@ from sqlalchemy.dialects.postgresql import insert from .db import Database, EvidenceReviewEventRow, PublicationRevisionRow, PublicationRow +from .origins import publication_origin_fields from .providers.base import Publication @@ -21,8 +22,9 @@ async def save(self, publication: Publication) -> dict[str, Any]: provenance = publication.provenance if provenance is None or not provenance.checksum: raise ValueError("Persistent publications require provenance and a source checksum") - payload = asdict(publication) - stable = asdict(publication) + publication_payload = asdict(publication) + payload = {**publication_payload, **publication_origin_fields(publication_payload)} + stable = {**payload, "provenance": dict(payload["provenance"])} stable["provenance"].pop("retrieved_at") digest = hashlib.sha256( json.dumps(stable, sort_keys=True, ensure_ascii=False).encode() @@ -62,9 +64,13 @@ async def save(self, publication: Publication) -> dict[str, Any]: @staticmethod def serialize(row: PublicationRow) -> dict[str, Any]: - return {**row.payload, "revision": row.revision, "synthetic": False, - "first_retrieved_at": row.first_retrieved_at, - "last_retrieved_at": row.last_retrieved_at} + return { + **row.payload, + **publication_origin_fields(row.payload), + "revision": row.revision, + "first_retrieved_at": row.first_retrieved_at, + "last_retrieved_at": row.last_retrieved_at, + } async def get(self, identifier: str) -> dict[str, Any] | None: async with self.database.sessions() as session: @@ -87,9 +93,15 @@ async def history(self, identifier: str) -> list[dict[str, Any]]: rows = await session.scalars(select(PublicationRevisionRow).where( PublicationRevisionRow.publication_id == identifier ).order_by(PublicationRevisionRow.revision)) - return [{"revision": row.revision, "payload": row.payload, - "content_hash": row.content_hash, "retrieved_at": row.retrieved_at} - for row in rows] + return [ + { + "revision": row.revision, + "payload": {**row.payload, **publication_origin_fields(row.payload)}, + "content_hash": row.content_hash, + "retrieved_at": row.retrieved_at, + } + for row in rows + ] class EvidenceReviewRepository: diff --git a/tests/test_providers.py b/tests/test_providers.py index b5593b0..d87f09b 100644 --- a/tests/test_providers.py +++ b/tests/test_providers.py @@ -1,5 +1,9 @@ +from unittest.mock import AsyncMock + import pytest +from httpx import MockTransport, Response +from openlongevity.origins import PublicationOrigin, publication_origin_fields from openlongevity.providers.base import SearchQuery from openlongevity.providers.pubmed import PubMedProvider @@ -27,3 +31,62 @@ def test_pubmed_xml_parser_preserves_provenance() -> None: assert records[0].doi == "10.1000/example" assert records[0].provenance is not None assert records[0].provenance.source_provider == "pubmed" + assert records[0].origin == PublicationOrigin.UNKNOWN + + +@pytest.mark.parametrize( + ("payload", "expected_origin", "expected_synthetic"), + [ + ({}, PublicationOrigin.UNKNOWN, None), + ({"origin": "manual"}, PublicationOrigin.MANUAL, None), + ({"origin": "provider"}, PublicationOrigin.PROVIDER, False), + ({"origin": "synthetic"}, PublicationOrigin.SYNTHETIC, True), + ({"origin": "provider", "synthetic": True}, PublicationOrigin.SYNTHETIC, True), + ({"origin": "not-a-real-origin"}, PublicationOrigin.UNKNOWN, None), + ({"identifier": "SYN-001", "origin": "provider"}, PublicationOrigin.SYNTHETIC, True), + ({"identifier": "SEED-0001"}, PublicationOrigin.SYNTHETIC, True), + ], +) +def test_publication_origin_fields_are_conservative( + payload: dict[str, object], + expected_origin: PublicationOrigin, + expected_synthetic: bool | None, +) -> None: + assert publication_origin_fields(payload) == { + "origin": expected_origin, + "synthetic": expected_synthetic, + } + + +async def test_pubmed_search_marks_standard_live_path_as_provider() -> None: + provider = PubMedProvider() + provider._request = AsyncMock( # type: ignore[method-assign] + side_effect=[ + Response(200, json={"esearchresult": {"idlist": ["123"]}}), + Response(200, text=_pubmed_xml("123", "Provider path aging study")), + ] + ) + records = await provider.search(SearchQuery("aging", limit=1)) + assert records[0].origin == PublicationOrigin.PROVIDER + + +async def test_pubmed_search_does_not_certify_injected_transport_as_provider() -> None: + provider = PubMedProvider(transport=MockTransport(lambda request: Response(500))) + provider._request = AsyncMock( # type: ignore[method-assign] + side_effect=[ + Response(200, json={"esearchresult": {"idlist": ["456"]}}), + Response(200, text=_pubmed_xml("456", "Injected transport aging study")), + ] + ) + records = await provider.search(SearchQuery("aging", limit=1)) + assert records[0].origin == PublicationOrigin.UNKNOWN + + +def _pubmed_xml(pmid: str, title: str) -> str: + return f""" + {pmid} +
{title} + Reported observation. + Research Journal
+
+ """ diff --git a/tests/test_publication_origin.py b/tests/test_publication_origin.py new file mode 100644 index 0000000..f9881b5 --- /dev/null +++ b/tests/test_publication_origin.py @@ -0,0 +1,144 @@ +"""Publication origin classification survives persistence, search, and history.""" + +import os +from dataclasses import replace +from uuid import uuid4 + +import pytest +from fastapi.testclient import TestClient +from httpx import ASGITransport, AsyncClient +from sqlalchemy import delete + +from openlongevity.api import create_app +from openlongevity.db import Database, PublicationRevisionRow, PublicationRow +from openlongevity.origins import PublicationOrigin +from openlongevity.providers.base import Provenance, Publication +from openlongevity.repository import PublicationRepository + +TEST_DATABASE_URL = os.getenv("TEST_DATABASE_URL") +pytestmark = [ + pytest.mark.postgres, + pytest.mark.skipif(not TEST_DATABASE_URL, reason="TEST_DATABASE_URL is required"), +] + + +async def test_publication_origin_survives_api_and_history_round_trip() -> None: + assert TEST_DATABASE_URL is not None + token = uuid4().hex + database = Database(TEST_DATABASE_URL) + repository = PublicationRepository(database) + identifiers = [f"TEST-ORIGIN-{token}"] + app = create_app(database_url=TEST_DATABASE_URL) + try: + initial = publication( + identifier=identifiers[0], + source_identifier=f"test-origin-{token}", + title="Origin audit study", + origin=PublicationOrigin.UNKNOWN, + ) + saved = await repository.save(initial) + assert saved["origin"] == "unknown" + assert saved["synthetic"] is None + assert saved["revision"] == 1 + + # A clearer documented manual origin is a content-contract change. + updated = await repository.save(replace(initial, origin=PublicationOrigin.MANUAL)) + assert updated["origin"] == "manual" + assert updated["synthetic"] is None + assert updated["revision"] == 2 + + async with app.router.lifespan_context(app): + async with AsyncClient( + transport=ASGITransport(app=app), base_url="http://test", + ) as client: + detail = (await client.get(f"/api/v1/publications/{identifiers[0]}")).json() + assert detail["origin"] == "manual" + assert detail["synthetic"] is None + + search = (await client.get( + "/api/v1/search", params={"query": "Origin audit"}, + )).json() + assert search["items"][0]["origin"] == "manual" + assert search["items"][0]["synthetic"] is None + + history = (await client.get( + f"/api/v1/publications/{identifiers[0]}/history" + )).json() + assert [item["revision"] for item in history] == [1, 2] + assert history[0]["payload"]["origin"] == "unknown" + assert history[0]["payload"]["synthetic"] is None + assert history[1]["payload"]["origin"] == "manual" + assert history[1]["payload"]["synthetic"] is None + assert history[0]["content_hash"] != history[1]["content_hash"] + finally: + await cleanup(database, identifiers) + + +async def test_synthetic_prefix_overrides_provider_shaped_metadata() -> None: + assert TEST_DATABASE_URL is not None + token = uuid4().hex + identifier = f"SEED-{token}" + database = Database(TEST_DATABASE_URL) + repository = PublicationRepository(database) + try: + saved = await repository.save(publication( + identifier=identifier, + source_identifier=f"provider-shaped-{token}", + title="Provider-shaped synthetic seed", + origin=PublicationOrigin.PROVIDER, + )) + assert saved["origin"] == "synthetic" + assert saved["synthetic"] is True + finally: + await cleanup(database, [identifier]) + + +@pytest.mark.skipif( + not TEST_DATABASE_URL, + reason="TEST_DATABASE_URL is required for the PostgreSQL-backed search contract test", +) +def test_seed_publication_is_labeled_synthetic_in_search() -> None: + app = create_app(database_url=TEST_DATABASE_URL) + with TestClient(app) as client: + response = client.get("/api/v1/search", params={"query": "senescence"}) + assert response.status_code == 200 + seed = next( + item for item in response.json()["items"] if item["identifier"] == "SEED-0001" + ) + assert seed["origin"] == "synthetic" + assert seed["synthetic"] is True + + +def publication( + *, + identifier: str, + source_identifier: str, + title: str, + origin: PublicationOrigin, +) -> Publication: + return Publication( + identifier=identifier, + title=title, + abstract="Synthetic integration fixture for persistence behavior.", + provenance=Provenance( + source_provider="test", + source_identifier=source_identifier, + source_url=f"https://example.invalid/{source_identifier}", + retrieved_at="2026-09-22T00:00:00+00:00", + checksum=f"checksum-{source_identifier}", + normalization_version="test", + parser_version="test", + ), + origin=origin, + ) + + +async def cleanup(database: Database, identifiers: list[str]) -> None: + async with database.sessions.begin() as session: + await session.execute(delete(PublicationRevisionRow).where( + PublicationRevisionRow.publication_id.in_(identifiers) + )) + await session.execute(delete(PublicationRow).where( + PublicationRow.identifier.in_(identifiers) + )) + await database.close()