From 2294451d7827c3da47099f2593614c4964ec2e41 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sat, 15 Aug 2026 19:15:51 +0900 Subject: [PATCH 1/5] docs(evidence): define language-agnostic semantic spans Replace paragraph-only scope with a provider-neutral design for exact source blocks, versioned embedding profiles, final-payload token budgets, recursive recovery, hierarchy, and evidence-backed evaluation. Propose ADR 0017 and PRD v0.5 without claiming implementation or release. --- DOCUMENTATION.md | 9 +- ...nguage-agnostic-semantic-span-budgeting.md | 220 +++++++ docs/adr/README.md | 6 +- docs/product/prd-v0.5-proposed.md | 235 ++++++++ ...nguage-agnostic-semantic-span-embedding.md | 40 ++ ...nguage-agnostic-semantic-span-embedding.md | 556 ++++++++++++++++++ ...agnostic-semantic-span-embedding-design.md | 396 +++++++++++++ 7 files changed, 1458 insertions(+), 4 deletions(-) create mode 100644 docs/adr/0017-language-agnostic-semantic-span-budgeting.md create mode 100644 docs/product/prd-v0.5-proposed.md create mode 100644 docs/research/language-agnostic-semantic-span-embedding.md create mode 100644 docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md create mode 100644 docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md diff --git a/DOCUMENTATION.md b/DOCUMENTATION.md index 230c5abe..225986a3 100644 --- a/DOCUMENTATION.md +++ b/DOCUMENTATION.md @@ -1,10 +1,14 @@ # TEPP Documentation Map -TEPP's approved PRD v0.4 and implementation plan are the primary product baseline. This index makes the technical, data, scientific, security/privacy, integration, quality, operating, and assurance contracts discoverable without duplicating that source material. +TEPP's approved PRD v0.4 and implementation plan are the primary product baseline. This index makes the technical, data, scientific, security/privacy, integration, quality, operating, and assurance contracts discoverable without duplicating that source material. Proposed documents remain non-authoritative until their decision and product status are accepted through review. | Area | Canonical document | |---|---| | Approved product requirements | [`docs/product/prd-v0.4-approved.md`](docs/product/prd-v0.4-approved.md) | +| Proposed semantic-span PRD delta | [`docs/product/prd-v0.5-proposed.md`](docs/product/prd-v0.5-proposed.md) | +| Language-agnostic semantic-span design | [`docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md`](docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md) | +| Language-agnostic semantic-span implementation plan | [`docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md`](docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md) | +| Semantic-span research traceability | [`docs/research/language-agnostic-semantic-span-embedding.md`](docs/research/language-agnostic-semantic-span-embedding.md) | | Whole-conversation documentation fitness | [`docs/DOCUMENTATION_ASSESSMENT.md`](docs/DOCUMENTATION_ASSESSMENT.md) | | Technical requirements | [`docs/TRD.md`](docs/TRD.md) | | Architecture | [`ARCHITECTURE.md`](ARCHITECTURE.md) | @@ -23,6 +27,7 @@ TEPP's approved PRD v0.4 and implementation plan are the primary product baselin | Requirement/research/evidence traceability | [`docs/TRACEABILITY.md`](docs/TRACEABILITY.md) | | Architecture decision index / ownership map | [`docs/adr/README.md`](docs/adr/README.md) | | ADR status, maturity, and supersession policy | [`docs/adr/ADR_POLICY.md`](docs/adr/ADR_POLICY.md) | +| Proposed semantic-span authority | [`docs/adr/0017-language-agnostic-semantic-span-budgeting.md`](docs/adr/0017-language-agnostic-semantic-span-budgeting.md) | | Delivery roadmap | [`docs/roadmaps/2026-08-05-tepp-delivery-roadmap.md`](docs/roadmaps/2026-08-05-tepp-delivery-roadmap.md) | | Foundation implementation plan | [`docs/superpowers/plans/2026-08-05-temporal-event-foundation.md`](docs/superpowers/plans/2026-08-05-temporal-event-foundation.md) | | Foundation validation ledger | [`docs/validation/temporal-event-foundation.md`](docs/validation/temporal-event-foundation.md) | @@ -56,4 +61,4 @@ The documentation graph is **design-sufficient** when a reviewer can reconstruct It is **protected-main-sufficient** only after the canonical documents are integrated on protected `main`, remain semantically current with live code, and their required exact-head documentation/security/review gates pass. An active documentation PR can therefore be design-sufficient while the protected branch remains documentation-insufficient. -At the time of this review, immutable evidence records/exact spans, the Rust workspace quality foundation, and typed six-clock values/uncertain intervals (PR #8) are implemented-main. PR #9 is the active-PR that replays Task 4 Allen interval algebra and bounded path-consistency reasoner work onto that protected-main temporal foundation. Superseded PRs #5 and #6 remain historical lineage only. Event ontology, PostgreSQL persistence, shared-latent topic estimation, GPU kernels, TDT/CHRONOS intelligence, longitudinal ESEM/DSEM, visual analytics, production HTTP services, and deployment assurance remain later accepted-target or deployment-owned work. +At the time of this proposal, immutable evidence records/exact spans and the Rust workspace quality foundation are implemented on protected main. The language-agnostic semantic-span pipeline, provider model-profile registry, final-payload budget gate, hierarchy, adapters, and retrieval benchmark described in ADR 0017 remain proposed research/design work until independently reviewed, accepted, implemented, and promoted under ADR 0014. \ No newline at end of file diff --git a/docs/adr/0017-language-agnostic-semantic-span-budgeting.md b/docs/adr/0017-language-agnostic-semantic-span-budgeting.md new file mode 100644 index 00000000..173adc98 --- /dev/null +++ b/docs/adr/0017-language-agnostic-semantic-span-budgeting.md @@ -0,0 +1,220 @@ +# ADR 0017 — Language-agnostic semantic-span and embedding-budget authority + +**Decision status:** Proposed +**Implementation maturity:** research-only +**Date:** 2026-08-15 +**Supersedes:** None; narrows the implementation of ADR 0004 multilingual alignment and complements ADR 0008 exact evidence spans, ADR 0010 LLM orchestration, ADR 0011 modular MSA authority, and ADR 0012 topic-measurement inputs. + +## Context + +TEPP must turn documentary evidence into meaning-bearing embedding inputs without assuming that a document uses one known language, whitespace-delimited words, one script, or a language-specific morphology pipeline. A document may mix Korean, English, Chinese, Japanese, Thai, Arabic, identifiers, code, formulas, emoji, tables, quotations, and translated passages within one source span. + +The current draft PR #56 uses blank-line splitting as the semantic-unit implementation. That is a useful source-structure hint, but it is not the required architecture. Blank lines do not establish semantic coherence, do not protect the final rendered embedding payload from a model context overflow, do not represent headings, lists, tables, code, captions, dialogue turns, or DOM regions, and do not restore parent or neighboring context at retrieval time. + +The target OpenAI profile currently relevant to this decision, `text-embedding-3-large`, has a documented maximum input of 8,192 tokens, uses `cl100k_base` for third-generation embedding token estimation, and returns 3,072 dimensions by default with an optional `dimensions` parameter. These values are provider facts, not universal TEPP constants. They can change by model or revision and therefore belong in a versioned model profile. + +## Decision + +TEPP owns a provider-neutral **language-agnostic semantic-span pipeline**. Language identification may be recorded for evaluation, governance, or later measurement-invariance analysis, but it MUST NOT select the core segmentation, packing, overflow, or fallback algorithm. + +The pipeline has seven ordered contracts. + +### 1. Immutable evidence and source structure + +`evidence_core` preserves immutable source bytes, decoded text, exact byte and Unicode-scalar offsets, and available structural evidence. Parsers emit typed source blocks such as: + +- `document_title`; +- `section_heading`; +- `paragraph_block`; +- `list_item`; +- `table_row_group`; +- `code_block`; +- `caption_block`; +- `dialogue_turn`; +- `dom_region`; +- `unknown_block`. + +A paragraph is one candidate block, not automatically one final embedding unit. + +### 2. Language-agnostic micro-units + +A future `semantic_preprocessor` crate converts source blocks into exact-span micro-units. It uses source structure, Unicode-safe grapheme and sentence boundary hints, line boundaries, punctuation categories, and bounded token windows. It does not require a language code, dictionary, word count, stopword list, stemming, morphology, TF-IDF, BM25, or translation into a pivot language. + +Language-tailored analyzers may later supply optional annotations behind an adapter, but failure or absence of such an analyzer cannot change the correctness of the base pipeline. + +### 3. Versioned embedding-model profiles + +Every embedding request is governed by an immutable `embedding_model_profile` containing at least: + +- provider and endpoint family; +- model identifier and observed revision; +- tokenizer profile and tokenizer artifact digest; +- maximum input tokens; +- default and requested vector dimensions; +- input role; +- metadata-template version; +- profile source and verification date. + +For `text-embedding-3-large`, the initial profile records `max_input_tokens = 8192`, `tokenizer_profile = cl100k_base`, and `default_dimensions = 3072`. TEPP code does not scatter these values as literals. + +### 4. Budget-aware semantic-span packing + +Micro-units are packed in document order. A candidate span may be merged only when all of the following hold: + +1. no mandatory structural boundary is crossed; +2. the final rendered payload remains within the active model profile; +3. the configurable target-size policy allows the merge; +4. semantic evidence does not indicate a strong boundary. + +The boundary score may combine structural break strength, block-type transition, dense adjacent-unit similarity drop, and length pressure. It MUST NOT contain language identity or TF-IDF/BM25 features. + +Pre-packing may reserve tokens for headings and other metadata, but the authoritative gate tokenizes the **final rendered payload**, not only its source text. No request may be sent when: + +\[ +\operatorname{tokens}(\operatorname{rendered\_payload}) +> +\operatorname{max\_input\_tokens}. +\] + +The hard overflow rate is therefore required to be zero. + +### 5. Recursive oversized-unit recovery + +When one candidate unit exceeds the budget, TEPP applies the following order: + +1. split by known child blocks; +2. split at Unicode-safe sentence or line boundaries; +3. split at punctuation or clause-like boundaries that preserve exact offsets; +4. split by tokenizer offsets as the final fallback. + +The final fallback may use a bounded overlap, but overlap is not the primary context strategy. An unsplittable unit that cannot be represented with valid offsets fails closed as `unit_unsplittable_under_budget`; it is never silently truncated. + +### 6. Hierarchical context graph + +TEPP indexes three coordinated levels: + +- `leaf_span` — precise retrieval unit; +- `section_span` — parent context and section summary; +- `document_span` — coarse document routing and document summary. + +Each leaf retains parent, previous-sibling, next-sibling, source-block, and exact-source references. Retrieval first identifies document or section candidates, retrieves leaf spans, and restores bounded parent and neighbor context. Summaries are separate, versioned derived artifacts and never replace source evidence. + +### 7. Provider and ecosystem boundaries + +- TEPP owns exact semantic-span identity, hierarchy, budgeting policy, evaluation, and evidence provenance. +- `contextual-orchestrator` may provide optional LLM boundary proposals, summaries, or verification through a versioned API; it does not own TEPP evidence or token-budget authority. +- `pg-llm-batch` may implement a batch transport and a Postgres `pg_tiktoken` token-count adapter; it does not own segmentation policy. +- `semantic-data-portal` may persist and retrieve TEPP vectors and graph relations as a downstream consumer. +- `EmbedRelay` governs later embedding-space migration; it does not translate source segmentation into a different semantic claim. +- Direct cross-repository application-table access remains prohibited under ADR 0011. + +## Non-goals + +This ADR does not: + +- assert that every language has been psychometrically validated; +- make language detection a prerequisite; +- translate all text to English or another pivot language; +- use TF-IDF, BM25, or bag-of-words weights for semantic-span construction; +- claim that a blank line, sentence, or paragraph is always a complete semantic unit; +- implement true late chunking for a closed embedding API that does not expose contextual token embeddings and pooling control; +- let an LLM directly write authoritative spans without deterministic offset and budget validation; +- establish a stable public release. + +## Alternatives considered + +1. **Blank-line paragraph splitting only** — rejected as the final architecture because formatting is not semantic coherence and the final payload is not budgeted. +2. **Fixed token windows with fixed overlap** — retained only as the final recovery mechanism because it has predictable size but often cuts concepts and creates duplicate retrieval. +3. **Language detection followed by language-specific tokenizers** — rejected as the base control flow because code switching, unknown languages, and long-tail scripts become correctness failures. +4. **Translation-first embedding** — rejected because translation can remove distinctions, introduces an additional model and measurement error, and breaks exact-source correspondence. +5. **LLM-only segmentation** — rejected because it is nondeterministic, expensive, vulnerable to prompt injection, and cannot be trusted for exact offsets or hard token limits. +6. **Structure + model budget + optional dense boundary + hierarchical context** — selected. + +## Consequences + +- exact source spans remain the canonical evidence identity; +- paragraph-only logic becomes one structural adapter rather than the product contract; +- model limits are configuration data with source, revision, and digest; +- the same algorithm works when language metadata is absent, mixed, wrong, or unresolved; +- provider calls are impossible until the final payload is counted; +- retrieval can use small precise leaves without abandoning larger context; +- a deterministic structure-only mode remains available when dense or LLM services fail; +- model-specific and provider-specific behavior stays behind ports, preserving standalone and MSA use. + +The approach costs more metadata, hierarchy storage, and evaluation work than fixed windows. Dense boundary refinement also adds embedding calls. Those costs are measured rather than hidden. + +## Failure and recovery + +TEPP returns typed, content-redacting failures: + +- `tokenizer_profile_unavailable`; +- `embedding_profile_unverified`; +- `embedding_payload_too_large`; +- `unit_unsplittable_under_budget`; +- `invalid_source_offset`; +- `semantic_similarity_unavailable`; +- `embedding_provider_unavailable`. + +`semantic_similarity_unavailable` may degrade to deterministic structure-only packing and records `boundary_mode = structure_only`. Tokenizer/profile/offset failures cannot degrade because they protect correctness. Provider retries and fallback are bounded and recorded; they never change span identity silently. + +## Security, privacy, and governance impact + +Document content is untrusted data. It cannot alter the model profile, metadata template, access list, network destination, tool authority, or token budget. LLM boundary proposals require exact source references and deterministic validation. No external resource is fetched during segmentation. + +TEPP preserves PII required for authorized work under ADR 0009 rather than blanket masking it. Provider disclosure is purpose-bound and recorded. Logs and metrics contain identifiers, counts, policy versions, hashes, and outcomes—not source text, secrets, or embedding vectors. + +Embedding vectors, tokenizer artifacts, summaries, and model-profile artifacts are versioned and integrity checked. An embedding vector is always bound to `embedding_space_id`, model profile, input role, payload hash, and source-span set. + +## Compatibility and migration + +The initial implementation is additive. Existing document and source-span contracts remain valid. Paragraph-only outputs from draft PR #56 are not migrated as authoritative semantic units. + +A future persistence migration uses normalized two-or-more-word `snake_case` objects: + +- `source_block`; +- `semantic_unit`; +- `semantic_unit_relation`; +- `embedding_model_profile`; +- `embedding_payload`; +- `embedding_vector_record`; +- `embedding_evaluation_run`. + +`semantic_unit_relation` stores parent/previous/next/derived-from relations without duplicating document text. `embedding_vector_record` references an `embedding_space_id`; vectors from different spaces are never compared directly. + +## Verification + +Acceptance requires falsifiable evidence. + +### Correctness + +- final rendered-payload overflow rate is exactly zero; +- source reconstruction from every leaf span is exact; +- byte and Unicode-scalar offsets remain valid under mixed scripts, combining marks, emoji ZWJ sequences, RTL text, and malformed-input rejection; +- no control-flow branch depends on a language code; +- unsplittable oversized inputs fail closed without truncation. + +### Retrieval quality + +Compare at least: + +1. fixed token windows with overlap; +2. paragraph-only indexing; +3. structure + budget; +4. structure + budget + dense boundary; +5. hierarchical retrieval with parent/neighbor restoration. + +Report Recall@k, nDCG@k, MRR, duplicate-hit rate, context-restoration success, latency, input tokens, storage, and cost. Human-gold boundaries report precision, recall, and span agreement. Results are stratified by script/language profile and mixed-language status for evaluation only. + +### Robustness fixtures + +Tests include whitespace-delimited and non-whitespace-delimited scripts, Korean/English code switching, Chinese, Japanese, Thai, Arabic RTL, emoji and combining marks, tables, lists, code, long unpunctuated text, one token-dense source block, repeated boilerplate, and adversarial prompt-like document text. + +### Model profile + +The `text-embedding-3-large` profile is tested against 8,192-token acceptance and 8,193-token refusal using the pinned tokenizer profile. The final metadata-rendered payload, not raw content alone, is counted. + +## Rollback and supersession + +Rollback disables dense/LLM refinement and uses deterministic structure-only packing with the last validated model profile. It does not restore blank-line-only segmentation as an authoritative semantic algorithm. + +Supersession requires an ADR that preserves exact evidence offsets, versioned model limits, final-payload token enforcement, language-independent base correctness, explicit degradation, and retrieval-quality evidence. A later open-weight model may add true late chunking, but that capability must remain separate from the `text-embedding-3-large` API profile unless the provider exposes the necessary token-level contract. \ No newline at end of file diff --git a/docs/adr/README.md b/docs/adr/README.md index 1a9a7b31..e779a4d4 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -22,6 +22,7 @@ Read [`ADR_POLICY.md`](ADR_POLICY.md) first. **Decision status and implementatio | [0014](0014-scientific-claim-promotion-and-release-evidence.md) | Scientific claim promotion and release evidence authority | Accepted | partial | Separates design, implementation, scientific/product claim, and release authority; repository SBOM/provenance generator implemented, full release bundle remaining. | | [0015](0015-autonomous-development-review-and-merge-authority.md) | Autonomous development, review, and merge authority separation | Accepted | active-PR | Separates model proposal, deterministic verification, publication, independent review, and merge/release authority. | | [0016](0016-tdt-chronos-event-intelligence-boundary.md) | TDT, CHRONOS, and Event Ontology intelligence boundary | Accepted | accepted-target | Separates observed evidence, detection/tracking, prediction/schema inference, temporal consistency, and promoted transition authority. | +| [0017](0017-language-agnostic-semantic-span-budgeting.md) | Language-agnostic semantic spans, model profiles, and final-payload token budgets | Proposed | research-only | Narrows ADR 0004/0008 implementation, preserves ADR 0010/0011/0012 ownership, and supersedes paragraph-only draft PR #56 as the target design. | ## Decision ownership summary @@ -42,10 +43,11 @@ Use the narrowest owning ADR when decisions overlap: - **persistence / manifests / leakage-safe split:** ADR 0013; - **claim maturity / release evidence:** ADR 0014; - **autonomous development/review/merge authority:** ADR 0015; -- **TDT/CHRONOS event intelligence:** ADR 0016. +- **TDT/CHRONOS event intelligence:** ADR 0016; +- **semantic-unit identity / embedding-model profile / final-payload budget / hierarchy:** ADR 0017. ## Change and supersession rule ADR status changes require a pull request, source traceability, tests/evidence for affected invariants, and corresponding PRD/Architecture/TRD/Traceability updates where the approved measurement target changes. Decisions that materially alter privacy authority, orchestration authority, service ownership, temporal semantics, ontology, persistence identity, scientific estimands, release evidence, or automation authority require a superseding ADR rather than silent drift. -Partial supersession must identify the exact moved decision scope in both the older ADR and this index. Historical ADR text remains evidence of why a decision existed; it must not be silently rewritten to make a later architecture appear original. +Partial supersession must identify the exact moved decision scope in both the older ADR and this index. Historical ADR text remains evidence of why a decision existed; it must not be silently rewritten to make a later architecture appear original. \ No newline at end of file diff --git a/docs/product/prd-v0.5-proposed.md b/docs/product/prd-v0.5-proposed.md new file mode 100644 index 00000000..04be9c13 --- /dev/null +++ b/docs/product/prd-v0.5-proposed.md @@ -0,0 +1,235 @@ +# Temporal Event Psychometrics Platform — Proposed PRD v0.5 delta + +**Status:** Proposed; not an approved or implemented capability +**Date:** 2026-08-15 +**Base:** Approved PRD v0.4 +**Owning decision:** ADR 0017 +**Change scope:** Section 6 multilingual evidence measurement, Section 12 LLM responsibilities, Section 16 persistence, Section 18 delivery phase 2, and Section 19 release evidence. + +## 1. Reason for the version proposal + +Approved PRD v0.4 establishes multilingual evidence, exact source spans, optional LLM semantic-unit proposals, and a phase for multilingual evidence and semantic units. It also mentions language-tailored boundaries. The current product requirement is stricter: + +> The base semantic-span and embedding-budget pipeline must remain correct without knowing, trusting, or branching on the language of the input. + +This proposal does not remove language metadata, native lexical channels, language-profile validation, or measurement-invariance studies. It separates those scientific evaluation responsibilities from the correctness of source segmentation and model-budget enforcement. + +## 2. Product outcome + +TEPP converts any accepted Unicode source into exact, hierarchical semantic spans whose final embedding payloads cannot exceed the active embedding model's verified input limit. + +A buyer can: + +- index mixed-language and unknown-language evidence without selecting a tokenizer by language; +- trace every retrieved unit to immutable source offsets; +- retrieve precise leaf evidence and restore its section, document, and neighboring context; +- change embedding providers or limits through a versioned profile rather than code changes; +- audit why a boundary was chosen and whether processing degraded to structure-only mode; +- prove that no request exceeded the model limit and no source was silently truncated. + +## 3. Functional requirements + +### FR-SS-001 — Language-independent base control flow + +The system SHALL accept absent, mixed, unresolved, or incorrect language metadata without changing the base segmentation, packing, overflow, or recursive fallback algorithm. + +Language metadata MAY be used for stratified evaluation, lexical rendering, model-invariance analysis, and optional annotation adapters. It SHALL NOT be a required control input. + +### FR-SS-002 — Typed source blocks + +The system SHALL preserve headings, paragraphs, list items, table row groups, code blocks, captions, dialogue turns, DOM regions, and unknown blocks as typed exact-span evidence. + +### FR-SS-003 — Micro-unit construction + +The system SHALL construct micro-units from source structure and Unicode-safe boundary hints without requiring whitespace-delimited words, morphology, stopwords, TF-IDF, BM25, or translation. + +### FR-SS-004 — Versioned model profile + +Every embedding request SHALL reference an immutable model profile containing provider, model/revision, tokenizer profile and digest, maximum input tokens, vector dimensions, input role, metadata-template version, source, and verification date. + +The initial `text-embedding-3-large` profile SHALL record: + +- maximum input: 8,192 tokens; +- tokenization profile: `cl100k_base`; +- default vector length: 3,072 dimensions; +- optional requested dimensions as a profile field. + +These are initial provider facts, not universal constants. + +### FR-SS-005 — Final-payload token gate + +The system SHALL render the complete payload, including selected title/heading metadata and separators, before the authoritative token count. + +It SHALL refuse any payload for which: + +\[ +\operatorname{tokens}(\operatorname{payload}) > +\operatorname{max\_input\_tokens}. +\] + +Overflow rate SHALL equal zero. + +### FR-SS-006 — Semantic packing + +The system SHALL pack adjacent micro-units under mandatory structure boundaries, a configurable target range, the hard model budget, and optional dense semantic-boundary evidence. + +No boundary score SHALL use a language identity, TF-IDF, BM25, or bag-of-words weight. + +### FR-SS-007 — Oversized-unit recovery + +The system SHALL recursively split an oversized unit by child structure, Unicode-safe sentence/line hints, punctuation/clause hints, and finally tokenizer offsets. It SHALL never silently truncate source text. + +### FR-SS-008 — Hierarchical context graph + +The system SHALL create leaf, section, and document spans with parent and neighboring-unit relations. A retrieved leaf SHALL be able to reconstruct bounded parent and neighbor context without replacing source evidence with a summary. + +### FR-SS-009 — Explicit degradation + +If semantic-similarity or optional LLM refinement is unavailable, the system MAY continue using deterministic structure-only packing and SHALL record the degraded boundary mode. + +Tokenizer-profile, source-offset, and model-profile failures SHALL fail closed. + +### FR-SS-010 — Ecosystem ports + +TEPP SHALL expose provider-neutral ports for token counting, embedding, optional semantic similarity, optional LLM refinement, vector persistence, and batch transport. + +The owner boundaries are: + +- TEPP: unit identity, hierarchy, budget, evaluation, provenance; +- contextual-orchestrator: optional LLM proposal/verification; +- pg-llm-batch: optional batch/token-count transport adapter; +- semantic-data-portal: downstream graph/vector consumer; +- EmbedRelay: embedding-space migration. + +No service may directly read or write another service's application tables. + +## 4. Non-functional requirements + +### NFR-SS-001 — Determinism + +With the same source bytes, parser version, policy, model profile, tokenizer artifact, and deterministic similarity fixture, span identities and payload hashes SHALL be reproducible. + +### NFR-SS-002 — Auditability + +Every span SHALL record source references, block types, boundary reasons, policy version, model profile, token count, payload hash, hierarchy, degradation status, and creation provenance. + +### NFR-SS-003 — Security + +Document text SHALL be treated as untrusted data. It cannot modify tool authority, network destinations, credentials, templates, profiles, or budgets. Logs SHALL not contain source text, secrets, or vector values. + +### NFR-SS-004 — Modularity + +The deterministic segmentation/budget core SHALL run standalone with fake or local ports. Provider SDKs, databases, and contextual-orchestrator SHALL remain optional adapters. + +### NFR-SS-005 — Quality + +New production logic requires 100% line and branch coverage, complete public/safety docstrings, property tests, fuzz tests, exact-head CI, independent review, SBOM/provenance updates, and CHANGELOG evidence. + +## 5. Acceptance metrics + +### Hard gates + +- payload overflow: 0; +- silent truncation: 0; +- invalid source-offset acceptance: 0; +- language-code control branches in the base algorithm: 0; +- unversioned model-profile requests: 0; +- production line/branch coverage: 100%; +- public and safety-contract docstrings: 100%. + +### Comparative metrics + +Against fixed-window and paragraph-only baselines, report: + +- Recall@1/5/10; +- nDCG@5/10; +- MRR; +- duplicate-hit rate; +- boundary precision/recall; +- exact-span agreement; +- parent/neighbor context-restoration success; +- indexing latency and throughput; +- provider input tokens and cost; +- vector and relation storage. + +Results SHALL include uncertainty and shall be stratified by source structure, script/language profile, and mixed-language status. Stratification evaluates robustness; it does not select the base algorithm. + +## 6. User stories + +### Evidence engineer + +As an evidence engineer, I can index a document whose language is absent or mixed and receive a complete manifest showing exact spans, token counts, boundary reasons, and hierarchy, so that I can audit the source without reverse-engineering a tokenizer pipeline. + +### Researcher + +As a researcher, I can compare fixed windows, paragraph-only units, semantic spans, and hierarchical retrieval on the same gold queries, so that a chunking claim is supported by retrieval evidence rather than intuition. + +### Platform operator + +As a platform operator, I can change a model profile without changing segmentation code, and the system refuses unverified limits or tokenizers, so that provider drift cannot create silent overflows. + +### Downstream consumer + +As a semantic-data-portal or LineageWeave consumer, I can retrieve a precise leaf and its parent/neighbor graph while preserving `embedding_space_id`, so that cross-model and cross-context comparisons remain valid. + +## 7. Scope sequencing + +### P0 — Contract and deterministic baseline + +- exact typed source blocks; +- model profiles and token-count port; +- final-payload budget gate; +- deterministic micro-units and recursive fallback; +- hierarchy and provenance; +- fixed-window and paragraph-only baselines; +- hostile Unicode and mixed-script tests. + +### P1 — Dense semantic refinement + +- adjacent-unit similarity port; +- boundary-score calibration; +- deterministic caching; +- retrieval benchmark and threshold selection; +- structure-only fallback. + +### P2 — Optional LLM and advanced context + +- evidence-bounded LLM boundary proposals and verifier; +- versioned section/document summaries; +- direct-vs-orchestrated ablations; +- open-weight/token-level late-chunking experiments where the model contract permits them. + +True late chunking is not claimed for `text-embedding-3-large` because the API does not expose the token-level contextual states and pooling control required by that technique. + +## 8. Data requirements + +Normalized two-or-more-word `snake_case` objects are proposed: + +- `source_block`; +- `semantic_unit`; +- `semantic_unit_relation`; +- `embedding_model_profile`; +- `embedding_payload`; +- `embedding_vector_record`; +- `embedding_evaluation_run`. + +The logical model remains in third normal form: + +- source text belongs to the source artifact/document authority; +- units reference source offsets rather than duplicating text; +- model/profile metadata is stored once and referenced; +- vectors bind to an explicit embedding space and payload; +- evaluation runs reference immutable corpora, queries, policies, and profiles. + +## 9. Release claim boundary + +Acceptance of this PRD delta would authorize implementation planning only. It would not establish: + +- support for every language; +- cross-language measurement equivalence; +- retrieval superiority; +- a production-ready OpenAI connector; +- a completed vector database; +- a released feature. + +Those claims require the acceptance evidence in this document and ADR 0014 promotion. \ No newline at end of file diff --git a/docs/research/language-agnostic-semantic-span-embedding.md b/docs/research/language-agnostic-semantic-span-embedding.md new file mode 100644 index 00000000..d61d17bf --- /dev/null +++ b/docs/research/language-agnostic-semantic-span-embedding.md @@ -0,0 +1,40 @@ +# Language-agnostic semantic-span embedding research traceability + +## Decision summary + +TEPP should not ask “what language is this?” before it can preserve evidence, form candidate meaning units, or enforce an embedding-model context limit. Source structure, Unicode-safe offsets, final rendered-payload token counts, optional dense semantic changes, and explicit hierarchy provide the base contract. + +Language metadata remains important for measurement invariance, fairness, lexical analysis, and result stratification. It is not the switch that decides whether core processing works. + +## Evidence-to-decision map + +| Evidence | Supported decision | Claim boundary | +|---|---|---| +| OpenAI embeddings documentation | `text-embedding-3-large` initial profile: 8,192-token maximum, 3,072 default dimensions, `dimensions` support; third-generation token estimation uses `cl100k_base` | Provider documentation, not a universal model limit or a guarantee that the model never changes | +| Unicode Standard Annex #29, Revision 47 | Unicode-safe default grapheme/word/sentence boundary guidance | Default boundaries are not semantic truth and are not sufficient for every script or locale | +| Chen et al. (2024), Dense X Retrieval | Retrieval-unit granularity affects retrieval and QA; proposition-scale units are an empirical alternative to passages | Does not prove one granularity is best for every corpus, language, or embedding model | +| Günther et al. (2024), Late Chunking | Independent short chunks can lose global context; token-level long-context models can pool contextualized chunk representations | The OpenAI embeddings API does not expose contextual token states/pooling, so TEPP does not claim true late chunking for `text-embedding-3-large` | +| TEPP approved PRD v0.4 | Exact source spans, multilingual evidence, optional LLM semantic units, hierarchical/relational evidence, no TF-IDF/BM25 inferential weighting | The approved baseline does not by itself prove the new segmentation implementation | + +## Design implications + +1. **No language gate.** Missing, mixed, or wrong language metadata cannot cause a different base algorithm. +2. **No character-count fallback.** Token limits are model/tokenizer facts. +3. **Count the final payload.** Metadata and separators consume tokens. +4. **Use hierarchy, not only overlap.** Leaves retrieve precisely; parent and neighbor relations restore context. +5. **Treat granularity as an empirical parameter.** Compare fixed, paragraph, structural, semantic, and hierarchical alternatives. +6. **Keep late chunking experimental by model capability.** It requires token-level contextual states and pooling control. +7. **Do not use TF-IDF or BM25 in boundary scoring.** Dense similarity is optional; structure-only remains deterministic. +8. **Preserve source evidence.** Summaries and embeddings are derived artifacts. + +## APA 7 references + +Chen, T., Wang, H., Chen, S., Yu, W., Ma, K., Zhao, X., Zhang, H., & Yu, D. (2024). Dense X retrieval: What retrieval granularity should we use? In *Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing* (pp. 15159–15177). Association for Computational Linguistics. https://doi.org/10.18653/v1/2024.emnlp-main.845 + +Günther, M., Mohr, I., Williams, D. J., Wang, B., & Xiao, H. (2024). *Late chunking: Contextual chunk embeddings using long-context embedding models* (arXiv:2409.04701). arXiv. https://doi.org/10.48550/arXiv.2409.04701 + +OpenAI. (2026). *How can I tell how many tokens a string will have before I try to embed it?* OpenAI Help Center. Retrieved August 15, 2026, from https://help.openai.com/en/articles/8984337-how-can-i-tell-how-many-tokens-a-string-will-have-before-i-try-to-embed-it + +OpenAI. (2026). *Vector embeddings*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/guides/embeddings + +Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html \ No newline at end of file diff --git a/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md b/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md new file mode 100644 index 00000000..2af9f8c7 --- /dev/null +++ b/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md @@ -0,0 +1,556 @@ +# Language-Agnostic Semantic Span Embedding Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Build a language-independent, exact-span, model-budgeted, hierarchical semantic-unit pipeline whose final `text-embedding-3-large` payload never exceeds 8,192 `cl100k_base` tokens and whose provider-specific behavior remains behind ports. + +**Architecture:** `evidence_core` owns immutable typed source blocks; a new `semantic_preprocessor` crate owns micro-units, model profiles, token budgeting, recursive splitting, span packing, and hierarchy. `tepp_api` owns versioned interchange. Provider, batch, vector-store, and optional LLM behavior enters through narrow adapters. + +**Tech Stack:** Rust 1.97.1 workspace, CPU `f64` reference behavior where numeric similarity is computed, `serde` wire DTOs, property/fuzz testing, pinned tokenizer adapter, optional contextual-orchestrator/pg-llm-batch/semantic-data-portal ports. + +## Global Constraints + +- Language identity MUST NOT select the base segmentation, packing, overflow, or fallback algorithm. +- TF-IDF, BM25, stopword deletion, stemming, morphology, and translation MUST NOT be required by the base pipeline. +- The complete rendered payload MUST be token-counted and MUST NOT exceed the model profile. +- The initial `text-embedding-3-large` profile uses `max_input_tokens = 8192`, `tokenizer_profile = cl100k_base`, and `default_dimensions = 3072`. +- Silent truncation is prohibited. +- Exact byte and Unicode-scalar source offsets are preserved. +- Database objects use two-or-more-word `snake_case` names and remain in third normal form. +- Production statement and branch coverage and public/safety docstrings remain 100%. +- LLM live tests use `NVIDIA_NIM_API_KEY`; `COPILOT_GITHUB_TOKEN` is prohibited. +- No task may claim production readiness, cross-language equivalence, or retrieval superiority without the defined evidence. + +--- + +### Task 1: Replace paragraph-only scope with accepted contracts + +**Files:** +- Create: `docs/adr/0017-language-agnostic-semantic-span-budgeting.md` +- Create: `docs/product/prd-v0.5-proposed.md` +- Create: `docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md` +- Modify: `docs/adr/README.md` +- Modify: `DOCUMENTATION.md` +- Modify: `docs/TRACEABILITY.md` +- Modify: `CHANGELOG.md` +- Delete from superseded branch only: `crates/evidence_core/src/semantic.rs` +- Delete from superseded branch only: `crates/evidence_core/tests/semantic_unit_contract.rs` + +**Interfaces:** +- Consumes: approved PRD v0.4, ADR 0004, ADR 0008, ADR 0011, ADR 0012. +- Produces: ADR 0017 and proposed PRD v0.5 requirements for all later tasks. + +- [ ] **Step 1: Add the documentation-contract test cases** + +Extend the documentation validator fixture so ADR 0017 must appear exactly once and every required section is present. + +- [ ] **Step 2: Run the documentation validator and confirm RED** + +Run: + +```bash +python3 scripts/validate_documentation.py +``` + +Expected: failure because ADR 0017 and the index row are absent. + +- [ ] **Step 3: Add ADR, PRD delta, design, index, traceability, and changelog** + +Use the exact documents approved in the design PR. Mark decision status and implementation maturity independently. + +- [ ] **Step 4: Remove paragraph-only production claims** + +Remove the `semantic_paragraph_units` / `refuse_document_bag_of_words` implementation and its claim rows from the superseded PR branch. Paragraph parsing returns later as one `SourceBlockKind`, not the semantic-unit authority. + +- [ ] **Step 5: Verify documentation contracts** + +Run: + +```bash +python3 scripts/validate_documentation.py +python3 scripts/check_docstrings.py +python3 -m coverage run --branch -m unittest discover -s tests/quality -p 'test_*.py' +python3 -m coverage report --fail-under=100 --show-missing +``` + +Expected: all pass, including 100% statement and branch coverage for quality scripts. + +- [ ] **Step 6: Commit** + +```bash +git add docs DOCUMENTATION.md CHANGELOG.md crates/evidence_core +git commit -m "docs(evidence): define language-agnostic semantic spans" +``` + +### Task 2: Add typed source blocks to `evidence_core` + +**Files:** +- Create: `crates/evidence_core/src/block.rs` +- Modify: `crates/evidence_core/src/lib.rs` +- Modify: `crates/evidence_core/src/error.rs` +- Test: `crates/evidence_core/tests/source_block_contract.rs` + +**Interfaces:** +- Consumes: `DocumentRecord`, `SourceSpan`, `EvidenceId`. +- Produces: + - `SourceBlockKind`; + - `SourceBlock::new(document, kind, span, parent, ordinal)`; + - accessors for block identity, kind, span, parent, ordinal. + +- [ ] **Step 1: Write failing exact-block tests** + +Cover headings, paragraph, list, table-row group, code, caption, dialogue, DOM, and unknown blocks. Assert cross-document parents, duplicate ordinals, zero-length spans, and invalid ownership fail closed. + +- [ ] **Step 2: Run focused tests and confirm RED** + +```bash +cargo test -p evidence_core --test source_block_contract --offline +``` + +Expected: compile failure for missing `SourceBlock` APIs. + +- [ ] **Step 3: Implement `SourceBlockKind` and `SourceBlock`** + +Store validated domain fields privately. Do not store a second authoritative text copy. + +- [ ] **Step 4: Add stable typed errors and docstrings** + +Add content-redacting errors for cross-document parentage and invalid source-block order. + +- [ ] **Step 5: Run focused and full crate checks** + +```bash +cargo fmt --all -- --check +cargo clippy -p evidence_core --all-targets --offline -- -D warnings +cargo test -p evidence_core --offline +cargo llvm-cov -p evidence_core --offline --lib --tests --fail-under-lines 100 --fail-under-regions 100 +``` + +Expected: all pass at 100%. + +- [ ] **Step 6: Commit** + +```bash +git add crates/evidence_core +git commit -m "feat(evidence): add typed exact-span source blocks" +``` + +### Task 3: Create embedding model profiles and token-count port + +**Files:** +- Create: `crates/semantic_preprocessor/Cargo.toml` +- Create: `crates/semantic_preprocessor/src/lib.rs` +- Create: `crates/semantic_preprocessor/src/profile.rs` +- Create: `crates/semantic_preprocessor/src/token.rs` +- Create: `crates/semantic_preprocessor/tests/model_profile_contract.rs` +- Modify: `Cargo.toml` +- Modify: `Cargo.lock` + +**Interfaces:** +- Consumes: `EvidenceId`, `ContentDigest` from `evidence_core`. +- Produces: + - `EmbeddingModelProfile`; + - `EmbeddingInputRole`; + - `TokenCounter` trait; + - `TokenOffset`; + - typed `SemanticPreprocessorError`. + +- [ ] **Step 1: Write failing model-profile tests** + +Assert: + +```rust +assert_eq!(profile.max_input_tokens(), 8192); +assert_eq!(profile.tokenizer_profile(), "cl100k_base"); +assert_eq!(profile.default_dimensions(), 3072); +``` + +Also reject zero limits, zero dimensions, empty provider/model/revision, noncanonical tokenizer digests, and requested dimensions above the default unless the profile explicitly allows them. + +- [ ] **Step 2: Write failing fake token-counter tests** + +The deterministic fake counter maps Unicode scalar boundaries to token offsets and refuses offsets that split UTF-8. + +- [ ] **Step 3: Run focused tests and confirm RED** + +```bash +cargo test -p semantic_preprocessor --test model_profile_contract --offline +``` + +Expected: crate or API missing. + +- [ ] **Step 4: Implement profile and port with no provider SDK** + +Keep model facts in constructors/fixtures and immutable DTOs. Domain code depends only on the trait. + +- [ ] **Step 5: Run crate quality gates** + +```bash +cargo fmt --all -- --check +cargo clippy -p semantic_preprocessor --all-targets --offline -- -D warnings +cargo test -p semantic_preprocessor --offline +cargo llvm-cov -p semantic_preprocessor --offline --lib --tests --fail-under-lines 100 --fail-under-regions 100 +``` + +- [ ] **Step 6: Commit** + +```bash +git add Cargo.toml Cargo.lock crates/semantic_preprocessor +git commit -m "feat(semantic): add versioned embedding model profiles" +``` + +### Task 4: Build language-independent micro-units + +**Files:** +- Create: `crates/semantic_preprocessor/src/micro_unit.rs` +- Create: `crates/semantic_preprocessor/src/unicode_boundary.rs` +- Test: `crates/semantic_preprocessor/tests/micro_unit_contract.rs` +- Test: `crates/semantic_preprocessor/tests/fixtures/mixed_script_cases.json` + +**Interfaces:** +- Consumes: ordered `SourceBlock` values. +- Produces: + - `MicroUnit`; + - `MicroUnitBuilder::build(&[SourceBlock])`; + - boundary reasons with exact offsets. + +- [ ] **Step 1: Write failing mixed-script tests** + +Include Korean/English code switching, Chinese, Japanese, Thai, Arabic RTL, emoji ZWJ, combining marks, code, table cells, and long unpunctuated text. Call the same builder without a language argument. + +- [ ] **Step 2: Add a static source check for language branching** + +The quality test scans `semantic_preprocessor` production source and fails on a base API parameter or branch named `language_code`, `locale_code`, or equivalent selector in micro-unit/packing modules. Metadata DTOs are excluded from this narrow check. + +- [ ] **Step 3: Run focused tests and confirm RED** + +```bash +cargo test -p semantic_preprocessor --test micro_unit_contract --offline +``` + +- [ ] **Step 4: Implement structure and Unicode-safe boundary hints** + +Use source-block children first, then Unicode-safe sentence/line hints. Preserve unknown content. Do not call morphology, stopword, TF-IDF, BM25, or translation logic. + +- [ ] **Step 5: Add property tests** + +Generate arbitrary valid Unicode strings and assert monotonic, nonoverlapping, reconstructable offsets and no panics. + +- [ ] **Step 6: Run quality gates and commit** + +```bash +cargo fmt --all -- --check +cargo clippy -p semantic_preprocessor --all-targets --offline -- -D warnings +cargo test -p semantic_preprocessor --offline +cargo llvm-cov -p semantic_preprocessor --offline --lib --tests --fail-under-lines 100 --fail-under-regions 100 +git add crates/semantic_preprocessor +git commit -m "feat(semantic): build language-independent micro-units" +``` + +### Task 5: Enforce final-payload budgets and recursive recovery + +**Files:** +- Create: `crates/semantic_preprocessor/src/payload.rs` +- Create: `crates/semantic_preprocessor/src/budget.rs` +- Create: `crates/semantic_preprocessor/src/split.rs` +- Test: `crates/semantic_preprocessor/tests/budget_contract.rs` + +**Interfaces:** +- Consumes: `MicroUnit`, `EmbeddingModelProfile`, `TokenCounter`. +- Produces: + - `PayloadTemplate`; + - `RenderedEmbeddingPayload`; + - `SemanticSpanPolicy`; + - `BudgetedUnit`; + - `RecursiveSplitter`. + +- [ ] **Step 1: Write the 8,192/8,193 boundary tests** + +Use the fake token counter to assert the complete metadata-rendered payload at 8,192 tokens is accepted and 8,193 is split or refused before a provider call. + +- [ ] **Step 2: Write recursive recovery tests** + +Cover child block, sentence/line, punctuation, token-offset fallback, bounded overlap, and an unsplittable invalid-offset case. + +- [ ] **Step 3: Run focused tests and confirm RED** + +```bash +cargo test -p semantic_preprocessor --test budget_contract --offline +``` + +- [ ] **Step 4: Implement deterministic payload rendering** + +Version the template, omit absent metadata, preserve source content, and hash the exact rendered bytes. + +- [ ] **Step 5: Implement recursive splitting and typed errors** + +Never truncate. Token-offset fallback maps only to valid source boundaries. + +- [ ] **Step 6: Run quality gates and commit** + +```bash +cargo fmt --all -- --check +cargo clippy -p semantic_preprocessor --all-targets --offline -- -D warnings +cargo test -p semantic_preprocessor --offline +cargo llvm-cov -p semantic_preprocessor --offline --lib --tests --fail-under-lines 100 --fail-under-regions 100 +git add crates/semantic_preprocessor +git commit -m "feat(semantic): enforce final embedding payload budgets" +``` + +### Task 6: Add semantic packing and explicit degradation + +**Files:** +- Create: `crates/semantic_preprocessor/src/similarity.rs` +- Create: `crates/semantic_preprocessor/src/packer.rs` +- Test: `crates/semantic_preprocessor/tests/semantic_packer_contract.rs` + +**Interfaces:** +- Consumes: `BudgetedUnit`, `AdjacentSimilarity`, `SemanticSpanPolicy`. +- Produces: + - `SemanticSpanPacker`; + - `BoundaryMode`; + - `BoundaryReason`; + - packed leaf units. + +- [ ] **Step 1: Write failing structure-only tests** + +Mandatory heading/table/code boundaries split deterministically. Similarity is not required. + +- [ ] **Step 2: Write failing dense-boundary tests** + +A fake similarity port returns known adjacent scores. Assert a sharp semantic drop creates a boundary without a language feature. + +- [ ] **Step 3: Write failing degradation tests** + +When similarity returns `Unavailable`, assert packing succeeds in `StructureOnly` mode and records the reason. Invalid scores (`NaN`, infinity, wrong length) fail closed. + +- [ ] **Step 4: Run focused tests and confirm RED** + +```bash +cargo test -p semantic_preprocessor --test semantic_packer_contract --offline +``` + +- [ ] **Step 5: Implement the packer** + +Recount the final payload after every accepted merge and again before emit. Do not average embeddings from different model spaces. + +- [ ] **Step 6: Run quality gates and commit** + +```bash +cargo fmt --all -- --check +cargo clippy -p semantic_preprocessor --all-targets --offline -- -D warnings +cargo test -p semantic_preprocessor --offline +cargo llvm-cov -p semantic_preprocessor --offline --lib --tests --fail-under-lines 100 --fail-under-regions 100 +git add crates/semantic_preprocessor +git commit -m "feat(semantic): pack spans with explicit fallback" +``` + +### Task 7: Build leaf/section/document hierarchy and wire DTOs + +**Files:** +- Create: `crates/semantic_preprocessor/src/hierarchy.rs` +- Create: `crates/semantic_preprocessor/tests/hierarchy_contract.rs` +- Create: `crates/tepp_api/src/semantic_span.rs` +- Create: `schemas/semantic-span-manifest-v1.schema.json` +- Create: `examples/semantic-span-manifest-v1.json` +- Modify: `crates/tepp_api/src/lib.rs` +- Test: `crates/tepp_api/tests/semantic_span_wire_contract.rs` + +**Interfaces:** +- Consumes: packed leaf units, source-block tree, model profile. +- Produces: + - `SemanticUnitHierarchy`; + - `SemanticSpanManifestV1`; + - JSON schema and strict wire reconstruction. + +- [ ] **Step 1: Write failing hierarchy tests** + +Assert one parent per leaf, same-document ownership, acyclic parent links, exact previous/next symmetry, stable order, and bounded parent/neighbor expansion. + +- [ ] **Step 2: Write failing wire tests** + +Reject unknown fields, unsupported versions, cross-document relations, wrong payload digests, unknown model profiles, and vector records without `embedding_space_id`. + +- [ ] **Step 3: Run focused tests and confirm RED** + +```bash +cargo test -p semantic_preprocessor --test hierarchy_contract --offline +cargo test -p tepp_api --test semantic_span_wire_contract --offline +``` + +- [ ] **Step 4: Implement hierarchy and DTOs** + +Summaries are optional derived artifacts. They cannot replace source spans or become source truth. + +- [ ] **Step 5: Validate schema and run quality gates** + +```bash +cargo fmt --all -- --check +cargo clippy --workspace --all-targets --all-features -- -D warnings +cargo nextest run --workspace --all-features +cargo test --doc --workspace --all-features +python3 scripts/validate_documentation.py +``` + +- [ ] **Step 6: Commit** + +```bash +git add crates/semantic_preprocessor crates/tepp_api schemas examples +git commit -m "feat(api): add semantic span hierarchy contracts" +``` + +### Task 8: Add provider adapters without moving authority + +**Files:** +- Create: `crates/tepp_api/src/embedding_port.rs` +- Create: `docs/connectors/pg-llm-batch-semantic-spans.md` +- Create: `docs/connectors/contextual-orchestrator-semantic-spans.md` +- Create: `docs/connectors/semantic-data-portal-semantic-spans.md` +- Create: `docs/connectors/embedrelay-semantic-spans.md` +- Test: `crates/tepp_api/tests/embedding_port_contract.rs` + +**Interfaces:** +- Consumes: validated manifest and rendered payload. +- Produces: + - provider-neutral `EmbeddingPort`; + - token-count adapter contract; + - optional LLM refinement contract; + - vector-store handoff; + - embedding-migration handoff. + +- [ ] **Step 1: Write failing port tests** + +Assert no API accepts provider credentials in DTOs, no adapter can change unit identity, and vector results require profile, payload hash, dimensions, and `embedding_space_id`. + +- [ ] **Step 2: Run focused tests and confirm RED** + +```bash +cargo test -p tepp_api --test embedding_port_contract --offline +``` + +- [ ] **Step 3: Implement ports and connector documents** + +Keep all provider SDKs out of domain crates. Use HTTPS and host authorization boundaries in actual service adapters. + +- [ ] **Step 4: Add deterministic mock integration** + +Run source blocks through fake token count, fake similarity, fake embedding, and manifest export. Assert provider failure does not mutate span identity. + +- [ ] **Step 5: Run quality gates and commit** + +```bash +cargo fmt --all -- --check +cargo clippy --workspace --all-targets --all-features -- -D warnings +cargo nextest run --workspace --all-features +cargo test --doc --workspace --all-features +git add crates/tepp_api docs/connectors +git commit -m "feat(api): add semantic embedding ecosystem ports" +``` + +### Task 9: Build realistic retrieval and robustness evaluation + +**Files:** +- Create: `crates/tepp_simulation/src/semantic_span_truth.rs` +- Create: `crates/validation_core/src/retrieval.rs` +- Create: `crates/validation_core/src/span_boundary.rs` +- Create: `crates/validation_core/tests/semantic_span_recovery.rs` +- Create: `docs/validation/semantic-span-benchmark.md` +- Create: `.github/workflows/semantic-span-study.yml` + +**Interfaces:** +- Consumes: policies, gold source blocks, queries, relevant spans, hierarchy, retrieval results. +- Produces: + - overflow/truncation report; + - boundary precision/recall/span F1; + - Recall@k, nDCG@k, MRR; + - duplicate/context-restoration metrics; + - stratified uncertainty report. + +- [ ] **Step 1: Write deterministic truth-corpus recovery tests** + +Generate known boundaries and query relevance under mixed scripts, duplicated boilerplate, translation/revision groups, and long-unit stress. + +- [ ] **Step 2: Write metric tests with hand-calculated examples** + +Cover ties, zero relevant documents, duplicate hits, malformed ranks, and confidence intervals. + +- [ ] **Step 3: Run focused tests and confirm RED** + +```bash +cargo test -p validation_core --test semantic_span_recovery --offline +``` + +- [ ] **Step 4: Implement metrics in Rust** + +Use stable `f64` calculations and content-redacting reports. Do not infer quality from cosine score alone. + +- [ ] **Step 5: Add scheduled live study** + +The workflow uses `NVIDIA_NIM_API_KEY` only when an approved provider-backed evaluation is enabled. It never uses `COPILOT_GITHUB_TOKEN`, never skips a required live lane silently, and publishes immutable evaluation artifacts. + +- [ ] **Step 6: Run full acceptance suite** + +```bash +cargo fmt --all -- --check +cargo clippy --workspace --all-targets --all-features -- -D warnings +cargo nextest run --workspace --all-features +cargo test --doc --workspace --all-features +cargo llvm-cov --workspace --all-features --fail-under-lines 100 --fail-under-regions 100 +python3 scripts/check_workspace_contract.py +python3 scripts/check_docstrings.py +python3 scripts/validate_documentation.py +cargo deny check +``` + +Expected: all local deterministic gates pass; live study remains a separate exact-head evidence lane. + +- [ ] **Step 7: Commit** + +```bash +git add crates/tepp_simulation crates/validation_core docs/validation .github/workflows +git commit -m "test(semantic): validate retrieval and boundary recovery" +``` + +### Task 10: Final evidence, review, and claim boundary + +**Files:** +- Modify: `docs/TRACEABILITY.md` +- Modify: `docs/validation/temporal-event-foundation.md` +- Modify: `CHANGELOG.md` +- Modify: `docs/research/standards-and-literature.md` +- Generate: release SBOM/provenance/checksum artifacts through existing tooling. + +**Interfaces:** +- Consumes: exact-head deterministic and live evaluation results. +- Produces: reviewable implementation/scientific maturity decision. + +- [ ] **Step 1: Update traceability with exact evidence** + +Mark each capability `implemented-main`, `active-PR`, `partial`, or `accepted-target` truthfully. + +- [ ] **Step 2: Run release evidence tools** + +```bash +python3 scripts/release_evidence.py generate +python3 scripts/release_evidence.py validate +cargo deny check +``` + +- [ ] **Step 3: Run all exact-head checks** + +Do not substitute earlier-head, local-only, queued, skipped, or cancelled evidence for required current-head checks. + +- [ ] **Step 4: Obtain independent review** + +Resolve actionable review threads, re-run affected tests, and require an independent approval under ADR 0015. + +- [ ] **Step 5: Promote claims only when evidence permits** + +A merged implementation may claim zero-overflow and exact-span behavior only after the hard gates pass. It may claim retrieval improvement or language-profile robustness only after the benchmark evidence passes under ADR 0014. + +- [ ] **Step 6: Commit final evidence updates** + +```bash +git add docs CHANGELOG.md +git commit -m "docs(semantic): record semantic span acceptance evidence" +``` \ No newline at end of file diff --git a/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md b/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md new file mode 100644 index 00000000..fa33cf16 --- /dev/null +++ b/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md @@ -0,0 +1,396 @@ +# Language-Agnostic Semantic Span Embedding Design + +**Status:** Proposed design +**Repository owner:** `ContextualWisdomLab/TEPP` +**Date:** 2026-08-15 +**Architecture authority:** ADR 0017 +**Product authority:** proposed PRD v0.5 delta + +## 1. Design decision + +TEPP is the canonical owner because semantic units are fallible, exact-span observations feeding its multilingual evidence, topic, event, and psychometric layers. This is not primarily an embedding-space migration concern, an LLM routing concern, a batch-provider concern, or a vector-catalog concern. + +The implementation is split so that TEPP can also operate as a reusable module: + +```text +source adapters + -> evidence_core + -> semantic_preprocessor + -> embedding ports + -> vector/graph consumers + -> TEPP estimators +``` + +Adjacent CWL systems connect through ports: + +```text +contextual-orchestrator -- optional LLM proposals / summaries / verification +pg-llm-batch -- optional batch transport and pg_tiktoken adapter +semantic-data-portal -- optional vector + relation persistence/search +EmbedRelay -- later embedding-space migration +``` + +## 2. Why PR #56 is not the target design + +Draft PR #56 equates semantic units with `text.split("\n\n")` and verifies two English paragraphs. That implementation preserves offsets but misses the governing requirements: + +- language-independent correctness; +- final payload token counting; +- a versioned embedding-model profile; +- structural units beyond paragraphs; +- recursive recovery for one oversized paragraph; +- semantic-boundary evidence; +- hierarchy and context restoration; +- mixed-script and hostile Unicode tests; +- provider and ecosystem ownership boundaries. + +The PR therefore must be superseded rather than expanded by small patches around the same public API. + +## 3. Architecture + +### 3.1 `evidence_core` + +Owns immutable source evidence only. + +Proposed interfaces: + +```rust +pub enum SourceBlockKind { + DocumentTitle, + SectionHeading, + ParagraphBlock, + ListItem, + TableRowGroup, + CodeBlock, + CaptionBlock, + DialogueTurn, + DomRegion, + UnknownBlock, +} + +pub struct SourceBlock { + block_id: EvidenceId, + document_id: EvidenceId, + block_kind: SourceBlockKind, + source_span: SourceSpan, + parent_block_id: Option, + ordinal_index: u32, +} +``` + +Invariants: + +- exact byte and Unicode-scalar offsets; +- source reconstruction; +- no text duplication as authority; +- parent belongs to the same document; +- ordinal ordering is total within a parent; +- unknown blocks are retained, not discarded. + +### 3.2 `semantic_preprocessor` + +Owns model-independent span construction and budget-aware packing. + +Proposed interfaces: + +```rust +pub struct EmbeddingModelProfile { + profile_id: EvidenceId, + provider_code: String, + model_identifier: String, + observed_revision: String, + tokenizer_profile: String, + tokenizer_digest: ContentDigest, + max_input_tokens: u32, + default_dimensions: u32, + requested_dimensions: Option, + input_role: EmbeddingInputRole, + metadata_template_version: String, + verified_at: Timestamp, +} + +pub trait TokenCounter { + type Error; + + fn count_tokens( + &self, + profile: &EmbeddingModelProfile, + payload: &str, + ) -> Result; + + fn token_offsets( + &self, + profile: &EmbeddingModelProfile, + payload: &str, + ) -> Result, Self::Error>; +} + +pub trait AdjacentSimilarity { + type Error; + + fn similarities( + &self, + units: &[MicroUnit], + ) -> Result, Self::Error>; +} + +pub struct SemanticSpanPolicy { + target_input_tokens: u32, + preferred_max_input_tokens: u32, + metadata_reserve_tokens: u32, + minimum_unit_tokens: u32, + maximum_overlap_tokens: u32, + semantic_drop_threshold: f64, +} + +pub enum BoundaryMode { + StructureOnly, + StructureAndDenseSimilarity, + StructureDenseAndLlmVerified, +} + +pub struct SemanticUnit { + unit_id: EvidenceId, + unit_level: SemanticUnitLevel, + source_spans: Vec, + parent_unit_id: Option, + previous_unit_id: Option, + next_unit_id: Option, + token_count: u32, + payload_digest: ContentDigest, + boundary_mode: BoundaryMode, + boundary_reasons: Vec, +} +``` + +The exact domain types can be adjusted to repository conventions, but the responsibilities and invariants are fixed. + +### 3.3 `tepp_api` + +Owns versioned DTOs and ports, not provider credentials. + +Data-plane operations: + +```text +POST /v1/semantic-span-runs +GET /v1/semantic-span-runs/{run_id} +GET /v1/semantic-span-runs/{run_id}/manifest +POST /v1/embedding-model-profiles/validate +POST /v1/semantic-span-evaluations +``` + +The API does not expose raw provider secrets or direct database handles. Bulk unit/vector interchange uses Arrow IPC or Parquet only after a schema version is fixed. + +## 4. Data flow + +```text +1. Ingest immutable source +2. Parse typed source blocks +3. Build Unicode-safe micro-units +4. Load verified embedding-model profile +5. Render candidate payload with title/heading metadata +6. Count final payload tokens +7. Merge adjacent units under structure + budget + optional similarity +8. Recursively split any oversized unit +9. Validate exact offsets and final token count +10. Build leaf / section / document hierarchy +11. Produce immutable span manifest +12. Call embedding provider through a port +13. Persist vector with embedding_space_id and payload hash +14. Evaluate retrieval and context restoration +``` + +No language detection step is required. A language/profile classifier may run in parallel and attach metadata for evaluation or psychometric invariance. + +## 5. Payload template + +A payload is deterministic and versioned. The P0 template is intentionally small: + +```text +[document_title] +{title} + +[heading_path] +{heading_1} > {heading_2} + +[content] +{source_text} +``` + +Rules: + +- omit an absent field rather than render `null`; +- normalize template line endings without rewriting source content; +- source text remains exact inside the content slot; +- metadata values are themselves exact source or approved metadata references; +- count the complete rendered payload; +- bind `metadata_template_version` and payload digest to the vector record. + +## 6. Packing algorithm + +Pseudocode: + +```text +micro_units = build_micro_units(source_blocks) + +for unit in micro_units: + if final_payload_tokens(unit) > hard_limit: + emit(recursive_split(unit)) + continue + + candidate = current + unit + if mandatory_structure_break(current, unit): + emit(current) + current = unit + elif final_payload_tokens(candidate) > preferred_max: + emit(current) + current = unit + elif semantic_drop(current.last, unit) >= threshold: + emit(current) + current = unit + else: + current = candidate + +emit(current) + +for emitted_span: + assert final_payload_tokens(emitted_span) <= hard_limit +``` + +`preferred_max` controls retrieval granularity. `hard_limit` comes from the model profile. The algorithm never assumes that filling all 8,192 tokens is desirable. + +## 7. Recursive split + +```text +split_by_child_blocks + -> split_by_unicode_sentence_or_line_boundary + -> split_by_punctuation_boundary + -> split_by_token_offsets_with_bounded_overlap + -> fail_closed +``` + +Every split produces exact source offsets. A token-offset adapter must map offsets back to valid UTF-8/Unicode-scalar boundaries. It may not split inside a code point, grapheme cluster where avoidable, or invalid byte sequence. + +## 8. Hierarchical retrieval + +Query flow: + +```text +query embedding + -> document/section candidate routing + -> leaf retrieval + -> rank fusion or rerank + -> parent + previous/next expansion + -> bounded context package +``` + +The system returns: + +- leaf evidence; +- parent heading path; +- bounded previous/next evidence; +- source offsets; +- vector/model/profile provenance; +- exact token and expansion budgets. + +This replaces large fixed overlaps with explicit graph restoration. + +## 9. Error handling + +| Error | Behavior | +|---|---| +| model profile missing/unverified | fail closed before segmentation run | +| tokenizer unavailable | fail closed; do not estimate by characters | +| similarity provider unavailable | continue structure-only and record degradation | +| LLM refinement unavailable | continue without LLM | +| final payload over hard limit | recursive split, then fail closed if impossible | +| invalid source offsets | reject unit and run | +| provider timeout/rate limit | bounded retry/fallback without changing unit identity | +| mixed/unknown language | normal path; metadata may remain unresolved | + +No error message echoes source text, API keys, vector values, or provider response bodies. + +## 10. Security + +- content is untrusted and cannot issue instructions; +- no active content, script, or external resource is executed; +- provider destinations are allowlisted by the host; +- secrets are resolved outside the domain core; +- PII is purpose-bound rather than blanket-masked; +- embedding vectors and payloads are restricted data; +- every derived artifact records source digest, policy, model profile, tokenizer digest, and code commit; +- LLM proposals require exact evidence identifiers and deterministic validation. + +## 11. Evaluation design + +### 11.1 Corpora + +Build a licensed or synthetic benchmark containing: + +- Korean/English code switching; +- Chinese and Japanese without whitespace dependence; +- Thai; +- Arabic RTL; +- English, French, German, Turkish, Vietnamese, Indonesian; +- emoji, combining marks, zero-width joiners; +- tables, list hierarchies, code, captions, dialogue; +- long unpunctuated spans; +- duplicated boilerplate and templates; +- translation/revision-linked documents. + +Language labels are used only to stratify results. + +### 11.2 Gold evidence + +Human reviewers annotate: + +- mandatory and preferred boundaries; +- self-contained answer spans; +- parent context required to interpret each leaf; +- query-to-evidence relevance; +- duplicate or boilerplate spans. + +### 11.3 Baselines + +- fixed 800-token / 100-token overlap; +- paragraph-only; +- structure + hard budget; +- structure + budget + dense boundary; +- hierarchy + context restoration. + +### 11.4 Metrics + +- overflow and silent truncation; +- exact offset recovery; +- boundary precision/recall and span F1; +- Recall@k, nDCG@k, MRR; +- duplicate-hit rate; +- context-restoration success; +- indexing time, query time, tokens, cost, storage; +- mixed-language performance gap with confidence intervals. + +### 11.5 Scientific claim gate + +A strategy is not promoted because one language or one dataset improves. The report includes uncertainty, per-profile results, failure examples, and practical effect size. A model-specific threshold is versioned and revalidated on model revision. + +## 12. Research position + +Dense X Retrieval shows that retrieval-unit granularity materially affects retrieval and downstream QA, and that proposition-sized units can outperform passage indexing. Late Chunking shows that short units can lose surrounding context and that models exposing contextual token states can pool chunks after encoding. Unicode Standard Annex #29 defines default grapheme, word, and sentence boundary guidance but does not make semantic or language validity claims. + +TEPP therefore: + +- treats retrieval granularity as an empirical choice; +- preserves explicit hierarchy for context restoration in P0; +- keeps true late chunking as an adapter-specific experiment; +- does not claim the closed OpenAI embedding API exposes late-chunking internals; +- validates semantic-unit quality with retrieval and human-gold evidence. + +## 13. APA 7 references + +Chen, T., Wang, H., Chen, S., Yu, W., Ma, K., Zhao, X., Zhang, H., & Yu, D. (2024). Dense X retrieval: What retrieval granularity should we use? In *Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing* (pp. 15159–15177). Association for Computational Linguistics. https://doi.org/10.18653/v1/2024.emnlp-main.845 + +Günther, M., Mohr, I., Williams, D. J., Wang, B., & Xiao, H. (2024). *Late chunking: Contextual chunk embeddings using long-context embedding models* (arXiv:2409.04701). arXiv. https://doi.org/10.48550/arXiv.2409.04701 + +OpenAI. (2026). *Vector embeddings*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/guides/embeddings + +Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html \ No newline at end of file From 1bf86319c4c3d038f7283ad97144d29c06193aea Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sat, 15 Aug 2026 19:55:52 +0900 Subject: [PATCH 2/5] docs(evidence): align semantic-span maturity records Restore the protected-main capability summary, add the proposed semantic-span authority to traceability and the changelog, and correct already-merged persistence and naruon maturity references without promoting the proposal beyond research-only. --- CHANGELOG.md | 3 ++- DOCUMENTATION.md | 2 +- docs/TRACEABILITY.md | 9 +++++---- 3 files changed, 8 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c1cc6e87..ca3a1d47 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ All notable changes to TEPP are documented here. The format follows Keep a Chang ### Added +- Proposed ADR 0017, PRD v0.5 delta, design specification, implementation plan, and APA 7 research traceability for language-agnostic exact-source semantic spans, versioned embedding-model profiles, final rendered-payload token budgets, recursive oversized-unit recovery, hierarchical context restoration, and provider-neutral ecosystem adapters. This records research/design authority only and does not claim implementation or retrieval improvement. - `persistence_postgres` backup/restore integrity: restored snapshots stay unusable until tenant, canonical `SHA-256`, knowledge-cutoff eligibility, temporal window order, and append-only triggers revalidate; SQL probes raise `restore integrity failed` (ADR 0013). - `persistence_postgres` concurrent document-write stress: atomic revise `DO` block that requires exactly one open `system_to` close, SQLSTATE mapping onto `ConcurrentWriteConflict` / `DuplicateDocumentRecord`, and live multi-session insert/revise/append-only proofs. No new migration number. - `tepp_api` naruon HTTP interchange: versioned `https` POST contracts for analysis-run create and modular export authorization that refuse table-access URLs, review/Copilot credential headers, reserved standard-header redefinition, principal-only export idempotency keys, and lexical inference claims (ADR 0011). @@ -95,7 +96,7 @@ All notable changes to TEPP are documented here. The format follows Keep a Chang - Required 100% production line and branch coverage and complete public API docstrings. - Required true-parameter recovery, RMSE, bias, interval coverage, temporal leakage, graph recovery, invariance, and CPU/GPU parity evidence. -- Expanded documentation contracts to require the canonical threat/privacy/assurance/API/orchestration/fitness documents, ADR policy, and every numbered ADR 0001–0016 to remain indexed and structurally complete. +- Expanded documentation contracts to require the canonical threat/privacy/assurance/API/orchestration/fitness documents, ADR policy, and every numbered ADR 0001–0017 to remain indexed and structurally complete. - Added deterministic validation that ADR files and the index have identical decision numbers and that every ADR declares valid decision status, implementation maturity, supersession scope, core decision sections, verification, and rollback behavior. - Added 100% statement and branch coverage for the repository quality-gate scripts. - Made a zero executable-code coverage denominator explicit for the skeleton-only slice rather than treating it as evidence of implemented behavior. diff --git a/DOCUMENTATION.md b/DOCUMENTATION.md index 225986a3..d2810d41 100644 --- a/DOCUMENTATION.md +++ b/DOCUMENTATION.md @@ -61,4 +61,4 @@ The documentation graph is **design-sufficient** when a reviewer can reconstruct It is **protected-main-sufficient** only after the canonical documents are integrated on protected `main`, remain semantically current with live code, and their required exact-head documentation/security/review gates pass. An active documentation PR can therefore be design-sufficient while the protected branch remains documentation-insufficient. -At the time of this proposal, immutable evidence records/exact spans and the Rust workspace quality foundation are implemented on protected main. The language-agnostic semantic-span pipeline, provider model-profile registry, final-payload budget gate, hierarchy, adapters, and retrieval benchmark described in ADR 0017 remain proposed research/design work until independently reviewed, accepted, implemented, and promoted under ADR 0014. \ No newline at end of file +At the time of this proposal, protected `main` contains the immutable evidence and exact-span foundation, Rust quality gates, temporal/event/membership and relation foundations, leakage-safe split and simulation contracts, recovery metrics, versioned API/export contracts, and a partial PostgreSQL persistence and reproducibility chain. The language-agnostic semantic-span pipeline, provider model-profile registry, final-payload budget gate, hierarchy, adapters, and retrieval benchmark described in ADR 0017 remain proposed research/design work until independently reviewed, accepted, implemented, and promoted under ADR 0014. Shared-latent estimation, GPU kernels, TDT/CHRONOS intelligence, longitudinal ESEM/DSEM, visual analytics, production HTTP services, and deployment assurance remain accepted-target or deployment-owned work. diff --git a/docs/TRACEABILITY.md b/docs/TRACEABILITY.md index c29d9743..43147f78 100644 --- a/docs/TRACEABILITY.md +++ b/docs/TRACEABILITY.md @@ -1,7 +1,7 @@ # TEPP Requirements, Research, and Evidence Traceability **Status:** Accepted cross-cutting traceability baseline -**Last reviewed:** 2026-08-13 +**Last reviewed:** 2026-08-15 The full APA 7th standards/literature register remains `docs/research/standards-and-literature.md`. This matrix links durable requirements to their owning decisions and implementation/evidence maturity without duplicating the bibliography. @@ -17,10 +17,11 @@ The full APA 7th standards/literature register remains `docs/research/standards- | time-varying cross-classified multiple membership | PRD; ADR 0003 | `membership_core` network on protected main; multilevel estimators remaining | partial | | leakage-safe availability/cutoff snapshots | PRD; ADR 0002/0013 | `corpus_split` on protected main | implemented-main | | recovery metrics (RMSE, bias, coverage, graph, temporal order, Monte Carlo SE gates) | PRD; Test Strategy; ADR 0007/0014 | `validation_core` on protected main (PR #19); SE-aware Monte Carlo gates included | implemented-main | -| PostgreSQL bitemporal/lineage persistence | ADR 0013; Architecture/ERD | `persistence_postgres` migration contracts, in-memory adapters, live SQL session/document SQL port, tenant RLS (`0002` + session GUC/role helpers), `DATABASE_URL` SQLx gate, optional `live-sqlx` `PgPool` driver, exact-head live PostgreSQL CI with isolation proof, append-only immutability triggers (`0004`), temporal interval ordering CHECKs (`0005`), typed membership assignment (`0006` implemented-main), event-relation/mention/instance SQL (#37–#39 implemented-main), source-artifact SQL (#40 implemented-main), audit-event SQL (#41 implemented-main), concurrent document-write stress (#43 implemented-main), backup/restore integrity revalidation (active PR); remaining physical ERD constraints | partial | +| PostgreSQL bitemporal/lineage persistence | ADR 0013; Architecture/ERD | `persistence_postgres` migration contracts, in-memory adapters, live SQL session/document SQL port, tenant RLS (`0002` + session GUC/role helpers), `DATABASE_URL` SQLx gate, optional `live-sqlx` `PgPool` driver, exact-head live PostgreSQL CI with isolation proof, append-only immutability triggers (`0004`), temporal interval ordering CHECKs (`0005`), typed membership assignment (`0006` implemented-main), event-relation/mention/instance SQL (#37–#39 implemented-main), source-artifact SQL (#40 implemented-main), audit-event SQL (#41 implemented-main), concurrent document-write stress (#43 implemented-main), and backup/restore integrity revalidation (#44 implemented-main); remaining physical ERD constraints | partial | | known-truth temporal/event simulation manifests | PRD; TRD; Test Strategy | `tepp_simulation` on protected main; recovery metrics in `validation_core` | implemented-main | | versioned service/API contracts and exports | PRD; API contract; ADR 0011/0013 | `tepp_api` analysis-run/export/JSON-LD/GraphML contracts on protected main (PR #21); HTTP service remaining accepted-target | partial | | immutable split/run/reproducibility manifests | ADR 0013; ERD | `tepp_api` reproducibility manifest contract on protected main; `persistence_postgres` append-only SQL insert/lookup for `reproducibility_manifest`, `corpus_split_manifest`, `model_run`, and `model_artifact` (migration `0003`); full physical ERD constraints remaining | partial | +| language-agnostic semantic spans and model-budgeted embedding inputs | Proposed PRD v0.5; ADR 0017; ADR 0008/0011 | PR #83 design specification, implementation plan, and research doctoring; no production implementation or retrieval-quality evidence | research-only | | multilingual shared latent semantic space | PRD; ADR 0004 | future semantic/concept/topic crates | accepted-target | | TRSL-TM temporal/relational topic posterior and backend compatibility | ADR 0012; ADR 0004 | future `topic_measurement` | accepted-target | | global P0 topic identity with activity/dormancy/reactivation | ADR 0012 | future topic lineage/activity state | accepted-target | @@ -36,10 +37,10 @@ The full APA 7th standards/literature register remains `docs/research/standards- | purpose-bound PII handling without blanket masking | ADR 0009; `docs/PRIVACY_DATA_GOVERNANCE.md` | future authorization/persistence/export/provider adapters | accepted-target | | tenant/purpose/role/lifetime access and identity separation | ADR 0009; Threat Model | future service/persistence boundaries | accepted-target | | standalone + modular CWL MSA / no cross-service DB coupling | ADR 0011; `docs/API_CONTRACT.md` | current standalone crates; future service ports | partial | -| naruon modular artifact consumer boundary | ADR 0011/0012; API contract | `docs/connectors/naruon-artifact-consumer.md` + PR #22 versioned consumer contract on protected main; `tepp_api` HTTP interchange (active PR); live HTTP service remaining | partial | +| naruon modular artifact consumer boundary | ADR 0011/0012; API contract | `docs/connectors/naruon-artifact-consumer.md` + PR #22 versioned consumer contract on protected main; `tepp_api` HTTP interchange (#42 implemented-main); live HTTP service remaining | partial | | contextual-orchestrator interpretation port boundary | ADR 0010/0011; LLM orchestration | `docs/connectors/contextual-orchestrator-interpretation-port.md`; live port remaining | partial | | Actions registry identities bound to protected-main tree (orphan disable) | Operability; GitHub Actions REST | `scripts/actions_workflow_fleet.py` + issue #20 tests/doctoring; live disable remains operator-authorized | active-PR | -| autonomous model proposal separated from verification/publication/review/merge | ADR 0015 | future safe OpenCode/NVIDIA autonomous-development workflow | accepted-target | +| autonomous model proposal separated from verification/publication/review/merge | ADR 0015 | credential-separated hourly NVIDIA NIM/OpenCode workflow on protected main; independent review and merge authority remain external gates | partial | | contextual-orchestrator execution boundary | ADR 0010/0011 | provider-neutral orchestration port; TEPP retains scientific authority | accepted-target | | foundation validation / release-readiness ledger | ADR 0014; Test Strategy | PR #24 `docs/validation/temporal-event-foundation.md` on protected main | implemented-main | | scientific claim promotion separated from design/implementation/release | ADR 0014; ADR policy | documentation/CI/domain validation/release evidence | partial | From ca587d32b0456c4622629c1c0dd3a17e47c8a2e2 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sat, 15 Aug 2026 22:02:25 +0900 Subject: [PATCH 3/5] docs(research): pin embedding profile evidence and divergence policy --- ...nguage-agnostic-semantic-span-embedding.md | 21 +++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/docs/research/language-agnostic-semantic-span-embedding.md b/docs/research/language-agnostic-semantic-span-embedding.md index d61d17bf..8ac8de33 100644 --- a/docs/research/language-agnostic-semantic-span-embedding.md +++ b/docs/research/language-agnostic-semantic-span-embedding.md @@ -10,12 +10,18 @@ Language metadata remains important for measurement invariance, fairness, lexica | Evidence | Supported decision | Claim boundary | |---|---|---| -| OpenAI embeddings documentation | `text-embedding-3-large` initial profile: 8,192-token maximum, 3,072 default dimensions, `dimensions` support; third-generation token estimation uses `cl100k_base` | Provider documentation, not a universal model limit or a guarantee that the model never changes | +| OpenAI model documentation and the January 2024 embedding-model announcement | `text-embedding-3-large` remains an embeddings model; its full output is up to 3,072 dimensions and the `dimensions` request parameter may shorten the vector | Provider documentation, not a universal vector length or evidence that every shortened dimension has equal retrieval quality | +| Current OpenAI Python SDK embeddings resource | The documented per-input maximum for embedding models is 8,192 tokens; the complete request also has a separate aggregate-token limit | A mutable SDK/API contract. TEPP stores the source URL, retrieval date, and artifact digest and may choose an operational hard limit below the provider maximum | +| OpenAI `tiktoken` model mapping and token-counting cookbook | `text-embedding-3-large` maps to `cl100k_base`; `encoding_for_model` is preferred over scattering the encoding name through product code | Tokenization is model/revision profile data. A provider or tokenizer revision requires a new verified profile rather than silently changing an existing one | | Unicode Standard Annex #29, Revision 47 | Unicode-safe default grapheme/word/sentence boundary guidance | Default boundaries are not semantic truth and are not sufficient for every script or locale | | Chen et al. (2024), Dense X Retrieval | Retrieval-unit granularity affects retrieval and QA; proposition-scale units are an empirical alternative to passages | Does not prove one granularity is best for every corpus, language, or embedding model | | Günther et al. (2024), Late Chunking | Independent short chunks can lose global context; token-level long-context models can pool contextualized chunk representations | The OpenAI embeddings API does not expose contextual token states/pooling, so TEPP does not claim true late chunking for `text-embedding-3-large` | | TEPP approved PRD v0.4 | Exact source spans, multilingual evidence, optional LLM semantic units, hierarchical/relational evidence, no TF-IDF/BM25 inferential weighting | The approved baseline does not by itself prove the new segmentation implementation | +## Provider-profile verification rule + +The model profile distinguishes the provider-documented ceiling from TEPP's operational hard limit. The latter must be less than or equal to the documented ceiling and may reserve a safety margin. Unit and property tests use a deterministic token-counter port to prove exact boundary behavior. A scheduled live provider probe may corroborate the external contract, but a transient provider call is not the only authority for offline segmentation correctness. If the provider rejects a payload that the current profile permits, the adapter fails closed, records the divergence, and requires a new verified profile or a lower operational limit; it does not silently truncate or mutate existing span identity. + ## Design implications 1. **No language gate.** Missing, mixed, or wrong language metadata cannot cause a different base algorithm. @@ -26,6 +32,7 @@ Language metadata remains important for measurement invariance, fairness, lexica 6. **Keep late chunking experimental by model capability.** It requires token-level contextual states and pooling control. 7. **Do not use TF-IDF or BM25 in boundary scoring.** Dense similarity is optional; structure-only remains deterministic. 8. **Preserve source evidence.** Summaries and embeddings are derived artifacts. +9. **Version external facts.** Provider limit, tokenizer mapping, dimensions, source URL, retrieval date, and artifact digest belong to an immutable profile. ## APA 7 references @@ -33,8 +40,14 @@ Chen, T., Wang, H., Chen, S., Yu, W., Ma, K., Zhao, X., Zhang, H., & Yu, D. (202 Günther, M., Mohr, I., Williams, D. J., Wang, B., & Xiao, H. (2024). *Late chunking: Contextual chunk embeddings using long-context embedding models* (arXiv:2409.04701). arXiv. https://doi.org/10.48550/arXiv.2409.04701 -OpenAI. (2026). *How can I tell how many tokens a string will have before I try to embed it?* OpenAI Help Center. Retrieved August 15, 2026, from https://help.openai.com/en/articles/8984337-how-can-i-tell-how-many-tokens-a-string-will-have-before-i-try-to-embed-it +OpenAI. (2024, January 25). *New embedding models and API updates*. https://openai.com/index/new-embedding-models-and-api-updates/ + +OpenAI. (2026). *Embeddings resource* [Source code]. GitHub. Retrieved August 15, 2026, from https://github.com/openai/openai-python/blob/main/src/openai/resources/embeddings.py + +OpenAI. (2026). *How to count tokens with tiktoken* [Jupyter Notebook]. OpenAI Cookbook. Retrieved August 15, 2026, from https://github.com/openai/openai-cookbook/blob/main/examples/How_to_count_tokens_with_tiktoken.ipynb + +OpenAI. (2026). *Model-to-encoding mapping* [Source code]. GitHub. Retrieved August 15, 2026, from https://github.com/openai/tiktoken/blob/main/tiktoken/model.py -OpenAI. (2026). *Vector embeddings*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/guides/embeddings +OpenAI. (2026). *text-embedding-3-large model*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/models/text-embedding-3-large -Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html \ No newline at end of file +Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html From 6d425fd9cc86e34faa227963165f33cf24d584ad Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Sat, 15 Aug 2026 22:04:05 +0900 Subject: [PATCH 4/5] docs(adr): separate provider ceilings from operational embedding limits --- ...nguage-agnostic-semantic-span-budgeting.md | 42 +++++++++++++------ 1 file changed, 29 insertions(+), 13 deletions(-) diff --git a/docs/adr/0017-language-agnostic-semantic-span-budgeting.md b/docs/adr/0017-language-agnostic-semantic-span-budgeting.md index 173adc98..5e952c91 100644 --- a/docs/adr/0017-language-agnostic-semantic-span-budgeting.md +++ b/docs/adr/0017-language-agnostic-semantic-span-budgeting.md @@ -11,7 +11,7 @@ TEPP must turn documentary evidence into meaning-bearing embedding inputs withou The current draft PR #56 uses blank-line splitting as the semantic-unit implementation. That is a useful source-structure hint, but it is not the required architecture. Blank lines do not establish semantic coherence, do not protect the final rendered embedding payload from a model context overflow, do not represent headings, lists, tables, code, captions, dialogue turns, or DOM regions, and do not restore parent or neighboring context at retrieval time. -The target OpenAI profile currently relevant to this decision, `text-embedding-3-large`, has a documented maximum input of 8,192 tokens, uses `cl100k_base` for third-generation embedding token estimation, and returns 3,072 dimensions by default with an optional `dimensions` parameter. These values are provider facts, not universal TEPP constants. They can change by model or revision and therefore belong in a versioned model profile. +The target OpenAI profile currently relevant to this decision, `text-embedding-3-large`, has a provider-documented per-input ceiling of 8,192 tokens, maps to `cl100k_base` in OpenAI's current tokenizer registry, and emits a full vector of up to 3,072 dimensions with an optional `dimensions` parameter. These are mutable provider facts, not universal TEPP constants. TEPP therefore records both the provider-documented ceiling and its own operational hard limit, together with exact source, retrieval date, tokenizer artifact digest, and verification method. ## Decision @@ -49,20 +49,29 @@ Every embedding request is governed by an immutable `embedding_model_profile` co - provider and endpoint family; - model identifier and observed revision; - tokenizer profile and tokenizer artifact digest; -- maximum input tokens; -- default and requested vector dimensions; +- provider-documented maximum input tokens; +- TEPP operational hard-limit tokens and explicit safety margin; +- full/default and requested vector dimensions; - input role; - metadata-template version; -- profile source and verification date. +- profile source URL, retrieval date, source artifact digest, and verification method. -For `text-embedding-3-large`, the initial profile records `max_input_tokens = 8192`, `tokenizer_profile = cl100k_base`, and `default_dimensions = 3072`. TEPP code does not scatter these values as literals. +The following invariant is database- and domain-enforced: + +\[ +\operatorname{operational\_hard\_limit} +\leq +\operatorname{provider\_documented\_maximum}. +\] + +For `text-embedding-3-large`, the initial profile records `provider_documented_max_input_tokens = 8192`, `operational_hard_limit_tokens <= 8192`, `tokenizer_profile = cl100k_base`, and `full_dimensions = 3072`. TEPP code does not scatter these values as literals. Reducing the operational limit or changing a provider/tokenizer fact creates a new profile revision rather than mutating an existing one. ### 4. Budget-aware semantic-span packing Micro-units are packed in document order. A candidate span may be merged only when all of the following hold: 1. no mandatory structural boundary is crossed; -2. the final rendered payload remains within the active model profile; +2. the final rendered payload remains within the active profile's operational hard limit; 3. the configurable target-size policy allows the merge; 4. semantic evidence does not indicate a strong boundary. @@ -73,7 +82,7 @@ Pre-packing may reserve tokens for headings and other metadata, but the authorit \[ \operatorname{tokens}(\operatorname{rendered\_payload}) > -\operatorname{max\_input\_tokens}. +\operatorname{operational\_hard\_limit\_tokens}. \] The hard overflow rate is therefore required to be zero. @@ -119,6 +128,7 @@ This ADR does not: - claim that a blank line, sentence, or paragraph is always a complete semantic unit; - implement true late chunking for a closed embedding API that does not expose contextual token embeddings and pooling control; - let an LLM directly write authoritative spans without deterministic offset and budget validation; +- treat one live provider probe as permanent evidence of an immutable external contract; - establish a stable public release. ## Alternatives considered @@ -134,7 +144,7 @@ This ADR does not: - exact source spans remain the canonical evidence identity; - paragraph-only logic becomes one structural adapter rather than the product contract; -- model limits are configuration data with source, revision, and digest; +- provider limits and TEPP operational limits are distinct versioned profile facts; - the same algorithm works when language metadata is absent, mixed, wrong, or unresolved; - provider calls are impossible until the final payload is counted; - retrieval can use small precise leaves without abandoning larger context; @@ -153,10 +163,13 @@ TEPP returns typed, content-redacting failures: - `unit_unsplittable_under_budget`; - `invalid_source_offset`; - `semantic_similarity_unavailable`; -- `embedding_provider_unavailable`. +- `embedding_provider_unavailable`; +- `embedding_provider_contract_diverged`. `semantic_similarity_unavailable` may degrade to deterministic structure-only packing and records `boundary_mode = structure_only`. Tokenizer/profile/offset failures cannot degrade because they protect correctness. Provider retries and fallback are bounded and recorded; they never change span identity silently. +If the provider rejects a payload that the verified profile permits, the adapter returns `embedding_provider_contract_diverged`, records the response class without source text, and fails closed. Operations may publish a new profile revision with a lower operational hard limit after verification; no existing profile, payload digest, span identity, or vector provenance is rewritten. + ## Security, privacy, and governance impact Document content is untrusted data. It cannot alter the model profile, metadata template, access list, network destination, tool authority, or token budget. LLM boundary proposals require exact source references and deterministic validation. No external resource is fetched during segmentation. @@ -187,11 +200,12 @@ Acceptance requires falsifiable evidence. ### Correctness -- final rendered-payload overflow rate is exactly zero; +- final rendered-payload overflow rate is exactly zero relative to the profile's operational hard limit; - source reconstruction from every leaf span is exact; - byte and Unicode-scalar offsets remain valid under mixed scripts, combining marks, emoji ZWJ sequences, RTL text, and malformed-input rejection; - no control-flow branch depends on a language code; -- unsplittable oversized inputs fail closed without truncation. +- unsplittable oversized inputs fail closed without truncation; +- every operational limit is less than or equal to its provider-documented ceiling and tied to source/retrieval/digest evidence. ### Retrieval quality @@ -211,10 +225,12 @@ Tests include whitespace-delimited and non-whitespace-delimited scripts, Korean/ ### Model profile -The `text-embedding-3-large` profile is tested against 8,192-token acceptance and 8,193-token refusal using the pinned tokenizer profile. The final metadata-rendered payload, not raw content alone, is counted. +Deterministic unit/property tests accept a final rendered payload exactly at the profile's operational hard limit and refuse or recursively split one token above it. The initial `text-embedding-3-large` profile additionally proves that its configured operational hard limit does not exceed the provider-documented 8,192-token ceiling, uses the pinned `cl100k_base` artifact, and binds the full 3,072-dimension output or an explicit requested dimension. + +A scheduled live provider contract probe may corroborate the external boundary using `NVIDIA_NIM_API_KEY` only when the tested provider is reached through the reviewed orchestration/gateway policy; direct OpenAI live verification uses an explicitly authorized provider credential and is not a required unit-test dependency. Live rejection at or below the operational limit is treated as contract divergence, not as permission to truncate. ## Rollback and supersession Rollback disables dense/LLM refinement and uses deterministic structure-only packing with the last validated model profile. It does not restore blank-line-only segmentation as an authoritative semantic algorithm. -Supersession requires an ADR that preserves exact evidence offsets, versioned model limits, final-payload token enforcement, language-independent base correctness, explicit degradation, and retrieval-quality evidence. A later open-weight model may add true late chunking, but that capability must remain separate from the `text-embedding-3-large` API profile unless the provider exposes the necessary token-level contract. \ No newline at end of file +Supersession requires an ADR that preserves exact evidence offsets, versioned provider and operational model limits, final-payload token enforcement, language-independent base correctness, explicit degradation, provider-divergence handling, and retrieval-quality evidence. A later open-weight model may add true late chunking, but that capability must remain separate from the `text-embedding-3-large` API profile unless the provider exposes the necessary token-level contract. From 18200b674cb29d84ab66853f50e1b5dc38096279 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 21 Aug 2026 06:20:18 +0900 Subject: [PATCH 5/5] docs(semantic): close design contract review gaps --- docs/product/prd-v0.5-proposed.md | 25 ++++++++- ...nguage-agnostic-semantic-span-embedding.md | 22 ++++---- ...nguage-agnostic-semantic-span-embedding.md | 53 ++++++++++++++++--- ...agnostic-semantic-span-embedding-design.md | 51 ++++++++++++++++-- 4 files changed, 129 insertions(+), 22 deletions(-) diff --git a/docs/product/prd-v0.5-proposed.md b/docs/product/prd-v0.5-proposed.md index 04be9c13..c9cda0da 100644 --- a/docs/product/prd-v0.5-proposed.md +++ b/docs/product/prd-v0.5-proposed.md @@ -136,6 +136,12 @@ New production logic requires 100% line and branch coverage, complete public/saf - unversioned model-profile requests: 0; - production line/branch coverage: 100%; - public and safety-contract docstrings: 100%. +- every scientific evaluation uses a versioned realistic synthetic-truth fixture, + records the true parameters and uncertainty method, and publishes a + reproducible seed manifest; +- parameter recovery, temporal ordering, graph recovery, invariance, and + CPU/GPU parity gates must meet their fixture-specific pre-registered + tolerances; a missing lane is a failed gate, not an omitted metric. ### Comparative metrics @@ -152,6 +158,23 @@ Against fixed-window and paragraph-only baselines, report: - provider input tokens and cost; - vector and relation storage. +Scientific acceptance additionally reports parameter-recovery bias and RMSE, +nominal interval coverage, temporal-ordering violations, graph precision/ +recall/F1, measurement-invariance decision accuracy, and CPU/GPU parity error. +Each metric is reported overall and by source structure, script/language +profile, and mixed-language status with confidence intervals or a documented +finite-sample uncertainty method. A fixture passes only when all required +metrics meet its versioned tolerance: ordering has zero backward transition +edges, invariance recovers every known invariant/non-invariant fixture, and +parity is within the estimator's declared numerical tolerance against the CPU +`f64` reference. Recovery and graph thresholds are declared in the fixture +manifest before execution rather than chosen after observing results. + +Claims are promoted only when the hard gates, all required strata, uncertainty +report, exact-head checks, and independent review pass. Otherwise the result is +labelled provisional and cannot be used to claim production accuracy, +cross-language equivalence, or retrieval superiority. + Results SHALL include uncertainty and shall be stratified by source structure, script/language profile, and mixed-language status. Stratification evaluates robustness; it does not select the base algorithm. ## 6. User stories @@ -232,4 +255,4 @@ Acceptance of this PRD delta would authorize implementation planning only. It wo - a completed vector database; - a released feature. -Those claims require the acceptance evidence in this document and ADR 0014 promotion. \ No newline at end of file +Those claims require the acceptance evidence in this document and ADR 0014 promotion. diff --git a/docs/research/language-agnostic-semantic-span-embedding.md b/docs/research/language-agnostic-semantic-span-embedding.md index 8ac8de33..e1bafc96 100644 --- a/docs/research/language-agnostic-semantic-span-embedding.md +++ b/docs/research/language-agnostic-semantic-span-embedding.md @@ -13,7 +13,7 @@ Language metadata remains important for measurement invariance, fairness, lexica | OpenAI model documentation and the January 2024 embedding-model announcement | `text-embedding-3-large` remains an embeddings model; its full output is up to 3,072 dimensions and the `dimensions` request parameter may shorten the vector | Provider documentation, not a universal vector length or evidence that every shortened dimension has equal retrieval quality | | Current OpenAI Python SDK embeddings resource | The documented per-input maximum for embedding models is 8,192 tokens; the complete request also has a separate aggregate-token limit | A mutable SDK/API contract. TEPP stores the source URL, retrieval date, and artifact digest and may choose an operational hard limit below the provider maximum | | OpenAI `tiktoken` model mapping and token-counting cookbook | `text-embedding-3-large` maps to `cl100k_base`; `encoding_for_model` is preferred over scattering the encoding name through product code | Tokenization is model/revision profile data. A provider or tokenizer revision requires a new verified profile rather than silently changing an existing one | -| Unicode Standard Annex #29, Revision 47 | Unicode-safe default grapheme/word/sentence boundary guidance | Default boundaries are not semantic truth and are not sufficient for every script or locale | +| Unicode Standard Annex #29, Revision 47 | Unicode-safe default grapheme/word/sentence boundary guidance (Unicode Consortium, 2025) | Default boundaries are not semantic truth and are not sufficient for every script or locale | | Chen et al. (2024), Dense X Retrieval | Retrieval-unit granularity affects retrieval and QA; proposition-scale units are an empirical alternative to passages | Does not prove one granularity is best for every corpus, language, or embedding model | | Günther et al. (2024), Late Chunking | Independent short chunks can lose global context; token-level long-context models can pool contextualized chunk representations | The OpenAI embeddings API does not expose contextual token states/pooling, so TEPP does not claim true late chunking for `text-embedding-3-large` | | TEPP approved PRD v0.4 | Exact source spans, multilingual evidence, optional LLM semantic units, hierarchical/relational evidence, no TF-IDF/BM25 inferential weighting | The approved baseline does not by itself prove the new segmentation implementation | @@ -25,14 +25,14 @@ The model profile distinguishes the provider-documented ceiling from TEPP's oper ## Design implications 1. **No language gate.** Missing, mixed, or wrong language metadata cannot cause a different base algorithm. -2. **No character-count fallback.** Token limits are model/tokenizer facts. -3. **Count the final payload.** Metadata and separators consume tokens. -4. **Use hierarchy, not only overlap.** Leaves retrieve precisely; parent and neighbor relations restore context. -5. **Treat granularity as an empirical parameter.** Compare fixed, paragraph, structural, semantic, and hierarchical alternatives. -6. **Keep late chunking experimental by model capability.** It requires token-level contextual states and pooling control. +2. **No character-count fallback.** Token limits are model/tokenizer facts (OpenAI, 2026a, 2026b, 2026c, 2026d). +3. **Count the final payload.** Metadata and separators consume tokens, so the complete rendered input must be counted (OpenAI, 2026a, 2026b). +4. **Use hierarchy, not only overlap.** Leaves retrieve precisely; parent and neighbor relations restore context as an explicit TEPP design contract. +5. **Treat granularity as an empirical parameter.** Compare fixed, paragraph, structural, semantic, and hierarchical alternatives because retrieval quality changes with unit granularity (Chen et al., 2024). +6. **Keep late chunking experimental by model capability.** It requires token-level contextual states and pooling control (Günther et al., 2024). 7. **Do not use TF-IDF or BM25 in boundary scoring.** Dense similarity is optional; structure-only remains deterministic. 8. **Preserve source evidence.** Summaries and embeddings are derived artifacts. -9. **Version external facts.** Provider limit, tokenizer mapping, dimensions, source URL, retrieval date, and artifact digest belong to an immutable profile. +9. **Version external facts.** Provider limit, tokenizer mapping, dimensions, source URL, retrieval date, and artifact digest belong to an immutable profile (OpenAI, 2024; OpenAI, 2026a, 2026b, 2026c, 2026d; Unicode Consortium, 2025). ## APA 7 references @@ -42,12 +42,12 @@ Günther, M., Mohr, I., Williams, D. J., Wang, B., & Xiao, H. (2024). *Late chun OpenAI. (2024, January 25). *New embedding models and API updates*. https://openai.com/index/new-embedding-models-and-api-updates/ -OpenAI. (2026). *Embeddings resource* [Source code]. GitHub. Retrieved August 15, 2026, from https://github.com/openai/openai-python/blob/main/src/openai/resources/embeddings.py +OpenAI. (2026a). *Embeddings resource* [Source code]. GitHub. Retrieved August 15, 2026, from https://github.com/openai/openai-python/blob/main/src/openai/resources/embeddings.py -OpenAI. (2026). *How to count tokens with tiktoken* [Jupyter Notebook]. OpenAI Cookbook. Retrieved August 15, 2026, from https://github.com/openai/openai-cookbook/blob/main/examples/How_to_count_tokens_with_tiktoken.ipynb +OpenAI. (2026b). *How to count tokens with tiktoken* [Jupyter Notebook]. OpenAI Cookbook. Retrieved August 15, 2026, from https://github.com/openai/openai-cookbook/blob/main/examples/How_to_count_tokens_with_tiktoken.ipynb -OpenAI. (2026). *Model-to-encoding mapping* [Source code]. GitHub. Retrieved August 15, 2026, from https://github.com/openai/tiktoken/blob/main/tiktoken/model.py +OpenAI. (2026c). *Model-to-encoding mapping* [Source code]. GitHub. Retrieved August 15, 2026, from https://github.com/openai/tiktoken/blob/main/tiktoken/model.py -OpenAI. (2026). *text-embedding-3-large model*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/models/text-embedding-3-large +OpenAI. (2026d). *text-embedding-3-large model*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/models/text-embedding-3-large Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html diff --git a/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md b/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md index 2af9f8c7..888f8591 100644 --- a/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md +++ b/docs/superpowers/plans/2026-08-15-language-agnostic-semantic-span-embedding.md @@ -94,7 +94,8 @@ git commit -m "docs(evidence): define language-agnostic semantic spans" - Consumes: `DocumentRecord`, `SourceSpan`, `EvidenceId`. - Produces: - `SourceBlockKind`; - - `SourceBlock::new(document, kind, span, parent, ordinal)`; + - `SourceBlock::new(kind, span, parent, ordinal)`; `document_id` is derived + from `SourceSpan::document_id()` and is never stored twice; - accessors for block identity, kind, span, parent, ordinal. - [ ] **Step 1: Write failing exact-block tests** @@ -115,7 +116,7 @@ Store validated domain fields privately. Do not store a second authoritative tex - [ ] **Step 4: Add stable typed errors and docstrings** -Add content-redacting errors for cross-document parentage and invalid source-block order. +Add content-redacting errors for cross-document parentage and invalid source-block order. The constructor accepts an optional parent block, rejects a parent from another document, and adds a mismatch regression test; no redundant block-level document owner is introduced. - [ ] **Step 5: Run focused and full crate checks** @@ -124,6 +125,8 @@ cargo fmt --all -- --check cargo clippy -p evidence_core --all-targets --offline -- -D warnings cargo test -p evidence_core --offline cargo llvm-cov -p evidence_core --offline --lib --tests --fail-under-lines 100 --fail-under-regions 100 +cargo +nightly-2026-08-01 llvm-cov --branch -p evidence_core --offline --lib --tests --json --summary-only --output-path coverage-branches.json +python3 scripts/check_coverage.py coverage-branches.json --kind branches ``` Expected: all pass at 100%. @@ -155,6 +158,12 @@ git commit -m "feat(evidence): add typed exact-span source blocks" - `TokenOffset`; - typed `SemanticPreprocessorError`. +`EmbeddingModelProfile` carries `profile_source_uri` and +`profile_source_digest` alongside `verified_at` and `tokenizer_digest`. Both +provenance fields are required in the immutable manifest and wire DTO so a +model-limit or tokenizer claim can be reproduced without relying on mutable +provider documentation. + - [ ] **Step 1: Write failing model-profile tests** Assert: @@ -169,7 +178,12 @@ Also reject zero limits, zero dimensions, empty provider/model/revision, noncano - [ ] **Step 2: Write failing fake token-counter tests** -The deterministic fake counter maps Unicode scalar boundaries to token offsets and refuses offsets that split UTF-8. +The deterministic fake counter maps Unicode scalar boundaries to token offsets +and refuses offsets that split UTF-8. `TokenOffset` uses half-open UTF-8 byte +ranges in the complete rendered payload; metadata tokens have no source span, +while content tokens carry their source-span and Unicode-scalar mapping. Tests +cover metadata, source offsets, non-monotonic ranges, and UTF-8/code-point +splits. - [ ] **Step 3: Run focused tests and confirm RED** @@ -270,7 +284,7 @@ Use the fake token counter to assert the complete metadata-rendered payload at 8 - [ ] **Step 2: Write recursive recovery tests** -Cover child block, sentence/line, punctuation, token-offset fallback, bounded overlap, and an unsplittable invalid-offset case. +Cover child block, sentence/line, punctuation, token-offset fallback, bounded overlap, and an unsplittable invalid-offset case. Also cover an oversized unit after a non-empty `current` (the current span is emitted before recursive output) and a first unit larger than `preferred_max` but within `hard_limit` (no empty span is emitted). - [ ] **Step 3: Run focused tests and confirm RED** @@ -284,7 +298,10 @@ Version the template, omit absent metadata, preserve source content, and hash th - [ ] **Step 5: Implement recursive splitting and typed errors** -Never truncate. Token-offset fallback maps only to valid source boundaries. +Never truncate. Token-offset fallback maps only to valid source boundaries. The +packer flushes a non-empty `current` before recursive splitting, initializes a +first preferred-over-limit unit without emitting an empty span, and emits only +non-empty spans so document order is preserved. - [ ] **Step 6: Run quality gates and commit** @@ -312,6 +329,12 @@ git commit -m "feat(semantic): enforce final embedding payload budgets" - `BoundaryReason`; - packed leaf units. +`AdjacentSimilarity::similarities` returns exactly +`units.len().saturating_sub(1)` finite scores, ordered for pairs +`(units[0], units[1])` through `(units[n-2], units[n-1])`. The packer rejects +`NaN`, infinity, and wrong-length responses instead of padding, truncating, or +reordering them. + - [ ] **Step 1: Write failing structure-only tests** Mandatory heading/table/code boundaries split deterministically. Similarity is not required. @@ -322,7 +345,7 @@ A fake similarity port returns known adjacent scores. Assert a sharp semantic dr - [ ] **Step 3: Write failing degradation tests** -When similarity returns `Unavailable`, assert packing succeeds in `StructureOnly` mode and records the reason. Invalid scores (`NaN`, infinity, wrong length) fail closed. +When similarity returns `Unavailable`, assert packing succeeds in `StructureOnly` mode and records the reason. Invalid scores (`NaN`, infinity, wrong length) fail closed. Assert the exact adjacent-pair order and `units.len().saturating_sub(1)` result length. - [ ] **Step 4: Run focused tests and confirm RED** @@ -353,6 +376,11 @@ git commit -m "feat(semantic): pack spans with explicit fallback" - Create: `crates/tepp_api/src/semantic_span.rs` - Create: `schemas/semantic-span-manifest-v1.schema.json` - Create: `examples/semantic-span-manifest-v1.json` +- Create: `examples/semantic-span-manifest-v1-voice.json` +- Create: `examples/semantic-span-manifest-v1-unknown-field.json` +- Create: `examples/semantic-span-manifest-v1-unsupported-version.json` +- Create: `examples/semantic-span-manifest-v1-digest-mismatch.json` +- Modify: `requirements-quality.txt` (pin the `jsonschema` validator and hash) - Modify: `crates/tepp_api/src/lib.rs` - Test: `crates/tepp_api/tests/semantic_span_wire_contract.rs` @@ -385,6 +413,12 @@ Summaries are optional derived artifacts. They cannot replace source spans or be - [ ] **Step 5: Validate schema and run quality gates** ```bash +jsonschema --version # pinned in the quality-tool requirements +jsonschema -i examples/semantic-span-manifest-v1.json schemas/semantic-span-manifest-v1.schema.json +jsonschema -i examples/semantic-span-manifest-v1-voice.json schemas/semantic-span-manifest-v1.schema.json +! jsonschema -i examples/semantic-span-manifest-v1-unknown-field.json schemas/semantic-span-manifest-v1.schema.json +! jsonschema -i examples/semantic-span-manifest-v1-unsupported-version.json schemas/semantic-span-manifest-v1.schema.json +! jsonschema -i examples/semantic-span-manifest-v1-digest-mismatch.json schemas/semantic-span-manifest-v1.schema.json cargo fmt --all -- --check cargo clippy --workspace --all-targets --all-features -- -D warnings cargo nextest run --workspace --all-features @@ -392,6 +426,11 @@ cargo test --doc --workspace --all-features python3 scripts/validate_documentation.py ``` +The validator version is pinned with its checksum in the repository quality +requirements, and the positive/negative fixtures run in the exact-head CI +lane. Rust DTO tests remain necessary but do not replace validation of the +standalone schema artifact. + - [ ] **Step 6: Commit** ```bash @@ -553,4 +592,4 @@ A merged implementation may claim zero-overflow and exact-span behavior only aft ```bash git add docs CHANGELOG.md git commit -m "docs(semantic): record semantic span acceptance evidence" -``` \ No newline at end of file +``` diff --git a/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md b/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md index fa33cf16..fecd1b0a 100644 --- a/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md +++ b/docs/superpowers/specs/2026-08-15-language-agnostic-semantic-span-embedding-design.md @@ -70,12 +70,24 @@ pub enum SourceBlockKind { pub struct SourceBlock { block_id: EvidenceId, - document_id: EvidenceId, block_kind: SourceBlockKind, source_span: SourceSpan, parent_block_id: Option, ordinal_index: u32, } + +impl SourceBlock { + pub fn new( + block_kind: SourceBlockKind, + source_span: SourceSpan, + parent: Option<&SourceBlock>, + ordinal_index: u32, + ) -> Result; + + pub fn document_id(&self) -> EvidenceId { + self.source_span.document_id() + } +} ``` Invariants: @@ -107,6 +119,8 @@ pub struct EmbeddingModelProfile { input_role: EmbeddingInputRole, metadata_template_version: String, verified_at: Timestamp, + profile_source_uri: String, + profile_source_digest: ContentDigest, } pub trait TokenCounter { @@ -125,6 +139,23 @@ pub trait TokenCounter { ) -> Result, Self::Error>; } +pub struct TokenOffset { + token_index: u32, + payload_byte_start: u32, + payload_byte_end: u32, + source_scalar_start: Option, + source_scalar_end: Option, + source_span: Option, +} + +`TokenOffset` coordinates are UTF-8 byte offsets into the complete rendered +payload. `payload_byte_start..payload_byte_end` is half-open and must end on +Unicode-scalar boundaries. Tokens emitted for title, heading, separators, and +other metadata have no `source_span`; content-slot tokens carry the exact +source span and Unicode-scalar subrange they represent. The adapter must reject +non-monotonic, out-of-payload, UTF-8-splitting, or source-inconsistent offsets; +it may not silently reinterpret payload coordinates as source coordinates. + pub trait AdjacentSimilarity { type Error; @@ -134,6 +165,12 @@ pub trait AdjacentSimilarity { ) -> Result, Self::Error>; } +The `similarities` result contains exactly +`units.len().saturating_sub(1)` finite scores in document order. Element `i` +corresponds to the adjacent pair `(units[i], units[i + 1])`. A consumer rejects +`NaN`, infinity, or any other result length before using a score; it never +silently pads, truncates, or reorders the provider result. + pub struct SemanticSpanPolicy { target_input_tokens: u32, preferred_max_input_tokens: u32, @@ -235,9 +272,16 @@ micro_units = build_micro_units(source_blocks) for unit in micro_units: if final_payload_tokens(unit) > hard_limit: + if current is not empty: + emit(current) + current = empty emit(recursive_split(unit)) continue + if current is empty: + current = unit + continue + candidate = current + unit if mandatory_structure_break(current, unit): emit(current) @@ -251,7 +295,8 @@ for unit in micro_units: else: current = candidate -emit(current) +if current is not empty: + emit(current) for emitted_span: assert final_payload_tokens(emitted_span) <= hard_limit @@ -393,4 +438,4 @@ Günther, M., Mohr, I., Williams, D. J., Wang, B., & Xiao, H. (2024). *Late chun OpenAI. (2026). *Vector embeddings*. OpenAI API documentation. Retrieved August 15, 2026, from https://developers.openai.com/api/docs/guides/embeddings -Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html \ No newline at end of file +Unicode Consortium. (2025). *Unicode Standard Annex #29: Unicode text segmentation* (Revision 47, Unicode 17.0.0). https://www.unicode.org/reports/tr29/tr29-47.html