diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 65f588c0..5060d3cd 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -61,6 +61,7 @@ boundaries above remain the target modular MSA architecture. | `tepp_simulation` | known-truth temporal/event data generation | | `validation_core` | RMSE, bias, coverage, graph, and Monte Carlo metrics | | `tepp_api` | versioned DTO, schema, and export contracts | +| `prompt_source` | prompt boilerplate is not unique latent content and not stopword deletion | | `corpus_background` | corpus-background wording is not unique latent content and not stopword deletion | | `modality_source` | non-lexical modality is not unique latent content and not stopword deletion | | `copied_text` | copied-text residue is not unique latent content and not stopword deletion | diff --git a/CHANGELOG.md b/CHANGELOG.md index ad9d0dde..d663bf79 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ All notable changes to TEPP are documented here. The format follows Keep a Chang ### Added +- `prompt_source` identity gate: instruction and prompt boilerplate is not unique latent content and is not erased by a stopword list; `identity_recovery_rate` reports exact kind matches, with a contract test comparing correct recovery with an all-unique collapse on a mixed known-truth fixture (ADR 0004/0012). - `corpus_background` identity gate: corpus-level background wording is not unique latent content and is not erased by a stopword list; recovered background kinds match known truth at a higher computed rate than collapsing every token to unique content (ADR 0004/0012). - `modality_source` identity gate: non-lexical modality is not unique latent content and is not erased by a stopword list; recovered modality kinds match known truth at a higher computed rate than collapsing every token to unique content (ADR 0004/0012). - `copied_text` identity gate: copied and boilerplate residue is not unique latent content and is not erased by a stopword list; recovered copied-text kinds match known truth at a higher computed rate than collapsing every token to unique content (ADR 0004/0012). diff --git a/Cargo.lock b/Cargo.lock index fcd7b90e..656738e2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1101,6 +1101,10 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "prompt_source" +version = "0.1.0" + [[package]] name = "provider_receipt" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index c9b5b199..02c29288 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -11,6 +11,7 @@ members = [ "crates/tepp_simulation", "crates/validation_core", "crates/tepp_api", + "crates/prompt_source", "crates/corpus_background", "crates/modality_source", "crates/copied_text", @@ -59,6 +60,7 @@ default-members = [ "crates/tepp_simulation", "crates/validation_core", "crates/tepp_api", + "crates/prompt_source", "crates/corpus_background", "crates/modality_source", "crates/copied_text", diff --git a/README.md b/README.md index 5c59f762..288df5c1 100644 --- a/README.md +++ b/README.md @@ -43,6 +43,7 @@ crates/corpus_split crates/tepp_simulation crates/validation_core crates/tepp_api +crates/prompt_source crates/corpus_background crates/modality_source crates/copied_text diff --git a/crates/prompt_source/Cargo.toml b/crates/prompt_source/Cargo.toml new file mode 100644 index 00000000..9a48d6d0 --- /dev/null +++ b/crates/prompt_source/Cargo.toml @@ -0,0 +1,17 @@ +[package] +name = "prompt_source" +description = "Prompt boilerplate is not unique content and not stopword deletion." +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +authors.workspace = true +repository.workspace = true +homepage.workspace = true +readme.workspace = true +keywords.workspace = true +categories.workspace = true +publish = false + +[lints] +workspace = true diff --git a/crates/prompt_source/src/error.rs b/crates/prompt_source/src/error.rs new file mode 100644 index 00000000..678cbbd1 --- /dev/null +++ b/crates/prompt_source/src/error.rs @@ -0,0 +1,53 @@ +//! Fail-closed prompt-source errors. + +use std::fmt; + +/// A fail-closed prompt-source error. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[non_exhaustive] +pub enum PromptSourceError { + /// Prompt boilerplate was treated as unique latent content. + PromptIsNotUniqueContent, + /// Prompt boilerplate was treated as stopword deletion. + PromptIsNotStopwordDeletion, + /// A recovery slice was empty or length-mismatched. + InvalidPromptPayload, +} + +impl fmt::Display for PromptSourceError { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + let message = match self { + Self::PromptIsNotUniqueContent => "prompt boilerplate is not unique latent content", + Self::PromptIsNotStopwordDeletion => "prompt boilerplate is not stopword deletion", + Self::InvalidPromptPayload => "invalid prompt-source payload", + }; + formatter.write_str(message) + } +} + +impl std::error::Error for PromptSourceError {} + +#[cfg(test)] +mod tests { + use super::PromptSourceError; + + #[test] + fn error_messages_are_stable() { + for (error, message) in [ + ( + PromptSourceError::PromptIsNotUniqueContent, + "prompt boilerplate is not unique latent content", + ), + ( + PromptSourceError::PromptIsNotStopwordDeletion, + "prompt boilerplate is not stopword deletion", + ), + ( + PromptSourceError::InvalidPromptPayload, + "invalid prompt-source payload", + ), + ] { + assert_eq!(error.to_string(), message); + } + } +} diff --git a/crates/prompt_source/src/kind.rs b/crates/prompt_source/src/kind.rs new file mode 100644 index 00000000..ad3d09d7 --- /dev/null +++ b/crates/prompt_source/src/kind.rs @@ -0,0 +1,132 @@ +//! Prompt boilerplate versus unique latent content. + +use crate::PromptSourceError; + +/// Closed vocabulary of prompt-related token treatments. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub enum PromptKind { + /// Instruction or prompt boilerplate, not unique document meaning. + PromptBoilerplate, + /// Token treatment reserved for unique latent content. + UniqueContent, +} + +impl PromptKind { + /// Return the stable wire kind name. + #[must_use] + pub const fn wire_name(self) -> &'static str { + match self { + Self::PromptBoilerplate => "prompt_boilerplate", + Self::UniqueContent => "unique_content", + } + } + + /// Parse a stable wire kind name. + /// + /// # Errors + /// + /// Returns [`PromptSourceError::InvalidPromptPayload`] for unrecognized + /// names. + pub fn from_wire_name(name: &str) -> Result { + match name { + "prompt_boilerplate" => Ok(Self::PromptBoilerplate), + "unique_content" => Ok(Self::UniqueContent), + _ => Err(PromptSourceError::InvalidPromptPayload), + } + } +} + +/// Refuse to treat prompt boilerplate as unique latent content. +/// +/// # Errors +/// +/// Returns [`PromptSourceError::PromptIsNotUniqueContent`] when `kind` is +/// [`PromptKind::PromptBoilerplate`]. +pub fn refuse_prompt_as_unique_content(kind: PromptKind) -> Result<(), PromptSourceError> { + match kind { + PromptKind::PromptBoilerplate => Err(PromptSourceError::PromptIsNotUniqueContent), + PromptKind::UniqueContent => Ok(()), + } +} + +/// Refuse to treat prompt boilerplate as stopword deletion. +/// +/// # Errors +/// +/// Returns [`PromptSourceError::PromptIsNotStopwordDeletion`] when `kind` is +/// [`PromptKind::PromptBoilerplate`]. +pub fn refuse_prompt_as_stopword_deletion(kind: PromptKind) -> Result<(), PromptSourceError> { + match kind { + PromptKind::PromptBoilerplate => Err(PromptSourceError::PromptIsNotStopwordDeletion), + PromptKind::UniqueContent => Ok(()), + } +} + +/// Fraction of recovered prompt kinds that match known truth. +/// +/// # Errors +/// +/// Returns [`PromptSourceError::InvalidPromptPayload`] when either slice is +/// empty or the lengths differ. +pub fn identity_recovery_rate( + truth: &[PromptKind], + decided: &[PromptKind], +) -> Result { + if truth.is_empty() || truth.len() != decided.len() { + return Err(PromptSourceError::InvalidPromptPayload); + } + let mut matches = 0_u32; + for (truth_kind, decided_kind) in truth.iter().zip(decided) { + if truth_kind == decided_kind { + matches += 1; + } + } + Ok(f64::from(matches) / truth.len() as f64) +} + +#[cfg(test)] +mod tests { + use super::{ + PromptKind, identity_recovery_rate, refuse_prompt_as_stopword_deletion, + refuse_prompt_as_unique_content, + }; + use crate::PromptSourceError; + + #[test] + fn local_branches_cover_kinds_payloads_and_wire_names() { + assert_eq!( + refuse_prompt_as_unique_content(PromptKind::PromptBoilerplate), + Err(PromptSourceError::PromptIsNotUniqueContent) + ); + assert_eq!( + refuse_prompt_as_stopword_deletion(PromptKind::PromptBoilerplate), + Err(PromptSourceError::PromptIsNotStopwordDeletion) + ); + refuse_prompt_as_unique_content(PromptKind::UniqueContent).expect("unique"); + refuse_prompt_as_stopword_deletion(PromptKind::UniqueContent).expect("unique"); + for kind in [PromptKind::PromptBoilerplate, PromptKind::UniqueContent] { + assert_eq!( + PromptKind::from_wire_name(kind.wire_name()).expect("round-trip"), + kind + ); + } + assert_eq!( + PromptKind::from_wire_name("template"), + Err(PromptSourceError::InvalidPromptPayload) + ); + let matched = identity_recovery_rate( + &[PromptKind::PromptBoilerplate], + &[PromptKind::PromptBoilerplate], + ) + .expect("rate"); + assert!((matched - 1.0).abs() < f64::EPSILON); + assert_eq!( + identity_recovery_rate(&[], &[]), + Err(PromptSourceError::InvalidPromptPayload) + ); + assert_eq!( + identity_recovery_rate(&[PromptKind::PromptBoilerplate], &[]), + Err(PromptSourceError::InvalidPromptPayload) + ); + } +} diff --git a/crates/prompt_source/src/lib.rs b/crates/prompt_source/src/lib.rs new file mode 100644 index 00000000..8f8893f4 --- /dev/null +++ b/crates/prompt_source/src/lib.rs @@ -0,0 +1,22 @@ +#![forbid(unsafe_code)] +#![deny(missing_docs)] +#![allow(clippy::cast_precision_loss)] +//! Prompt boilerplate is not unique latent content. +//! +//! Instruction and prompt text stays explicit method structure. It is not +//! unique document meaning and is not erased by a stopword list +//! (ADR 0004/0012). + +mod error; +mod kind; + +/// Fail-closed prompt-source errors. +pub use error::PromptSourceError; +/// Closed vocabulary of prompt-related token treatments. +pub use kind::PromptKind; +/// Fraction of recovered prompt kinds that match known truth. +pub use kind::identity_recovery_rate; +/// Refuse to treat prompt boilerplate as stopword deletion. +pub use kind::refuse_prompt_as_stopword_deletion; +/// Refuse to treat prompt boilerplate as unique latent content. +pub use kind::refuse_prompt_as_unique_content; diff --git a/crates/prompt_source/tests/crate_contract.rs b/crates/prompt_source/tests/crate_contract.rs new file mode 100644 index 00000000..7f82bf4d --- /dev/null +++ b/crates/prompt_source/tests/crate_contract.rs @@ -0,0 +1,7 @@ +//! Integration contract for the `prompt_source` package identity. + +#[test] +fn package_identity_is_stable() { + let observed = std::hint::black_box(env!("CARGO_PKG_NAME")); + assert_eq!(observed, "prompt_source"); +} diff --git a/crates/prompt_source/tests/prompt_source_contract.rs b/crates/prompt_source/tests/prompt_source_contract.rs new file mode 100644 index 00000000..966976ec --- /dev/null +++ b/crates/prompt_source/tests/prompt_source_contract.rs @@ -0,0 +1,67 @@ +//! Prompt boilerplate is not unique content and not stopword deletion. + +use prompt_source::{ + PromptKind, PromptSourceError, identity_recovery_rate, refuse_prompt_as_stopword_deletion, + refuse_prompt_as_unique_content, +}; + +#[test] +fn prompt_boilerplate_cannot_become_unique_content_or_stopword_deletion() { + assert_eq!( + refuse_prompt_as_unique_content(PromptKind::PromptBoilerplate), + Err(PromptSourceError::PromptIsNotUniqueContent) + ); + assert_eq!( + refuse_prompt_as_stopword_deletion(PromptKind::PromptBoilerplate), + Err(PromptSourceError::PromptIsNotStopwordDeletion) + ); + refuse_prompt_as_unique_content(PromptKind::UniqueContent).expect("unique"); + refuse_prompt_as_stopword_deletion(PromptKind::UniqueContent).expect("unique"); +} + +#[test] +fn recovered_kinds_match_known_truth_better_than_a_unique_content_collapse() { + let truth = [ + PromptKind::PromptBoilerplate, + PromptKind::UniqueContent, + PromptKind::PromptBoilerplate, + ]; + let recovered = truth; + let collapsed = [ + PromptKind::UniqueContent, + PromptKind::UniqueContent, + PromptKind::UniqueContent, + ]; + let recovered_rate = identity_recovery_rate(&truth, &recovered).expect("recovered"); + let collapsed_rate = identity_recovery_rate(&truth, &collapsed).expect("collapsed"); + let expected = { + let mut matches = 0_u32; + for (truth_kind, decided_kind) in truth.iter().zip(recovered.iter()) { + if truth_kind == decided_kind { + matches += 1; + } + } + f64::from(matches) / f64::from(u32::try_from(truth.len()).expect("len")) + }; + assert!((recovered_rate - expected).abs() < f64::EPSILON); + assert!(recovered_rate > collapsed_rate); +} + +#[test] +fn empty_or_mismatched_kind_payloads_fail_closed() { + assert_eq!( + identity_recovery_rate(&[], &[]), + Err(PromptSourceError::InvalidPromptPayload) + ); + assert_eq!( + identity_recovery_rate(&[PromptKind::PromptBoilerplate], &[]), + Err(PromptSourceError::InvalidPromptPayload) + ); + assert_eq!( + identity_recovery_rate( + &[PromptKind::PromptBoilerplate, PromptKind::UniqueContent], + &[PromptKind::PromptBoilerplate] + ), + Err(PromptSourceError::InvalidPromptPayload) + ); +} diff --git a/docs/TRACEABILITY.md b/docs/TRACEABILITY.md index 1f3fac43..b05bf311 100644 --- a/docs/TRACEABILITY.md +++ b/docs/TRACEABILITY.md @@ -25,7 +25,7 @@ The full APA 7th standards/literature register remains `docs/research/standards- | TRSL-TM temporal/relational topic posterior and backend compatibility | ADR 0012; ADR 0004 | future `topic_measurement` | accepted-target | | global P0 topic identity with activity/dormancy/reactivation | ADR 0012 | future topic lineage/activity state | accepted-target | | no default stopword deletion / no TF-IDF-BM25 inferential weighting | ADR 0004/0012; PRD/TRD | future semantic/method-source model | accepted-target | -| report template/section/copied/style/modality method effects | ADR 0004/0012; PRD/TRD | simulation truth factors implemented; `corpus_background` background-versus-unique-content identity on the active PR; estimator-side method model remains future | partial | +| report template/section/copied/style/modality method effects | ADR 0004/0012; PRD/TRD | simulation truth factors implemented; `prompt_source` prompt-versus-unique-content identity on the active PR; estimator-side method model remains future | partial | | candidate K statistical/Pareto gates + blinded LLM review | ADR 0012; research | future `model_selection` | accepted-target | | compositional topic correlation / stable clustering | ADR 0005/0012; research | future `network_analysis` | accepted-target | | posterior ESEM / longitudinal invariance / DSEM | ADR 0005 | `psychometric_fit` ESEM loading and DSEM lag gates on the active PR; `psychometric_core` input gates remain #49; invariance/multilevel remain accepted-target | active-PR | diff --git a/docs/adr/0004-shared-multilingual-latent-space.md b/docs/adr/0004-shared-multilingual-latent-space.md index 00934224..965869e8 100644 --- a/docs/adr/0004-shared-multilingual-latent-space.md +++ b/docs/adr/0004-shared-multilingual-latent-space.md @@ -1,6 +1,7 @@ # ADR 0004 — Shared multilingual latent semantic space **Decision status:** Accepted +**Implementation maturity:** accepted-target — prompt-versus-unique-content identity in `prompt_source` on the active PR; shared-space estimators remain accepted-target **Implementation maturity:** accepted-target — corpus-background-versus-unique-content identity in `corpus_background` on the active PR; shared-space estimators remain accepted-target **Implementation maturity:** accepted-target — modality-versus-unique-content identity in `modality_source` on the active PR; shared-space estimators remain accepted-target **Implementation maturity:** accepted-target — copied-versus-unique-content identity in `copied_text` on the active PR; shared-space estimators remain accepted-target diff --git a/docs/adr/0012-temporal-relational-shared-latent-topic-measurement.md b/docs/adr/0012-temporal-relational-shared-latent-topic-measurement.md index d47d1f02..c43bfbde 100644 --- a/docs/adr/0012-temporal-relational-shared-latent-topic-measurement.md +++ b/docs/adr/0012-temporal-relational-shared-latent-topic-measurement.md @@ -1,6 +1,7 @@ # ADR 0012 — Temporal Relational Shared-Latent Topic Measurement **Decision status:** Accepted +**Implementation maturity:** accepted-target — prompt-versus-unique-content identity in `prompt_source` on the active PR; estimator-side method model remains accepted-target **Implementation maturity:** accepted-target — corpus-background-versus-unique-content identity in `corpus_background` on the active PR; estimator-side method model remains accepted-target **Implementation maturity:** accepted-target — modality-versus-unique-content identity in `modality_source` on the active PR; estimator-side method model remains accepted-target **Implementation maturity:** accepted-target — copied-versus-unique-content identity in `copied_text` on the active PR; estimator-side method model remains accepted-target diff --git a/docs/adr/README.md b/docs/adr/README.md index 75ccdd13..db4d279d 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -25,6 +25,7 @@ Read [`ADR_POLICY.md`](ADR_POLICY.md) first. **Decision status and implementatio | [0002](0002-six-clock-temporal-semantics.md) | Six-clock temporal semantics and fail-closed historical leakage prevention | Accepted | active-PR | Typed clocks/intervals are implemented-main; `document_clocks` refuses omitted assertion/document time on the active PR. Later graph/split enforcement remains target work. | | [0002](0002-six-clock-temporal-semantics.md) | Six-clock temporal semantics and fail-closed historical leakage prevention | Accepted | active-PR | Provenance-vs-transition gate in `citation_edge` on the active PR; remaining graph/split enforcement stays accepted-target. | | [0003](0003-relational-event-multiple-membership.md) | Relational event ontology and time-varying cross-classified multiple membership | Accepted | partial | Weighted time-varying membership network/roles are active-PR (PR #12); full multilevel estimators, graph ontology, and persistence remain accepted-target. ADR 0016 owns event-intelligence tasks. | +| [0004](0004-shared-multilingual-latent-space.md) | One shared multilingual latent space with explicit invariance status | Accepted | accepted-target | Prompt-versus-unique-content identity is `prompt_source` on the active PR; ADR 0012 owns the full topic-estimator contract. | | [0004](0004-shared-multilingual-latent-space.md) | One shared multilingual latent space with explicit invariance status | Accepted | accepted-target | Corpus-background-versus-unique-content identity is `corpus_background` on the active PR; ADR 0012 owns the full topic-estimator contract. | | [0004](0004-shared-multilingual-latent-space.md) | One shared multilingual latent space with explicit invariance status | Accepted | accepted-target | Modality-versus-unique-content identity is `modality_source` on the active PR; ADR 0012 owns the full topic-estimator contract. | | [0004](0004-shared-multilingual-latent-space.md) | One shared multilingual latent space with explicit invariance status | Accepted | accepted-target | Copied-versus-unique-content identity is `copied_text` on the active PR; ADR 0012 owns the full topic-estimator contract. | @@ -57,6 +58,7 @@ Read [`ADR_POLICY.md`](ADR_POLICY.md) first. **Decision status and implementatio | [0010](0010-adaptive-llm-orchestration.md) | Adaptive LLM orchestration and test-time compute | Accepted | partial | `tepp_api` router/ablation/orchestrator binding implemented-main; evidence-bounded `interpretation_gateway` is on the active PR; live NIM execution and production ablation evidence remain accepted-target. | | [0010](0010-adaptive-llm-orchestration.md) | Adaptive LLM orchestration and test-time compute | Accepted | partial | `tepp_api` router/ablation/orchestrator binding implemented-main; live NIM execution and production ablation evidence remain accepted-target. | | [0011](0011-standalone-modular-msa-boundary.md) | Standalone operation and modular CWL MSA boundary | Accepted | partial | Owns cross-service persistence/credential/API authority; no direct cross-service application-table coupling. | +| [0012](0012-temporal-relational-shared-latent-topic-measurement.md) | Temporal Relational Shared-Latent Topic Measurement (TRSL-TM) | Accepted | accepted-target | Prompt-versus-unique-content identity is `prompt_source` on the active PR; estimator-side method model, backend, and K gates remain accepted-target. | | [0012](0012-temporal-relational-shared-latent-topic-measurement.md) | Temporal Relational Shared-Latent Topic Measurement (TRSL-TM) | Accepted | accepted-target | Corpus-background-versus-unique-content identity is `corpus_background` on the active PR; estimator-side method model, backend, and K gates remain accepted-target. | | [0012](0012-temporal-relational-shared-latent-topic-measurement.md) | Temporal Relational Shared-Latent Topic Measurement (TRSL-TM) | Accepted | accepted-target | Modality-versus-unique-content identity is `modality_source` on the active PR; estimator-side method model, backend, and K gates remain accepted-target. | | [0012](0012-temporal-relational-shared-latent-topic-measurement.md) | Temporal Relational Shared-Latent Topic Measurement (TRSL-TM) | Accepted | accepted-target | Copied-versus-unique-content identity is `copied_text` on the active PR; estimator-side method model, backend, and K gates remain accepted-target. | diff --git a/docs/research/prompt-source-identity.md b/docs/research/prompt-source-identity.md new file mode 100644 index 00000000..884ca696 --- /dev/null +++ b/docs/research/prompt-source-identity.md @@ -0,0 +1,43 @@ +# Prompt boilerplate is not unique content (doctoring) + +## Scope + +`prompt_source` keeps instruction and prompt boilerplate out of unique +latent content and out of global stopword deletion. Recovery is the +computed share of recovered kinds that match known truth. + +This slice does not persist method sources, allocate migration `0008`, +or replace `method_effects`, `section_source`, `style_source`, +`copied_text`, `modality_source`, `corpus_background`, or +`stopword_deletion`. + +## Authority + +### Normative TEPP contract + +- `docs/adr/0004-shared-multilingual-latent-space.md` and + `docs/adr/0012-temporal-relational-shared-latent-topic-measurement.md` + — method and template sources are modeled explicitly and are not + inferential topic weights or stopword deletions. + +### Supporting literature + +Brown et al. (2020) provide primary evidence that textual prompts and +demonstrations condition language-model task behavior, while Reynolds and +McDonell (2021) study prompt programming as a method for directing model +behavior. Neither study defines TEPP's latent-content labels. The statement +that prompt boilerplate is not unique latent content is therefore a normative +TEPP measurement contract derived from ADR 0004 and ADR 0012, not a universal +empirical claim about every prompt or corpus. + +Brown, T. B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., +Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, +A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D., Wu, J., +Winter, C., … Amodei, D. (2020). Language models are few-shot learners. +*Advances in Neural Information Processing Systems, 33*, 1877–1901. +https://papers.neurips.cc/paper/2020/hash/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html + +Reynolds, L., & McDonell, K. (2021). Prompt programming for large language +models: Beyond the few-shot paradigm. In *Extended abstracts of the 2021 CHI +conference on human factors in computing systems*. Association for Computing +Machinery. https://doi.org/10.1145/3411763.3451760 diff --git a/docs/research/standards-and-literature.md b/docs/research/standards-and-literature.md index ae0dfcec..e6783f0a 100644 --- a/docs/research/standards-and-literature.md +++ b/docs/research/standards-and-literature.md @@ -38,7 +38,13 @@ Bianchi, F., Terragni, S., Hovy, D., Nozza, D., & Fersini, E. (2021). Cross-ling Nguyen, T. P., Minh, N. V., Nguyen, T., Van, L. N., Nguyen, D. A., Sang, D. V., & Le, T. (2025). XTRA: Cross-lingual topic modeling with topic and representation alignments. In *Findings of the Association for Computational Linguistics: EMNLP 2025*. Association for Computational Linguistics. -TEPP retains a logistic-normal CPU reference while allowing adapter backends that satisfy shared-latent, posterior, temporal, relational, and measurement-invariance contracts. Corpus-background wording is modeled as explicit structure, not unique latent content and not a stopword deletion (Chemudugunta et al., 2007). +Brown, T. B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D., Wu, J., Winter, C., … Amodei, D. (2020). Language models are few-shot learners. *Advances in Neural Information Processing Systems, 33*, 1877–1901. https://papers.neurips.cc/paper/2020/hash/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html + +Reynolds, L., & McDonell, K. (2021). Prompt programming for large language models: Beyond the few-shot paradigm. In *Extended abstracts of the 2021 CHI conference on human factors in computing systems*. Association for Computing Machinery. https://doi.org/10.1145/3411763.3451760 + +Liu, P., Yuan, W., Fu, J., Jiang, Z., Hayashi, H., & Neubig, G. (2023). Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. *ACM Computing Surveys, 55*(9), Article 195. https://doi.org/10.1145/3560815 + +TEPP retains a logistic-normal CPU reference while allowing adapter backends that satisfy shared-latent, posterior, temporal, relational, and measurement-invariance contracts. Brown et al. (2020) and Reynolds and McDonell (2021) provide primary research context for prompts as task-conditioning and prompt-programming mechanisms; they do not define TEPP's latent-content labels. As a normative ADR 0004/0012 contract, instruction and prompt boilerplate is therefore modeled as explicit method structure, not unique latent content and not a stopword deletion. Liu et al. (2023) is secondary survey background only and is not evidence for that repository-specific classification. ## Topic-model evaluation and LLM judges diff --git a/docs/validation/temporal-event-foundation.md b/docs/validation/temporal-event-foundation.md index b397358e..775cbd36 100644 --- a/docs/validation/temporal-event-foundation.md +++ b/docs/validation/temporal-event-foundation.md @@ -36,6 +36,7 @@ This report tracks exact-head scientific and engineering evidence required befor | Mention-confidence Brier score | `event_core` | active-PR | calibration vs binary truth | perfect 0 / half 0.25 RMSE | ADR 0003; `docs/research/mention-confidence-brier.md` | | Checkpoint is not the estimator | `checkpoint_authority` | accepted-target | active PR | refuse checkpoint-as-estimator + unvalidated artifact + recovery vs estimator collapse | ADR 0001/0014 | | Versioned API/export contracts | `tepp_api` | implemented-main | naruon HTTP interchange | unknown-field/version/limit + naruon HTTPS interchange tests | Task 12 / PR #21; live HTTP service remaining | +| Prompt-versus-unique-content identity | `prompt_source` | accepted-target | active PR | refuse prompt-as-unique/stopword + recovery vs unique-content collapse | ADR 0004/0012 | | Corpus-background-versus-unique-content identity | `corpus_background` | accepted-target | active PR | refuse background-as-unique/stopword + recovery vs unique-content collapse | ADR 0004/0012 | | Modality-versus-unique-content identity | `modality_source` | accepted-target | active PR | refuse modality-as-unique/stopword + recovery vs unique-content collapse | ADR 0004/0012 | | Copied-versus-unique-content identity | `copied_text` | accepted-target | active PR | refuse copied-text-as-unique/stopword + recovery vs unique-content collapse | ADR 0004/0012 | diff --git a/scripts/check_workspace_contract.py b/scripts/check_workspace_contract.py index e3139024..1f7a7043 100644 --- a/scripts/check_workspace_contract.py +++ b/scripts/check_workspace_contract.py @@ -23,6 +23,7 @@ "tepp_simulation", "validation_core", "tepp_api", + "prompt_source", "corpus_background", "modality_source", "copied_text",