diff --git a/CHANGELOG.md b/CHANGELOG.md index 30fb1c89..34b9e5d4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,7 @@ All notable changes to TEPP are documented here. The format follows Keep a Chang ### Added +- `event_core` mention-confidence Brier score: known-truth binary outcomes recover a computed Brier of 0 for perfect forecasts and 0.25 for constant 0.5, with empty or mismatched streams failing closed. - `membership_core` nested ICC: CPU `f64` unbalanced ANOVA recovers a known cluster ICC and refuses to treat cross-classified or multiple-membership designs as a single hierarchy (ADR 0003). - `persistence_postgres` typed `text_segment` SQL: insert/lookup of exact UTF-8 half-open byte spans on the existing `0006` table, cutoff-eligible document reads (`available_time <= knowledge_cutoff`), and live recovery of a known `hello` span. No new migration number (`#45` still owns `0007`). - Hourly contextual-orchestrator discovery records all provider models but routes OpenCode only through general-chat candidates, excluding embedding, image, reranker, transcription, moderation, safety, and other endpoint-only identifiers before price selection. diff --git a/DOCUMENTATION.md b/DOCUMENTATION.md index d0f517de..7c004f85 100644 --- a/DOCUMENTATION.md +++ b/DOCUMENTATION.md @@ -34,6 +34,7 @@ TEPP's approved PRD v0.4 and implementation plan are the primary product baselin | Hourly NIM product-development operations | [`docs/operations/HOURLY_NIM_PRODUCT_DEVELOPMENT.md`](docs/operations/HOURLY_NIM_PRODUCT_DEVELOPMENT.md) | | Actions workflow fleet audit | [`docs/operations/ACTIONS_WORKFLOW_FLEET.md`](docs/operations/ACTIONS_WORKFLOW_FLEET.md) | | Actions fleet research doctoring | [`docs/research/actions-workflow-fleet.md`](docs/research/actions-workflow-fleet.md) | +| Mention-confidence Brier doctoring | [`docs/research/mention-confidence-brier.md`](docs/research/mention-confidence-brier.md) | | Event-intelligence status-gate doctoring | [`docs/research/event-intelligence-status-gates.md`](docs/research/event-intelligence-status-gates.md) | | Retention/deletion/legal-hold doctoring | [`docs/research/retention-deletion-legal-hold.md`](docs/research/retention-deletion-legal-hold.md) | | Provider-payload minimization doctoring | [`docs/research/provider-payload-minimization.md`](docs/research/provider-payload-minimization.md) | diff --git a/crates/event_core/src/confidence.rs b/crates/event_core/src/confidence.rs index d16597eb..f2bf42bb 100644 --- a/crates/event_core/src/confidence.rs +++ b/crates/event_core/src/confidence.rs @@ -39,6 +39,29 @@ impl EventConfidence { } } +/// Mean squared error of mention probabilities against binary truth. +/// +/// # Errors +/// +/// Returns [`EventError::InvalidWirePayload`] when the slices are empty or +/// have unequal length. +pub fn mention_brier_score( + forecasts: &[EventConfidence], + outcomes: &[bool], +) -> Result { + if forecasts.is_empty() || forecasts.len() != outcomes.len() { + return Err(EventError::InvalidWirePayload); + } + let mut square_sum = 0.0_f64; + for (forecast, outcome) in forecasts.iter().zip(outcomes) { + let target = if *outcome { 1.0 } else { 0.0 }; + let residual = forecast.value() - target; + square_sum += residual * residual; + } + #[allow(clippy::cast_precision_loss)] + Ok(square_sum / forecasts.len() as f64) +} + #[cfg(test)] mod tests { use super::EventConfidence; @@ -56,5 +79,9 @@ mod tests { EventConfidence::new(f64::NAN), Err(EventError::InvalidEventConfidence) ); + let one = EventConfidence::certain().expect("certain"); + assert!((one.value() - 1.0).abs() < 1e-15); + let miss = super::mention_brier_score(&[one], &[false]).expect("miss"); + assert!((miss - 1.0).abs() < 1e-15); } } diff --git a/crates/event_core/src/lib.rs b/crates/event_core/src/lib.rs index 9a1715b1..13050cb0 100644 --- a/crates/event_core/src/lib.rs +++ b/crates/event_core/src/lib.rs @@ -19,6 +19,8 @@ mod role; /// Finite confidence on the closed unit interval. pub use confidence::EventConfidence; +/// Mean squared error of mention probabilities against binary truth. +pub use confidence::mention_brier_score; /// Fail-closed event-ontology errors. pub use error::EventError; /// Opaque event-instance identifier. diff --git a/crates/event_core/tests/confidence_calibration_contract.rs b/crates/event_core/tests/confidence_calibration_contract.rs new file mode 100644 index 00000000..631e1468 --- /dev/null +++ b/crates/event_core/tests/confidence_calibration_contract.rs @@ -0,0 +1,37 @@ +//! Mention confidence recovers known Brier scores against binary truth. + +use event_core::{EventConfidence, EventError, mention_brier_score}; + +#[test] +fn perfectly_calibrated_forecasts_recover_zero_brier() { + let forecasts = [ + EventConfidence::new(0.0).expect("0"), + EventConfidence::new(1.0).expect("1"), + EventConfidence::new(0.0).expect("0"), + EventConfidence::new(1.0).expect("1"), + ]; + let outcomes = [false, true, false, true]; + let score = mention_brier_score(&forecasts, &outcomes).expect("brier"); + assert!(score.abs() < 1e-15, "perfect Brier {score}"); +} + +#[test] +fn constant_half_recovers_quarter_and_mismatches_fail_closed() { + let forecasts = [ + EventConfidence::new(0.5).expect("half"), + EventConfidence::new(0.5).expect("half"), + ]; + let outcomes = [false, true]; + let score = mention_brier_score(&forecasts, &outcomes).expect("half"); + let residual = score - 0.25; + let rmse = (residual * residual).sqrt(); + assert!(rmse < 1e-15, "Brier RMSE {rmse}"); + assert_eq!( + mention_brier_score(&forecasts, &[true]), + Err(EventError::InvalidWirePayload) + ); + assert_eq!( + mention_brier_score(&[], &[]), + Err(EventError::InvalidWirePayload) + ); +} diff --git a/docs/TRACEABILITY.md b/docs/TRACEABILITY.md index 3cf3e316..c783444c 100644 --- a/docs/TRACEABILITY.md +++ b/docs/TRACEABILITY.md @@ -13,8 +13,8 @@ The full APA 7th standards/literature register remains `docs/research/standards- | six distinct clocks and uncertain intervals | PRD; ADR 0002; ISO 24617-1:2012; Hobbs & Pan (2017) | merged PR #8 `temporal_core` on protected main; PR #5 historical lineage only | implemented-main | | Allen relation algebra/bounded closure | ADR 0002; Allen (1983) | merged PR #9 `temporal_core` path-consistency on protected main | implemented-main | | forward-only transition subgraph | PRD; ADR 0002/0003 | `relation_graph` on protected main | implemented-main | -| event ontology/evidence mentions | PRD; ADR 0003 | `event_core` mention/instance separation on protected main; `persistence_postgres` mention SQL implemented-main refuses mention-as-instance; event-instance SQL (#39 implemented-main) refuses inverted windows; full intelligence stack remaining | partial | -| time-varying cross-classified multiple membership | PRD; ADR 0003 | `membership_core` network and Kish ESS on protected main; nested ICC + cross-classified/multiple-membership refusal (this PR); full multilevel/MMMC/ESEM estimators remaining | partial | +| event ontology/evidence mentions | PRD; ADR 0003 | `event_core` mention/instance separation on protected main; `persistence_postgres` mention SQL implemented-main refuses mention-as-instance; event-instance SQL (#39 implemented-main) refuses inverted windows; Brier calibration on the active PR; full intelligence stack remaining | partial | +| time-varying cross-classified multiple membership | PRD; ADR 0003 | `membership_core` network on protected main; multilevel estimators remaining | partial | | leakage-safe availability/cutoff snapshots | PRD; ADR 0002/0013 | `corpus_split` on protected main | implemented-main | | recovery metrics (RMSE, bias, coverage, graph, temporal order, Monte Carlo SE gates) | PRD; Test Strategy; ADR 0007/0014 | `validation_core` on protected main (PR #19); SE-aware Monte Carlo gates included | implemented-main | | PostgreSQL bitemporal/lineage persistence | ADR 0013; Architecture/ERD | `persistence_postgres` migration contracts, in-memory adapters, live SQL session/document SQL port, tenant RLS (`0002` + session GUC/role helpers), `DATABASE_URL` SQLx gate, optional `live-sqlx` `PgPool` driver, exact-head live PostgreSQL CI with isolation proof, append-only immutability triggers (`0004`), temporal interval ordering CHECKs (`0005`), typed membership assignment (`0006` implemented-main), event-relation/mention/instance SQL (#37–#39 implemented-main), source-artifact SQL (#40 implemented-main), audit-event SQL (#41 implemented-main), concurrent document-write stress (#43 implemented-main), backup/restore integrity revalidation (#44 implemented-main), typed `text_segment` SQL insert/cutoff lookup (active PR); remaining physical ERD constraints including `document_record` FK on `text_segment` | partial | diff --git a/docs/research/mention-confidence-brier.md b/docs/research/mention-confidence-brier.md new file mode 100644 index 00000000..afb7382b --- /dev/null +++ b/docs/research/mention-confidence-brier.md @@ -0,0 +1,27 @@ +# Mention-confidence Brier score + +## Scope + +This note doctors the `event_core` calibration contract for fallible event mentions: + +1. mention confidence is a probability on `[0, 1]`; +2. `mention_brier_score` is the mean squared error against binary truth; +3. empty or length-mismatched streams fail closed. + +TDT/CHRONOS promotion remains on the event-intelligence active PR. No database migration is allocated. + +## Authoritative sources + +Brier, G. W. (1950). Verification of forecasts expressed in terms of probability. *Monthly Weather Review, 78*(1), 1–3. https://doi.org/10.1175/1520-0493(1950)078<0001:VOFEIT>2.0.CO;2 + +Gneiting, T., & Raftery, A. E. (2007). Strictly proper scoring rules, prediction, and estimation. *Journal of the American Statistical Association, 102*(477), 359–378. https://doi.org/10.1198/016214506000001437 + +## Application + +Brier (1950) defines the mean squared error of a probability forecast. Gneiting and Raftery (2007) treat the Brier score as a strictly proper scoring rule, so a mention that is certain when true and impossible when false is uniquely optimal. TEPP therefore scores mention confidence against known binary outcomes rather than treating a high score as an event instance (Brier, 1950; Gneiting & Raftery, 2007). + +## Verification + +- forecasts `(0,1,0,1)` against outcomes `(false,true,false,true)` recover Brier `0`; +- constant `0.5` against mixed outcomes recovers `0.25` with computed residual RMSE; +- empty and mismatched streams return `InvalidWirePayload`. diff --git a/docs/validation/temporal-event-foundation.md b/docs/validation/temporal-event-foundation.md index 7b358266..963a75a4 100644 --- a/docs/validation/temporal-event-foundation.md +++ b/docs/validation/temporal-event-foundation.md @@ -23,6 +23,7 @@ This report tracks exact-head scientific and engineering evidence required befor | Leakage-safe splits | `corpus_split` | implemented-main | — | cutoff + co-partition tests | Task 9 / PR #17 | | Truth corpora / manifests | `tepp_simulation` | implemented-main | — | deterministic generator tests | Task 10 / PR #18 | | Recovery metrics | `validation_core` | implemented-main | — | RMSE/bias/coverage/MC gates | Task 11 / PR #19 | +| Mention-confidence Brier score | `event_core` | active-PR | calibration vs binary truth | perfect 0 / half 0.25 RMSE | ADR 0003; `docs/research/mention-confidence-brier.md` | | Checkpoint is not the estimator | `checkpoint_authority` | accepted-target | active PR | refuse checkpoint-as-estimator + unvalidated artifact + recovery vs estimator collapse | ADR 0001/0014 | | Versioned API/export contracts | `tepp_api` | implemented-main | naruon HTTP interchange | unknown-field/version/limit + naruon HTTPS interchange tests | Task 12 / PR #21; live HTTP service remaining | | TDT/CHRONOS evidence-status gates | `event_core` | active-PR | PR #50 | admission + first-story rates | known-stream miss/FA; full tracking/calibration/schema extraction remains future; ADR 0016; `docs/research/event-intelligence-status-gates.md` |