From 3f44f226d3826438f08fce62b2632c224dcd200f Mon Sep 17 00:00:00 2001 From: Fernando Fernandes Date: Mon, 21 Sep 2026 01:27:13 +0200 Subject: [PATCH 1/2] Sync AgentGuidelines 0.0.34 --- .../skills/agent-guidelines-audit/SKILL.md | 1 + AgentGuidelines/CHANGELOG.md | 7 ++++ AgentGuidelines/Guidelines/Packages.md | 4 ++ AgentGuidelines/README.md | 4 +- .../Scripts/validate_guidelines.swift | 12 ++++++ AgentGuidelines/Tests/run_tests.swift | 38 +++++++++++++++++++ AgentGuidelines/VERSION | 2 +- 7 files changed, 65 insertions(+), 3 deletions(-) diff --git a/AgentGuidelines/.agents/skills/agent-guidelines-audit/SKILL.md b/AgentGuidelines/.agents/skills/agent-guidelines-audit/SKILL.md index 75510ae..ead9b72 100644 --- a/AgentGuidelines/.agents/skills/agent-guidelines-audit/SKILL.md +++ b/AgentGuidelines/.agents/skills/agent-guidelines-audit/SKILL.md @@ -31,6 +31,7 @@ Review the actual change rather than only checking whether files exist: - Check tests for the required framework, mirrored paths, shared tags, Given/When/Then structure, deterministic seams, and coverage of changed behavior and failure paths. - Trace every new or changed stateful, asynchronous, fallible, or lifecycle-oriented behavior and verify that its owning artifact emits privacy-safe AppLogger events for the meaningful success, failure, cancellation, recovery, and state-transition outcomes needed to diagnose it. Dependency declaration and target linkage alone do not establish logging coverage. Accept silence for pure values or utilities only when there is no meaningful event boundary and the implementation handoff records that deliberate decision. - Check logging ownership, subsystem, categories, emoji, privacy, severity, metadata stability, noise controls, and focused formatter or sink tests when logging changed. +- For every in-scope package that emits logs, resolve its canonical emoji from local instructions or documentation and the package list on the current `main` branch of the [ThatFactory Swift Package Collection](https://github.com/thatfactory/swift-package-collection). Enumerate the registered package entries and require each canonical emoji to identify exactly one package, treating visually identical presentation-selector variants as a collision. Confirm the package documentation, package-local logging gateway, and registry entry use the same emoji. When creating a package or assigning or changing its emoji, verify the intended emoji is unused before accepting it and require the registry update in the publication work. A collision, a missing registry entry for a published package, an unreachable current registry, or disagreement among these sources blocks audit completion and release readiness. - For every Apple-platform application or Swift package in scope, except the AppLogger provider repository itself, verify integration with the shared [Logging guide](../../../Guidelines/Logging.md): confirm the AppLogger dependency is declared, the `AppLogger` library product is linked to every target that emits diagnostics, and any new project has it available in its primary runtime target before its first log call. Search the actual package or Xcode dependency graph rather than relying on an `import` alone, and treat `print`, direct `Logger` instances, or duplicate logging backends as incomplete integration when they emit project diagnostics. When implementation is authorized, add or repair the dependency and target linkage and migrate affected calls while preserving the guide's ownership, subsystem, category, emoji, privacy, severity, and noise rules; report an exact blocker when target or platform constraints make safe integration ambiguous. - Inspect dependency manifests, resolver or lock files, Xcode package references, vendored source or binary frameworks, and equivalent dependency declarations. Compare the change with the baseline and identify every new third-party dependency or expansion of an existing third-party dependency into a new target or runtime role. Apply the shared [external dependency policy](../../../Guidelines/Development.md#external-dependencies): require explicit repository-owner approval before the dependency is introduced and require the durable exception record in repository documentation. Do not infer approval merely from an execution plan, pull-request description, implementation convenience, package popularity, or the dependency already appearing in the diff. Treat an unapproved or undocumented third-party dependency as a blocker to completion. Do not flag Apple system frameworks, the Swift standard library, ThatFactory-owned packages, or guideline-mandated tooling used only for its documented tooling role. If a newly resolved transitive third-party package will be linked into or shipped with the product, verify that its owning direct dependency is covered by an approved exception rather than dismissing it solely because it is transitive. - Search dependency manifests, generated directories, project files, and documentation for CocoaPods or Carthage adoption. The shared [external dependency policy](../../../Guidelines/Development.md#external-dependencies) forbids both without an exception path and requires Swift Package Manager for package dependencies. When implementation is authorized, remove newly introduced adoption and its generated or configuration files; report pre-existing adoption as a completion blocker when safe migration is outside the task scope. diff --git a/AgentGuidelines/CHANGELOG.md b/AgentGuidelines/CHANGELOG.md index dcb012e..d82dd17 100644 --- a/AgentGuidelines/CHANGELOG.md +++ b/AgentGuidelines/CHANGELOG.md @@ -2,6 +2,13 @@ All notable changes to this project are documented in this file. +## [0.0.34] - 2026-09-20 + +### Changed + +- Required every ThatFactory package emoji to be unique in the Swift Package Collection and synchronized across package documentation, the logging gateway, and the collection registry. +- Extended the completion audit to block package publication when emoji uniqueness cannot be verified or the registered and emitted identities disagree. + ## [0.0.33] - 2026-09-19 ### Added diff --git a/AgentGuidelines/Guidelines/Packages.md b/AgentGuidelines/Guidelines/Packages.md index c33cfbb..2de2bec 100644 --- a/AgentGuidelines/Guidelines/Packages.md +++ b/AgentGuidelines/Guidelines/Packages.md @@ -114,6 +114,10 @@ Accept a deviation only when the nearest applicable `AGENTS.md`, or durable docu Packages own any diagnostics emitted by their implementation. Follow the shared [logging guide](Logging.md) for AppLogger usage, subsystem identity, package emoji prefixes, domain-owned categories, concise messages, privacy, and test coverage. A consuming application must not reproduce package-internal logs. +Before assigning or changing a package's canonical emoji, inspect the package list on the current `main` branch of the [ThatFactory Swift Package Collection](https://github.com/thatfactory/swift-package-collection) and confirm that no other package uses it. Each package emoji must be unique across ThatFactory so a log prefix identifies one package unambiguously. Treat visually identical emoji spellings that differ only by presentation selectors as the same emoji; do not use an encoding variation to create an apparent distinction. + +Declare the selected emoji in the package's local instructions or documentation, use that exact emoji in its package-local logging gateway, and add or update the matching Swift Package Collection entry as part of the package's publication work. An existing collision, a missing registry entry for a published package, or disagreement among the registry, documentation, and emitted prefix blocks release readiness until reconciled. + ## Development workflow 1. Read the package's local `AGENTS.md`, README, DocC, and public API before changing behavior. diff --git a/AgentGuidelines/README.md b/AgentGuidelines/README.md index cec2d20..caab867 100644 --- a/AgentGuidelines/README.md +++ b/AgentGuidelines/README.md @@ -91,7 +91,7 @@ From the consumer repository root, install a tagged release: git subtree add \ --prefix=AgentGuidelines \ https://github.com/thatfactory/agent-guidelines.git \ - 0.0.33 \ + 0.0.34 \ --squash ``` @@ -147,7 +147,7 @@ Review the target release's changelog, then pull it deliberately: git subtree pull \ --prefix=AgentGuidelines \ https://github.com/thatfactory/agent-guidelines.git \ - 0.0.33 \ + 0.0.34 \ --squash ``` diff --git a/AgentGuidelines/Scripts/validate_guidelines.swift b/AgentGuidelines/Scripts/validate_guidelines.swift index 39b58d6..7c2dddf 100755 --- a/AgentGuidelines/Scripts/validate_guidelines.swift +++ b/AgentGuidelines/Scripts/validate_guidelines.swift @@ -576,6 +576,14 @@ func validatePackageCompilerSettingsGuideline(_ errors: inout [String]) { for feature in packageUpcomingFeatures where !contents.contains(".enableUpcomingFeature(\"\(feature)\")") { errors.append("Guidelines/Packages.md: missing required SwiftPM upcoming feature '\(feature)'") } + let emojiRequired = [ + "Each package emoji must be unique across ThatFactory": "uniqueness policy", + "visually identical emoji spellings": "presentation-selector collision policy", + "matching Swift Package Collection entry": "registry synchronization", + ] + for (value, description) in emojiRequired where !contents.contains(value) { + errors.append("Guidelines/Packages.md: missing package emoji \(description): '\(value)'") + } } /// Validates the shared App Store metadata workflow. @@ -679,6 +687,10 @@ func validateAuditSkill(_ errors: inout [String]) { "AppLogger": "AppLogger integration audit", "Logging.md": "shared Logging guide reference", "Dependency declaration and target linkage alone": "lifecycle observability coverage audit", + "Enumerate the registered package entries": "package emoji registry enumeration", + "visually identical presentation-selector variants as a collision": + "package emoji presentation-selector collision audit", + "blocks audit completion and release readiness": "package emoji audit stopping rule", "## Audit documentation consistency": "documentation drift audit", "Known stale documentation blocks completion": "stale documentation stopping rule", "## Audit documentation formatting": "documentation formatting audit", diff --git a/AgentGuidelines/Tests/run_tests.swift b/AgentGuidelines/Tests/run_tests.swift index 847eeab..57fb159 100755 --- a/AgentGuidelines/Tests/run_tests.swift +++ b/AgentGuidelines/Tests/run_tests.swift @@ -290,6 +290,25 @@ let tests: [(String, () throws -> Void)] = [ } } ), + ( + "repository validator rejects missing package emoji uniqueness policy", + { + try withTemporaryDirectory { temporary in + let fixture = temporary.appendingPathComponent("repository") + try copyRepositoryFixture(to: fixture) + let guideline = fixture.appendingPathComponent("Guidelines/Packages.md") + var contents = try String(contentsOf: guideline, encoding: .utf8) + contents = contents.replacingOccurrences( + of: "Each package emoji must be unique across ThatFactory", + with: "Package emojis should be recognizable" + ) + try write(contents, to: guideline) + let result = try run([fixture.appendingPathComponent("Scripts/validate_guidelines.swift").path]) + try require(!result.succeeded, "missing package emoji uniqueness policy unexpectedly passed") + try require(result.output.contains("missing package emoji uniqueness policy"), result.output) + } + } + ), ( "repository validator rejects missing package audit section", { @@ -326,6 +345,25 @@ let tests: [(String, () throws -> Void)] = [ } } ), + ( + "repository validator rejects package emoji audit drift", + { + try withTemporaryDirectory { temporary in + let fixture = temporary.appendingPathComponent("repository") + try copyRepositoryFixture(to: fixture) + let skill = fixture.appendingPathComponent(".agents/skills/agent-guidelines-audit/SKILL.md") + var contents = try String(contentsOf: skill, encoding: .utf8) + contents = contents.replacingOccurrences( + of: "Enumerate the registered package entries", + with: "Inspect the package registry" + ) + try write(contents, to: skill) + let result = try run([fixture.appendingPathComponent("Scripts/validate_guidelines.swift").path]) + try require(!result.succeeded, "package emoji audit drift unexpectedly passed") + try require(result.output.contains("missing package emoji registry enumeration"), result.output) + } + } + ), ( "repository validator rejects missing observability adoption guidance", { diff --git a/AgentGuidelines/VERSION b/AgentGuidelines/VERSION index cd9d21e..bb951c8 100644 --- a/AgentGuidelines/VERSION +++ b/AgentGuidelines/VERSION @@ -1 +1 @@ -0.0.33 +0.0.34 From 2392bfe3b534b3b6fe397b97cbca59593e5ffe37 Mon Sep 17 00:00:00 2001 From: Fernando Fernandes Date: Mon, 21 Sep 2026 01:33:25 +0200 Subject: [PATCH 2/2] Add immutable lexicon model runtime --- AGENTS.md | 25 +- CHANGELOG.md | 11 + README.md | 20 +- .../LexiconKit/LexiconKit.docc/LexiconKit.md | 12 +- Sources/LexiconKit/LexiconLogging.swift | 18 ++ Sources/LexiconKit/LexiconModel.swift | 303 ++++++++++++++++++ Sources/LexiconKit/LexiconModelTypes.swift | 54 ++++ Tests/LexiconKitTests/LexiconModelTests.swift | 200 ++++++++++++ VERSION | 2 +- 9 files changed, 617 insertions(+), 28 deletions(-) create mode 100644 Sources/LexiconKit/LexiconLogging.swift create mode 100644 Sources/LexiconKit/LexiconModel.swift create mode 100644 Sources/LexiconKit/LexiconModelTypes.swift create mode 100644 Tests/LexiconKitTests/LexiconModelTests.swift diff --git a/AGENTS.md b/AGENTS.md index 9b4dd56..2bc24f4 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -30,9 +30,6 @@ Read only the guides relevant to the task: For an application that uses Redux, also read [Redux architecture](AgentGuidelines/Guidelines/Architecture/Redux.md). -Keep the following observability contract in the consumer repository's root `AGENTS.md` so implementation agents treat runtime diagnostics as part of lifecycle work. Copy it unchanged and update it when the marker version changes in this template. - -```md ## Runtime Observability @@ -40,11 +37,7 @@ Treat privacy-safe runtime observability as part of implementing or changing sta Every ThatFactory package log starts with its canonical emoji and uses its own stable subsystem. Never log credentials, account or record identifiers, share URLs, captured content, images, or other user-generated values as public metadata. Keep pure values and utilities silent when they have no meaningful diagnostic event; record that deliberate decision in the implementation handoff instead of adding initializer or property-access noise. Follow [Logging](AgentGuidelines/Guidelines/Logging.md) for ownership, privacy, severity, message design, and tests. -``` - -Keep the following marked external-dependency contract in the consumer repository's root `AGENTS.md` so implementation agents receive the rule directly before they make dependency choices. Copy it unchanged and update it when the marker version changes in this template. -```md ## External Dependency Policy @@ -58,11 +51,7 @@ Apple system frameworks and the Swift standard library are not third-party depen Follow [Development workflow](AgentGuidelines/Guidelines/Development.md) for the detailed policy. -``` -Keep the following documentation-maintenance contract in the consumer repository's root `AGENTS.md` so implementation agents receive it directly rather than only through a linked guide. Copy it unchanged and update it when the marker version changes in this template. - -```md ## Documentation Maintenance @@ -72,11 +61,7 @@ Update documentation when a change alters durable or core feature behavior or an Do not create documentation churn for incidental implementation details that are not durable and do not affect an existing documented claim. Follow [Documentation](AgentGuidelines/Guidelines/Documentation.md) for detailed scope and the completion checklist. -``` - -Keep the following marked code-review contract in the consumer repository's root `AGENTS.md` so it is loaded directly for root-level Codex and pull-request work. Copy it unchanged and update it when the marker version changes in this template; a Markdown link to the detailed workflow is not an instruction include. -```md ## Code Review Rules @@ -98,16 +83,10 @@ Automatic Codex review is the initial Codex review. Do not request a manual Code ## Codex review scope For consumer pull requests, do not substantively review `AgentGuidelines/**` after exact tagged-tree provenance has been verified. Verify its `VERSION`, compare its tree with the matching central tag, and verify the required `.gitattributes` rule. If provenance does not match exactly, review the subtree contents and stop the merge. Report substantive guideline feedback against the central `agent-guidelines` pull request. -``` - -The marked block is intentional controlled duplication of the shared review policy. The tracked, synchronized subtree is reviewed centrally in `thatfactory/agent-guidelines`; the root-level instructions ensure the review contract and subtree scope are loaded even when Codex starts from the repository root. - ## Physical folder map -Replace these examples with exact repository paths: - | Role | Physical folder | -|---|---| +| --- | --- | | Package sources | `Sources/LexiconKit/` | | DocC catalog | `Sources/LexiconKit/LexiconKit.docc/` | | Unit tests | `Tests/LexiconKitTests/` | @@ -123,4 +102,4 @@ Replace these examples with exact repository paths: - Do not encode CEFR, product progression, CloudKit, or application-specific deduplication policy. - Store only opaque host-owned media references, never framework image objects or temporary URLs. -- LexiconKit currently emits no runtime diagnostics because its public surface contains only pure domain values and initializers. Do not log value construction or vocabulary content. If a stateful or fallible lifecycle is added, route its diagnostics through a package-local `LexiconLogging` gateway with subsystem `com.thatfactory.lexiconkit` and canonical emoji 📖. +- Model open success and failure are logged through `LexiconLogging` with subsystem `com.thatfactory.lexiconkit` and canonical emoji 📖. Never log model paths, lookup terms, glosses, or vocabulary content. diff --git a/CHANGELOG.md b/CHANGELOG.md index e0d2299..0f21717 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,17 @@ All notable changes to LexiconKit are documented here. ## Unreleased +## 0.3.0 — 2026-09-21 + +### Added + +- Add a memory-mapped, language-neutral lexicon model reader with synchronous exact term and gloss lookup. +- Add explicit model verification for CI and installed-resource integrity checks. + +### Changed + +- Adopt Agent Guidelines `0.0.34`. + ## 0.2.0 — 2026-09-20 ### Added diff --git a/README.md b/README.md index cdebf4b..d13bccc 100644 --- a/README.md +++ b/README.md @@ -11,9 +11,9 @@ # LexiconKit -LexiconKit is a reusable, UI-agnostic domain package for personal vocabulary collections, definitions, provenance, lightweight metadata, and opaque references to host-owned media. +LexiconKit is a reusable, UI-agnostic domain package for personal vocabulary collections and immutable lexical model lookup. -LexiconKit provides persistence-friendly values for vocabulary entries, terms, optional grammatical gender, definitions, definition provenance, tags, and opaque host-owned media references. Dictionary lookup, translation, language-specific display articles, exercises, persistence frameworks, synchronization, and UI remain outside its boundary. +LexiconKit provides persistence-friendly vocabulary values plus a synchronous, language-neutral reader for versioned packed lexicon artifacts. Translation, fuzzy or semantic inference, language-specific display articles, exercises, persistence frameworks, synchronization, and UI remain outside its boundary. ```swift let entry = LexiconEntry( @@ -32,13 +32,27 @@ let entry = LexiconEntry( ) ``` +Open a bundled model lazily and perform exact lemma or inflected-form lookup without deserializing the corpus: + +```swift +let model = try LexiconModel( + contentsOf: modelURL, + manifestURL: manifestURL +) +let result = try model.matchSense( + for: "See", + gloss: "lake", + languageCodes: ["en"] +) +``` + ## Documentation API documentation is published with DocC after a GitHub release. See the [LexiconKit documentation](https://thatfactory.github.io/lexiconkit/documentation/lexiconkit/). ## Runtime diagnostics -LexiconKit currently emits no runtime diagnostics. Its public surface contains only pure domain values and initializers; persistence, synchronization, validation workflows, and application lifecycle remain host responsibilities. Logging value construction would add noise and could expose vocabulary content. If the package later gains stateful or fallible runtime behavior, its diagnostics will use a package-local `LexiconLogging` gateway, subsystem `com.thatfactory.lexiconkit`, and the canonical 📖 prefix. +LexiconKit logs privacy-safe model-open success and failure through its package-local gateway, subsystem `com.thatfactory.lexiconkit`, and canonical 📖 prefix. It never logs model paths, lookup terms, glosses, or vocabulary content. Individual lookups remain silent. ## Requirements diff --git a/Sources/LexiconKit/LexiconKit.docc/LexiconKit.md b/Sources/LexiconKit/LexiconKit.docc/LexiconKit.md index 6702345..593cd9a 100644 --- a/Sources/LexiconKit/LexiconKit.docc/LexiconKit.md +++ b/Sources/LexiconKit/LexiconKit.docc/LexiconKit.md @@ -1,6 +1,6 @@ # ``LexiconKit`` -Model persistence-friendly vocabulary values without coupling consumers to UI, learning progression, or storage frameworks. +Model persistence-friendly vocabulary values and query immutable lexical models without coupling consumers to UI, learning progression, or storage frameworks. ## Overview @@ -10,6 +10,8 @@ Create a ``LexiconEntry`` from a ``LexiconTerm`` and one or more ``LexiconDefini All public values are `Codable`, `Hashable`, and `Sendable`. Stable entry and definition identifiers plus explicit creation and modification timestamps let a host persist and merge values using its own policy. +``LexiconModel`` memory-maps a versioned immutable artifact and exposes synchronous exact term and gloss matching. The storage format, dense identifiers, and lookup indexes remain private. Model initialization validates structural metadata without scanning the full corpus; ``LexiconModel/verifyArtifact(at:against:)`` provides the explicit full-checksum path for CI and resource installation. + ## Topics ### Entries @@ -27,3 +29,11 @@ All public values are `Codable`, `Hashable`, and `Sendable`. Stable entry and de - ``LexiconTag`` - ``LexiconAssetReference`` + +### Lexical models + +- ``LexiconModel`` +- ``LexiconModelLookupResult`` +- ``LexiconModelSense`` +- ``LexiconModelGloss`` +- ``LexiconPartOfSpeech`` diff --git a/Sources/LexiconKit/LexiconLogging.swift b/Sources/LexiconKit/LexiconLogging.swift new file mode 100644 index 0000000..01d6c5a --- /dev/null +++ b/Sources/LexiconKit/LexiconLogging.swift @@ -0,0 +1,18 @@ +import AppLogger + +enum LexiconLogging { + static func modelOpened() { + logger().log(level: .info, "📖 Lexicon model opened") + } + + static func modelOpenFailed() { + logger().log(level: .error, "📖 Lexicon model open failed") + } + + private static func logger() -> AppLogger { + AppLogger( + subsystem: "com.thatfactory.lexiconkit", + category: "model" + ) + } +} diff --git a/Sources/LexiconKit/LexiconModel.swift b/Sources/LexiconKit/LexiconModel.swift new file mode 100644 index 0000000..0faf8f2 --- /dev/null +++ b/Sources/LexiconKit/LexiconModel.swift @@ -0,0 +1,303 @@ +import CryptoKit +public import Foundation + +/// A synchronous, read-only lexical model backed by a mapped immutable artifact. +public final class LexiconModel: @unchecked Sendable { + private static let headerSize = 256 + private static let magic = Data("TFLEX001".utf8) + + private let counts: [Int] + private let data: Data + private let sections: [(offset: Int, length: Int)] + + /// The model's BCP 47 language code. + public let languageCode: String + + /// The semantic model version. + public let modelVersion: Int + + /// The runtime schema version. + public let schemaVersion: Int + + /// Opens and maps an immutable model without scanning or deserializing its corpus. + public init(contentsOf modelURL: URL, manifestURL: URL) throws { + do { + let manifest = try JSONDecoder().decode( + Manifest.self, + from: Data(contentsOf: manifestURL) + ) + let data = try Data(contentsOf: modelURL, options: .mappedIfSafe) + guard data.count >= Self.headerSize, data.prefix(8) == Self.magic else { + throw LexiconModelError.corruptArtifact("header") + } + let formatVersion = Self.uint32(data, at: 8) + guard formatVersion == 1 else { + throw LexiconModelError.unsupportedFormat(formatVersion) + } + let schemaVersion = Self.uint32(data, at: 12) + let modelVersion = Self.uint32(data, at: 16) + guard schemaVersion == 1 else { + throw LexiconModelError.unsupportedFormat(schemaVersion) + } + guard Int64(data.count) == manifest.artifact.byteCount, + Int(schemaVersion) == manifest.schemaVersion, + Int(modelVersion) == manifest.modelVersion + else { throw LexiconModelError.corruptArtifact("manifest-metadata") } + + var counts: [Int] = [] + for index in 0..<6 { + guard let count = Int(exactly: Self.uint64(data, at: 24 + index * 8)) else { + throw LexiconModelError.corruptArtifact("record-count") + } + counts.append(count) + } + var sections: [(offset: Int, length: Int)] = [] + for index in 0..<7 { + guard + let offset = Int(exactly: Self.uint64(data, at: 72 + index * 16)), + let length = Int(exactly: Self.uint64(data, at: 80 + index * 16)), + offset >= Self.headerSize, + length >= 0, + offset <= data.count, + length <= data.count - offset + else { throw LexiconModelError.corruptArtifact("section-bounds") } + sections.append((offset, length)) + } + guard Self.hasRecordLayout(sections[0].length, count: counts[5], stride: 8), + Self.hasRecordLayout(sections[2].length, count: counts[0], stride: 16), + Self.hasRecordLayout(sections[3].length, count: counts[1], stride: 12), + Self.hasRecordLayout(sections[4].length, count: counts[2], stride: 12), + Self.hasRecordLayout(sections[5].length, count: counts[3], stride: 12), + Self.hasRecordLayout(sections[6].length, count: counts[4], stride: 8) + else { throw LexiconModelError.corruptArtifact("section-length") } + + self.counts = counts + self.data = data + self.languageCode = manifest.languageCode + self.modelVersion = Int(modelVersion) + self.schemaVersion = Int(schemaVersion) + self.sections = sections + LexiconLogging.modelOpened() + } catch { + LexiconLogging.modelOpenFailed() + throw error + } + } + + /// Performs the full checksum verification intended for CI, tests, or downloaded-resource installation. + public static func verifyArtifact(at modelURL: URL, against manifestURL: URL) throws { + let manifest = try JSONDecoder().decode( + Manifest.self, + from: Data(contentsOf: manifestURL) + ) + let attributes = try FileManager.default.attributesOfItem(atPath: modelURL.path) + guard (attributes[.size] as? NSNumber)?.int64Value == manifest.artifact.byteCount else { + throw LexiconModelError.corruptArtifact("artifact-byte-count") + } + let handle = try FileHandle(forReadingFrom: modelURL) + defer { try? handle.close() } + var hasher = SHA256() + while let bytes = try handle.read(upToCount: 1_048_576), !bytes.isEmpty { + hasher.update(data: bytes) + } + let digest = hasher.finalize().map { String(format: "%02x", $0) }.joined() + guard digest == manifest.artifact.sha256 else { + throw LexiconModelError.corruptArtifact("artifact-checksum") + } + } + + /// Returns all exact lexical senses for a lemma, source form, or generated alias. + public func lookup(_ term: String) throws -> LexiconModelLookupResult { + let normalizedTerm = term.trimmingCharacters(in: .whitespacesAndNewlines) + .precomposedStringWithCanonicalMapping + guard !normalizedTerm.isEmpty else { return .notFound } + let postings = try postings(for: normalizedTerm) + let exact = postings.filter { ($0.flags & 3) != 0 } + return try result(for: exact.isEmpty ? postings : exact) + } + + /// Narrows exact term candidates to senses with an exact normalized gloss in an allowed language. + public func matchSense( + for term: String, + gloss: String, + languageCodes: Set + ) throws -> LexiconModelLookupResult { + let termResult = try lookup(term) + let candidates: [LexiconModelSense] + switch termResult { + case .multiple(let senses): candidates = senses + case .notFound: return .notFound + case .unique(let sense): candidates = [sense] + } + let normalized = Self.normalizedGloss(gloss) + let matched = candidates.filter { candidate in + candidate.glosses.contains { modelGloss in + languageCodes.contains(modelGloss.languageCode) + && Self.normalizedGloss(modelGloss.text) == normalized + } + } + return Self.result(for: matched) + } + + // MARK: - Private + + private func result(for postings: [(lexeme: UInt32, flags: UInt8)]) throws -> LexiconModelLookupResult { + var matchedSenses: [LexiconModelSense] = [] + for posting in postings { + matchedSenses.append(contentsOf: try senses(forLexeme: posting.lexeme)) + } + return Self.result(for: matchedSenses) + } + + private static func result(for senses: [LexiconModelSense]) -> LexiconModelLookupResult { + switch senses.count { + case 0: .notFound + case 1: .unique(senses[0]) + default: .multiple(senses) + } + } + + private func postings(for term: String) throws -> [(lexeme: UInt32, flags: UInt8)] { + var low = 0 + var high = counts[3] + while low < high { + let middle = (low + high) / 2 + let text = try string(try uint32(inSection: 5, record: middle, stride: 12, field: 0)) + if text.utf8.lexicographicallyPrecedes(term.utf8) { low = middle + 1 } else { high = middle } + } + guard low < counts[3] else { return [] } + let key = try string(try uint32(inSection: 5, record: low, stride: 12, field: 0)) + guard key == term else { return [] } + let start = Int(try uint32(inSection: 5, record: low, stride: 12, field: 4)) + let count = Int(try uint32(inSection: 5, record: low, stride: 12, field: 8)) + guard start <= counts[4], count <= counts[4] - start else { + throw LexiconModelError.corruptArtifact("posting-range") + } + return try (start..<(start + count)).map { index in + let value = try uint64(inSection: 6, record: index, stride: 8, field: 0) + let lexeme = UInt32(truncatingIfNeeded: value) + let flags = UInt8(truncatingIfNeeded: value >> 32) + guard Int(lexeme) < counts[0], flags != 0, flags & ~7 == 0 else { + throw LexiconModelError.corruptArtifact("posting") + } + return (lexeme, flags) + } + } + + private func senses(forLexeme lexemeID: UInt32) throws -> [LexiconModelSense] { + guard Int(lexemeID) < counts[0] else { + throw LexiconModelError.corruptArtifact("lexeme-id") + } + let index = Int(lexemeID) + let lemma = try string(try uint32(inSection: 2, record: index, stride: 16, field: 0)) + let rawPartOfSpeech = try string(try uint32(inSection: 2, record: index, stride: 16, field: 4)) + let senseStart = Int(try uint32(inSection: 2, record: index, stride: 16, field: 8)) + let senseCount = Int(try uint32(inSection: 2, record: index, stride: 16, field: 12)) + guard senseStart <= counts[1], senseCount <= counts[1] - senseStart else { + throw LexiconModelError.corruptArtifact("sense-range") + } + let partOfSpeech: LexiconPartOfSpeech = rawPartOfSpeech == "noun" ? .noun : .other(rawPartOfSpeech) + return try (senseStart..<(senseStart + senseCount)).map { senseIndex in + let glossStart = Int(try uint32(inSection: 3, record: senseIndex, stride: 12, field: 0)) + let glossCount = Int(try uint32(inSection: 3, record: senseIndex, stride: 12, field: 4)) + guard glossStart <= counts[2], glossCount <= counts[2] - glossStart else { + throw LexiconModelError.corruptArtifact("gloss-range") + } + let genderOffset = try offset(inSection: 3, record: senseIndex, stride: 12, field: 8, width: 1) + let gender: LexiconGrammaticalGender? + switch data[genderOffset] { + case 0: gender = nil + case 1: gender = .masculine + case 2: gender = .feminine + case 3: gender = .neuter + default: throw LexiconModelError.corruptArtifact("gender") + } + let glosses = try (glossStart..<(glossStart + glossCount)).map { glossIndex in + LexiconModelGloss( + languageCode: try string( + try uint32(inSection: 4, record: glossIndex, stride: 12, field: 0) + ), + text: try string( + try uint32(inSection: 4, record: glossIndex, stride: 12, field: 4) + ) + ) + } + return LexiconModelSense( + lemma: lemma, + partOfSpeech: partOfSpeech, + grammaticalGender: gender, + glosses: glosses + ) + } + } + + private func string(_ identifier: UInt32) throws -> String { + guard Int(identifier) < counts[5] else { + throw LexiconModelError.corruptArtifact("string-id") + } + let record = try offset(inSection: 0, record: Int(identifier), stride: 8, field: 0, width: 8) + let byteOffset = Int(Self.uint32(data, at: record)) + let byteCount = Int(Self.uint32(data, at: record + 4)) + guard byteOffset <= sections[1].length, byteCount <= sections[1].length - byteOffset else { + throw LexiconModelError.corruptArtifact("string-range") + } + let start = sections[1].offset + byteOffset + guard let value = String(data: data[start..<(start + byteCount)], encoding: .utf8) else { + throw LexiconModelError.corruptArtifact("string-utf8") + } + return value + } + + private func uint32(inSection section: Int, record: Int, stride: Int, field: Int) throws -> UInt32 { + Self.uint32(data, at: try offset(inSection: section, record: record, stride: stride, field: field, width: 4)) + } + + private func uint64(inSection section: Int, record: Int, stride: Int, field: Int) throws -> UInt64 { + Self.uint64(data, at: try offset(inSection: section, record: record, stride: stride, field: field, width: 8)) + } + + private func offset( + inSection section: Int, + record: Int, + stride: Int, + field: Int, + width: Int + ) throws -> Int { + guard record >= 0, stride > 0, field >= 0, field <= stride - width, + record <= (sections[section].length - width - field) / stride + else { throw LexiconModelError.corruptArtifact("record-bounds") } + return sections[section].offset + record * stride + field + } + + private static func normalizedGloss(_ value: String) -> String { + value.trimmingCharacters(in: .whitespacesAndNewlines) + .precomposedStringWithCanonicalMapping + .folding(options: [.caseInsensitive], locale: Locale(identifier: "en_US_POSIX")) + .split(whereSeparator: \.isWhitespace) + .joined(separator: " ") + } + + private static func hasRecordLayout(_ length: Int, count: Int, stride: Int) -> Bool { + length % stride == 0 && length / stride == count + } + + private static func uint32(_ data: Data, at offset: Int) -> UInt32 { + (0..<4).reduce(0) { $0 | UInt32(data[offset + $1]) << UInt32($1 * 8) } + } + + private static func uint64(_ data: Data, at offset: Int) -> UInt64 { + (0..<8).reduce(0) { $0 | UInt64(data[offset + $1]) << UInt64($1 * 8) } + } + + private struct Manifest: Decodable { + struct Artifact: Decodable { + let byteCount: Int64 + let sha256: String + } + + let artifact: Artifact + let languageCode: String + let modelVersion: Int + let schemaVersion: Int + } +} diff --git a/Sources/LexiconKit/LexiconModelTypes.swift b/Sources/LexiconKit/LexiconModelTypes.swift new file mode 100644 index 0000000..fd5e739 --- /dev/null +++ b/Sources/LexiconKit/LexiconModelTypes.swift @@ -0,0 +1,54 @@ +import Foundation + +/// A lexical part of speech without imposing a language-specific closed vocabulary. +public enum LexiconPartOfSpeech: Hashable, Sendable { + /// A noun entry. + case noun + + /// Another source-defined part of speech. + case other(String) +} + +/// One model-provided definition gloss. +public struct LexiconModelGloss: Equatable, Hashable, Sendable { + public let languageCode: String + public let text: String + + public init(languageCode: String, text: String) { + self.languageCode = languageCode + self.text = text + } +} + +/// One lexical sense returned from an immutable model. +public struct LexiconModelSense: Equatable, Hashable, Sendable { + public let glosses: [LexiconModelGloss] + public let grammaticalGender: LexiconGrammaticalGender? + public let lemma: String + public let partOfSpeech: LexiconPartOfSpeech + + public init( + lemma: String, + partOfSpeech: LexiconPartOfSpeech, + grammaticalGender: LexiconGrammaticalGender?, + glosses: [LexiconModelGloss] + ) { + self.glosses = glosses + self.grammaticalGender = grammaticalGender + self.lemma = lemma + self.partOfSpeech = partOfSpeech + } +} + +/// The exact candidates returned by a model lookup. +public enum LexiconModelLookupResult: Equatable, Sendable { + case multiple([LexiconModelSense]) + case notFound + case unique(LexiconModelSense) +} + +/// Stable errors raised while opening or reading an immutable model. +public enum LexiconModelError: Error, Equatable, Sendable { + case corruptArtifact(String) + case unsupportedFormat(UInt32) +} diff --git a/Tests/LexiconKitTests/LexiconModelTests.swift b/Tests/LexiconKitTests/LexiconModelTests.swift new file mode 100644 index 0000000..92bcb1a --- /dev/null +++ b/Tests/LexiconKitTests/LexiconModelTests.swift @@ -0,0 +1,200 @@ +import CryptoKit +import Foundation +import Testing + +@testable import LexiconKit + +struct LexiconModelTests { + @Test func exactLemmaAndFormLookupReturnTheSameSense() throws { + let fixture = try ModelFixture() + let model = try LexiconModel(contentsOf: fixture.modelURL, manifestURL: fixture.manifestURL) + + let lemma = try #require(model.lookup("Haus").uniqueSense) + let form = try #require(model.lookup("Häuser").uniqueSense) + + #expect(lemma == form) + #expect(lemma.lemma == "Haus") + #expect(lemma.grammaticalGender == .neuter) + } + + @Test func exactGlossNarrowsAnAmbiguousHomograph() throws { + let fixture = try ModelFixture() + let model = try LexiconModel(contentsOf: fixture.modelURL, manifestURL: fixture.manifestURL) + + let unresolved = try model.lookup("See") + let lake = try model.matchSense(for: "See", gloss: " LAKE ", languageCodes: ["en"]) + + #expect(unresolved.multipleSenses?.count == 2) + #expect(lake.uniqueSense?.grammaticalGender == .masculine) + } + + @Test func nonNounAndUnknownTermsRemainExplicit() throws { + let fixture = try ModelFixture() + let model = try LexiconModel(contentsOf: fixture.modelURL, manifestURL: fixture.manifestURL) + + #expect(try model.lookup("laufen").uniqueSense?.partOfSpeech == .other("verb")) + #expect(try model.lookup("unbekannt") == .notFound) + } + + @Test func fullVerificationRejectsChangedBytes() throws { + let fixture = try ModelFixture() + var bytes = try Data(contentsOf: fixture.modelURL) + bytes.append(0) + try bytes.write(to: fixture.modelURL) + + #expect(throws: LexiconModelError.corruptArtifact("artifact-byte-count")) { + try LexiconModel.verifyArtifact(at: fixture.modelURL, against: fixture.manifestURL) + } + } + + @Test func lookupRejectsMalformedReachedPosting() throws { + let fixture = try ModelFixture() + var bytes = try Data(contentsOf: fixture.modelURL) + let postingSection = Int(Self.uint64(bytes, at: 168)) + bytes.replaceSubrange(postingSection..<(postingSection + 4), with: Data(repeating: 0xFF, count: 4)) + try bytes.write(to: fixture.modelURL) + try fixture.updateManifestChecksum() + let model = try LexiconModel(contentsOf: fixture.modelURL, manifestURL: fixture.manifestURL) + + #expect(throws: LexiconModelError.corruptArtifact("posting")) { + try model.lookup("Haus") + } + } + + @Test func initializerRejectsUnsupportedFormat() throws { + let fixture = try ModelFixture() + var bytes = try Data(contentsOf: fixture.modelURL) + bytes.replaceSubrange(8..<12, with: Data([2, 0, 0, 0])) + try bytes.write(to: fixture.modelURL) + try fixture.updateManifestChecksum() + + #expect(throws: LexiconModelError.unsupportedFormat(2)) { + try LexiconModel(contentsOf: fixture.modelURL, manifestURL: fixture.manifestURL) + } + } + + @Test func initializerRejectsTruncatedArtifact() throws { + let fixture = try ModelFixture() + var bytes = try Data(contentsOf: fixture.modelURL) + bytes.removeLast(1) + try bytes.write(to: fixture.modelURL) + try fixture.updateManifestChecksum() + + #expect(throws: LexiconModelError.corruptArtifact("section-bounds")) { + try LexiconModel(contentsOf: fixture.modelURL, manifestURL: fixture.manifestURL) + } + } + + private static func uint64(_ data: Data, at offset: Int) -> UInt64 { + (0..<8).reduce(0) { $0 | UInt64(data[offset + $1]) << UInt64($1 * 8) } + } +} + +extension LexiconModelLookupResult { + fileprivate var multipleSenses: [LexiconModelSense]? { + guard case .multiple(let senses) = self else { return nil } + return senses + } + + fileprivate var uniqueSense: LexiconModelSense? { + guard case .unique(let sense) = self else { return nil } + return sense + } +} + +private final class ModelFixture { + let manifestURL: URL + let modelURL: URL + + init() throws { + let root = FileManager.default.temporaryDirectory.appending(path: UUID().uuidString) + try FileManager.default.createDirectory(at: root, withIntermediateDirectories: true) + modelURL = root.appending(path: "German.lexicon") + manifestURL = root.appending(path: "manifest.json") + try Self.artifact().write(to: modelURL) + try updateManifestChecksum() + } + + func updateManifestChecksum() throws { + let bytes = try Data(contentsOf: modelURL) + let digest = SHA256.hash(data: bytes).map { String(format: "%02x", $0) }.joined() + let manifest: [String: Any] = [ + "artifact": ["byteCount": bytes.count, "sha256": digest], + "languageCode": "de", + "modelVersion": 1, + "schemaVersion": 1, + ] + try JSONSerialization.data(withJSONObject: manifest).write(to: manifestURL) + } + + private static func artifact() -> Data { + let strings = ["Haus", "Häuser", "See", "en", "house", "lake", "laufen", "noun", "sea", "verb"] + let identifiers = Dictionary(uniqueKeysWithValues: strings.enumerated().map { ($1, UInt32($0)) }) + var stringRecords = Data() + var stringBytes = Data() + for string in strings { + append(UInt32(stringBytes.count), to: &stringRecords) + append(UInt32(string.utf8.count), to: &stringRecords) + stringBytes.append(Data(string.utf8)) + } + var lexemes = Data() + for row: [UInt32] in [ + [identifiers["Haus"]!, identifiers["noun"]!, 0, 1], + [identifiers["See"]!, identifiers["noun"]!, 1, 2], + [identifiers["laufen"]!, identifiers["verb"]!, 3, 1], + ] { + for value in row { append(value, to: &lexemes) } + } + var senses = Data() + for row: (UInt32, UInt32, UInt8) in [(0, 1, 3), (1, 1, 1), (2, 1, 2), (3, 0, 0)] { + append(row.0, to: &senses) + append(row.1, to: &senses) + senses.append(row.2) + senses.append(contentsOf: [0, 0, 0]) + } + var glosses = Data() + for text in ["house", "lake", "sea"] { + append(identifiers["en"]!, to: &glosses) + append(identifiers[text]!, to: &glosses) + append(identifiers[text]!, to: &glosses) + } + var lookups = Data() + for row: (UInt32, UInt32, UInt32) in [ + (identifiers["Haus"]!, 0, 1), + (identifiers["Häuser"]!, 1, 1), + (identifiers["See"]!, 2, 1), + (identifiers["laufen"]!, 3, 1), + ] { + append(row.0, to: &lookups) + append(row.1, to: &lookups) + append(row.2, to: &lookups) + } + var postings = Data() + for value: UInt64 in [1 << 32, (2 << 32), 1 | (1 << 32), 2 | (1 << 32)] { + append(value, to: &postings) + } + let sections = [stringRecords, stringBytes, lexemes, senses, glosses, lookups, postings] + var artifact = Data(repeating: 0, count: 256) + var sectionOffsets: [(UInt64, UInt64)] = [] + for section in sections { + sectionOffsets.append((UInt64(artifact.count), UInt64(section.count))) + artifact.append(section) + } + artifact.replaceSubrange(0..<8, with: Data("TFLEX001".utf8)) + var header = Data() + for value: UInt32 in [1, 1, 1, 1] { append(value, to: &header) } + for value: UInt64 in [3, 4, 3, 4, 4, UInt64(strings.count)] { append(value, to: &header) } + for section in sectionOffsets { + append(section.0, to: &header) + append(section.1, to: &header) + } + header.append(Data(repeating: 0, count: 64)) + artifact.replaceSubrange(8..<(8 + header.count), with: header) + return artifact + } + + private static func append(_ value: T, to data: inout Data) { + var littleEndian = value.littleEndian + withUnsafeBytes(of: &littleEndian) { data.append(contentsOf: $0) } + } +} diff --git a/VERSION b/VERSION index 0ea3a94..0d91a54 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -0.2.0 +0.3.0