From 8c14be475399f90f9b591d622659af9b896acce9 Mon Sep 17 00:00:00 2001 From: Alex-Wengg Date: Thu, 1 Oct 2026 21:55:30 -0400 Subject: [PATCH 1/2] Short-reply model: Swift host, hotkey demo, parity check ShortReplyManager (macOS 15) runs the FluidInference/short-reply-0.6b-coreml package: Qwen3 chat prompt via QwenBPETokenizer (which gains decode), left-padded to the fixed prefill length, prefill K/V written into the decoder MLState, greedy decode with an echo guard, seeded sampling for regenerate. ShortReplyDemo is a menu-bar app: with a reply box focused, 9 drafts and pastes the reply, 0 regenerates; the post above the box is read through Accessibility (Chrome and Safari); demo.sh --x opens x.com plus a macmon + log terminal; mock-feed/ is a local fictional feed for recordings. ShortReplyCheck replays benchmark posts for parity and latency. README gains a Short replies section. --- .gitignore | 1 + Package.swift | 3 +- README.md | 40 +- Sources/FluidUse/Qwen/QwenBPETokenizer.swift | 22 ++ .../ShortReply/ShortReplyManager.swift | 357 ++++++++++++++++++ Sources/ShortReplyCheck/main.swift | 84 +++++ Sources/ShortReplyDemo/README.md | 48 +++ Sources/ShortReplyDemo/ReplyPanel.swift | 110 ++++++ Sources/ShortReplyDemo/ReplyPanelModel.swift | 130 +++++++ Sources/ShortReplyDemo/SelectionReader.swift | 205 ++++++++++ Sources/ShortReplyDemo/demo.sh | 44 +++ Sources/ShortReplyDemo/main.swift | 122 ++++++ Sources/ShortReplyDemo/mock-feed/index.html | 166 ++++++++ Tests/FluidUseTests/ShortReplyTests.swift | 49 +++ 14 files changed, 1352 insertions(+), 29 deletions(-) create mode 100644 Sources/FluidUse/ShortReply/ShortReplyManager.swift create mode 100644 Sources/ShortReplyCheck/main.swift create mode 100644 Sources/ShortReplyDemo/README.md create mode 100644 Sources/ShortReplyDemo/ReplyPanel.swift create mode 100644 Sources/ShortReplyDemo/ReplyPanelModel.swift create mode 100644 Sources/ShortReplyDemo/SelectionReader.swift create mode 100755 Sources/ShortReplyDemo/demo.sh create mode 100644 Sources/ShortReplyDemo/main.swift create mode 100644 Sources/ShortReplyDemo/mock-feed/index.html create mode 100644 Tests/FluidUseTests/ShortReplyTests.swift diff --git a/.gitignore b/.gitignore index e72abde..fa9b1b5 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,7 @@ DerivedData/ *.xcodeproj .DS_Store .venv/ +.mobius/ Tools/doom/sauerkraut/models/ __pycache__/ _vizdoom.ini diff --git a/Package.swift b/Package.swift index ee32956..8866260 100644 --- a/Package.swift +++ b/Package.swift @@ -54,10 +54,11 @@ let package = Package( .executableTarget(name: "ImageSortCheck", dependencies: ["ImageSort", "FluidUse"]), .executableTarget(name: "ImageSortDemo", dependencies: ["ImageSort"], exclude: ["README.md"]), .executableTarget(name: "KevCheck", dependencies: ["FluidUse"]), - .executableTarget(name: "InternDecisionCheck", dependencies: ["FluidUse"]), .executableTarget(name: "GLiClassServe", dependencies: ["FluidUse"]), .executableTarget( name: "KevGuessWhoDemo", dependencies: ["FluidUse", "SortAnything"], exclude: ["README.md", "demo.sh"]), + .executableTarget(name: "ShortReplyCheck", dependencies: ["FluidUse"]), + .executableTarget(name: "ShortReplyDemo", dependencies: ["FluidUse"], exclude: ["README.md", "demo.sh", "mock-feed"]), .testTarget( name: "FluidUseTests", dependencies: ["FluidUse", "LayaTetris"], resources: [.copy("Fixtures")] diff --git a/README.md b/README.md index b72dae4..0c99dc3 100644 --- a/README.md +++ b/README.md @@ -153,39 +153,23 @@ On an M5 Pro a short ticket with two questions takes 18 ms and a Wikipedia bio w `swift run -c release KevGuessWhoDemo` plays Guess Who over 80 Wikipedia people with it ([Sources/KevGuessWhoDemo](Sources/KevGuessWhoDemo/README.md)). -## Intern-Decision +## Short replies -`InternDecisionManager` runs [Intern-Decision-0.8B](https://huggingface.co/internlm/Intern-Decision-0.8B) (Shanghai AI -Laboratory, Qwen3.5 backbone, Apache-2.0) on the GPU through Core ML. A request is a JSON state and up to 16 named -questions (multiple choice, yes/no, or a score); every question is answered in one call from the logits before its -`` marker, with the checkpoint's calibration temperature. The prompt is byte for byte the checkpoint's own -compiler and chat template, so answers match its reference engine (0 of 60 differ on its published suites, max -probability delta 0.004). The pinned snapshot downloads from -[FluidInference/intern-decision-0.8b-coreml](https://huggingface.co/FluidInference/intern-decision-0.8b-coreml) on first -use (~3.4 GB, macOS 14 / iOS 17). Text only; images are not exported. +`ShortReplyManager` drafts one short reply to a social post with a sub-1B model: +[short-reply-0.6b-coreml](https://huggingface.co/FluidInference/short-reply-0.6b-coreml), a Qwen3-0.6B fine-tune +(Apache-2.0) in one 724 MB package whose prefill runs on the Neural Engine and whose decode runs on the GPU. About +190 ms per reply on an M5 Pro; drafts are for a person to review, nothing is posted automatically. ```swift -let model = try await InternDecisionManager.load(from: try await InternDecisionModelStore.ensure()) -let result = try await model.decide( - state: ["channel": "email", "message": "Charged twice for my annual renewal. Refund the duplicate before Friday."], - questions: [ - ("team", .choice("Which team should handle this?", options: [("billing", "Refunds."), ("technical", "Bugs.")])), - ("frustrated", .noul("Is the customer frustrated?")), - ("urgency", .score("How urgent is this?", levels: ["Low", "Medium", "High"])), - ]) -print(result.answers.map { "\($0.field): \($0.decision) \($0.confidence)" }) +let replies = try await ShortReplyManager.load(from: modelDirectory) // config.json, tokenizer.json, .mlpackage +let draft = try await replies.draft(for: "Finally passed my driving test on the third try.") +print(draft.reply, draft.timing.totalSeconds) // "Congrats on passing!" 0.19 ``` -On an M5 Pro that request (319 tokens, three fields) takes 61 ms in the 320-token bucket; the checkpoint's own PyTorch -path on the same Mac takes 150 ms (bf16). Buckets are 320, 512 and 1,024 tokens and the pass costs the bucket, not the -request. `swift run -c release InternDecisionCheck bench ` reproduces the number. - -A second snapshot, `InternDecisionModelStore.ensure(.showdown)`, is the same model fine-tuned to pick Pokémon Showdown -battle actions ([FluidInference/intern-decision-0.8b-showdown-coreml](https://huggingface.co/FluidInference/intern-decision-0.8b-showdown-coreml)), -distilled from Intern-Decision-4B on a MacBook. It answers the same typed questions; given a battle state and the legal -moves and switches as `choice` options it matches its 4B teacher (24-6 vs poke-env's max-power player, 9-21 vs its -heuristic player over 30 battles). Buckets are 512, 640 and 1,024 tokens with int8 weights: about 0.85 GB in memory -with one bucket in use, 1.9 GB to download, the same 90 ms per decision as fp16. +`ShortReplyDemo` is a menu-bar app: open a post's reply box in any app (X in Chrome or Safari, Slack, Mail), press +**9**, and the draft is pasted into the box; **0** regenerates it. +`Sources/ShortReplyDemo/demo.sh --x` launches it with a macmon + log terminal +([Sources/ShortReplyDemo](Sources/ShortReplyDemo/README.md)). ## Demo diff --git a/Sources/FluidUse/Qwen/QwenBPETokenizer.swift b/Sources/FluidUse/Qwen/QwenBPETokenizer.swift index 7026c6b..e6eb45a 100644 --- a/Sources/FluidUse/Qwen/QwenBPETokenizer.swift +++ b/Sources/FluidUse/Qwen/QwenBPETokenizer.swift @@ -10,6 +10,9 @@ public final class QwenBPETokenizer: Sendable { /// Added tokens, longest first, so a longer token wins over one it contains. private let addedTokens: [(content: String, id: Int)] private let byteToCharacter: [Character] + /// Reverse tables for `decode`. + private let tokenForID: [Int: String] + private let byteForCharacter: [Character: UInt8] /// Encoded pieces; ordinary text repeats the same words, so this saves most of the merge loops. private let cache = OSAllocatedUnfairLock<[String: [Int]]>(initialState: [:]) @@ -44,6 +47,12 @@ public final class QwenBPETokenizer: Sendable { return (content, id) }.sorted { $0.content.count > $1.content.count } byteToCharacter = Self.bytesToUnicode() + var tokenForID = [Int: String](minimumCapacity: vocab.count) + for (token, id) in vocab { tokenForID[id] = token } + self.tokenForID = tokenForID + var byteForCharacter = [Character: UInt8](minimumCapacity: 256) + for (byte, character) in byteToCharacter.enumerated() { byteForCharacter[character] = UInt8(byte) } + self.byteForCharacter = byteForCharacter _ = try Self.splitter() } @@ -72,6 +81,19 @@ public final class QwenBPETokenizer: Sendable { return ids } + /// Text for `ids`, dropping added (special) tokens; invalid byte sequences decode lossily. + public func decode(_ ids: [Int]) -> String { + let special = Set(addedTokens.map(\.id)) + var bytes: [UInt8] = [] + for id in ids where !special.contains(id) { + guard let token = tokenForID[id] else { continue } + for character in token { + if let byte = byteForCharacter[character] { bytes.append(byte) } + } + } + return String(decoding: bytes, as: UTF8.self) + } + public func id(for token: String) -> Int? { addedTokens.first { $0.content == token }?.id ?? vocabulary[token] } diff --git a/Sources/FluidUse/ShortReply/ShortReplyManager.swift b/Sources/FluidUse/ShortReply/ShortReplyManager.swift new file mode 100644 index 0000000..bf67791 --- /dev/null +++ b/Sources/FluidUse/ShortReply/ShortReplyManager.swift @@ -0,0 +1,357 @@ +import CoreML +import Foundation +import os + +/// Drafts one short reply to a social post with the FluidUse short-reply model (Qwen3-0.6B fine-tune) on Core ML. +/// +/// The package holds two functions over one set of weights: `prefill` (stateless, fixed prompt length, runs on the +/// Neural Engine) and `decode` (stateful KV cache, runs on the GPU). The host tokenizes the chat prompt, left-pads it to +/// the prefill length, copies the prefill K/V into the decoder's state, then decodes greedily until `<|im_end|>`. +@available(macOS 15.0, iOS 18.0, *) +public actor ShortReplyManager { + public struct Timing: Sendable { + public let prefillSeconds: Double + public let decodeSeconds: Double + public let generatedTokens: Int + public var totalSeconds: Double { prefillSeconds + decodeSeconds } + } + + public struct Draft: Sendable { + public let reply: String + public let timing: Timing + /// The post actually sent to the model (links removed; cut to the prefill length when too long). + public let post: String + public let trimmed: Bool + } + + /// `config.json` next to the package. + struct Config: Decodable { + let package: String + let prefillLength: Int + let cacheLength: Int + let decodeLengths: [Int] + let layers: Int + let kvHeads: Int + let headDim: Int + let padID: Int + let stopIDs: [Int] + let maxNewTokens: Int + let systemPrompt: String + let maxPostCharacters: Int + } + + private let logger = Logger(subsystem: "FluidUse", category: "ShortReply") + private let config: Config + private let tokenizer: QwenBPETokenizer + private let prefill: MLModel + private let decode: MLModel + private let stopIDs: Set + + /// Loads `/config.json`, the package it names, and `tokenizer.json`. + public static func load( + from directory: URL, prefillUnits: MLComputeUnits = .cpuAndNeuralEngine, + decodeUnits: MLComputeUnits = .cpuAndGPU + ) async throws -> ShortReplyManager { + let config = try JSONDecoder().decode( + Config.self, from: Data(contentsOf: directory.appendingPathComponent("config.json"))) + var package = directory.appendingPathComponent(config.package) + if package.pathExtension == "mlpackage" { + package = try await MLModel.compileModel(at: package) + } + return try ShortReplyManager( + config: config, compiled: package, tokenizer: directory.appendingPathComponent("tokenizer.json"), + prefillUnits: prefillUnits, decodeUnits: decodeUnits) + } + + /// Models are created here, inside the actor, so the non-Sendable `MLModel`s never cross an isolation boundary. + init( + config: Config, compiled: URL, tokenizer: URL, prefillUnits: MLComputeUnits, decodeUnits: MLComputeUnits + ) throws { + self.config = config + self.tokenizer = try QwenBPETokenizer(tokenizerJsonURL: tokenizer) + let prefillConfiguration = MLModelConfiguration() + prefillConfiguration.computeUnits = prefillUnits + prefillConfiguration.functionName = "prefill" + let decodeConfiguration = MLModelConfiguration() + decodeConfiguration.computeUnits = decodeUnits + decodeConfiguration.functionName = "decode" + self.prefill = try MLModel(contentsOf: compiled, configuration: prefillConfiguration) + self.decode = try MLModel(contentsOf: compiled, configuration: decodeConfiguration) + self.stopIDs = Set(config.stopIDs) + } + + /// One prompt through both functions, so the first real call pays no compile cost. + public func warmUp() async throws { + _ = try await draft(for: "Warm-up post.") + } + + /// The prompt exactly as `reply.py` builds it (Qwen3 chat template, thinking disabled). + func promptTokens(for post: String) throws -> [Int] { + let text = + "<|im_start|>system\n\(config.systemPrompt)<|im_end|>\n" + + "<|im_start|>user\nPost: \(post)\nReply:<|im_end|>\n" + + "<|im_start|>assistant\n\n\n\n\n" + return try tokenizer.encode(text) + } + + /// Whitespace-normalized, links dropped, and cut (by tokens, from the end) so the prompt fits the prefill length. + func preparePost(_ rawPost: String) throws -> (post: String, trimmed: Bool) { + var post = Self.stripLinks(rawPost).split(whereSeparator: \.isWhitespace).joined(separator: " ") + guard !post.isEmpty else { throw ShortReplyError.emptyPost } + if post.count > config.maxPostCharacters { post = String(post.prefix(config.maxPostCharacters)) } + let overhead = try promptTokens(for: "").count + let budget = config.prefillLength - overhead + guard budget > 8 else { throw ShortReplyError.invalidAsset("prefill length too short for the system prompt") } + let postTokens = try tokenizer.encode(post) + guard postTokens.count > budget else { return (post, false) } + let cut = tokenizer.decode(Array(postTokens.prefix(budget - 1))).trimmingCharacters(in: .whitespaces) + return (cut + "…", true) + } + + static func stripLinks(_ text: String) -> String { + text.replacingOccurrences( + of: #"(https?://\S+|\b(?:pic\.x\.com|pic\.twitter\.com|t\.co|x\.com)/\S+)"#, with: "", + options: .regularExpression) + } + + /// `variation` 0 decodes greedily (the benchmarked behavior); higher values sample (temperature 0.7, top-p 0.9) + /// with a seed derived from the value, so "regenerate" gives a different, repeatable reply. `avoiding` rejects + /// replies equal to earlier drafts (up to a few attempts). + public func draft( + for rawPost: String, variation: Int = 0, avoiding previous: Set = [] + ) async throws -> Draft { + for attempt in 0..<(variation == 0 ? 1 : 4) { + let draft = try await decodeDraft(for: rawPost, variation: variation == 0 ? 0 : variation + attempt * 1000) + if variation == 0 || !previous.contains(draft.reply) { return draft } + } + return try await decodeDraft(for: rawPost, variation: variation) + } + + private func decodeDraft(for rawPost: String, variation: Int) async throws -> Draft { + let (post, trimmed) = try preparePost(rawPost) + var rng = SeededGenerator(seed: UInt64(20_260_929 &+ variation)) + let tokens = try promptTokens(for: post) + let length = config.prefillLength + guard tokens.count <= length else { throw ShortReplyError.postTooLong(config.maxPostCharacters) } + // Left-pad to the fixed prefill length; padded cache slots stay masked for every later query. + let pad = length - tokens.count + let ids = Array(repeating: config.padID, count: pad) + tokens + + let started = ContinuousClock.now + let state = decode.makeState() + let prefillOutput = try Self.predict(prefill, try prefillInputs(ids: ids, pad: pad), state: nil) + try copyCache(from: prefillOutput, into: state, promptLength: length) + var logits = try lastLogits(prefillOutput) + let prefillSeconds = seconds(since: started) + + let decodeStarted = ContinuousClock.now + var generated: [Int] = [] + var position = length + while generated.count < config.maxNewTokens { + let token = variation == 0 ? argmax(logits) : sample(logits, temperature: 0.7, topP: 0.9, using: &rng) + if stopIDs.contains(token) { break } + generated.append(token) + let output = try Self.predict( + decode, try decodeInputs(token: token, position: position, pad: pad), state: state) + logits = try lastLogits(output) + position += 1 + } + let decodeSeconds = seconds(since: decodeStarted) + let reply = ShortReplyManager.cleanReply(tokenizer.decode(generated)) + let timing = Timing( + prefillSeconds: prefillSeconds, decodeSeconds: decodeSeconds, generatedTokens: generated.count) + logger.info( + "reply in \(Int(timing.totalSeconds * 1000)) ms (prefill \(Int(prefillSeconds * 1000)), \(generated.count) tokens)" + ) + return Draft(reply: reply, timing: timing, post: post, trimmed: trimmed) + } + + /// Synchronous prediction (an async context would pick Core ML's async overload). Called only from the actor, + /// which serializes it: Core ML's synchronous prediction is not thread-safe across callers. + private static func predict( + _ model: MLModel, _ input: MLFeatureProvider, state: MLState? + ) throws -> MLFeatureProvider { + if let state { return try model.prediction(from: input, using: state) } + return try model.prediction(from: input) + } + + // MARK: - Feature providers + + private func prefillInputs(ids: [Int], pad: Int) throws -> MLDictionaryFeatureProvider { + let length = ids.count + let inputIDs = try MLMultiArray(shape: [1, NSNumber(value: length)], dataType: .int32) + for (index, id) in ids.enumerated() { inputIDs[index] = NSNumber(value: Int32(id)) } + // Causal mask over the prompt; padded columns masked for every row. + let mask = try MLMultiArray( + shape: [1, 1, NSNumber(value: length), NSNumber(value: length)], dataType: .float32) + mask.withUnsafeMutableBytes { buffer, _ in + let values = buffer.bindMemory(to: Float32.self) + for row in 0.. row || column < pad) ? -1e4 : 0 + } + } + } + return try MLDictionaryFeatureProvider(dictionary: ["input_ids": inputIDs, "attention_mask": mask]) + } + + private func decodeInputs(token: Int, position: Int, pad: Int) throws -> MLDictionaryFeatureProvider { + let cache = config.cacheLength + guard position < cache else { throw ShortReplyError.postTooLong(config.maxPostCharacters) } + let inputIDs = try MLMultiArray(shape: [1, 1], dataType: .int32) + inputIDs[0] = NSNumber(value: Int32(token)) + let mask = try MLMultiArray(shape: [1, 1, 1, NSNumber(value: cache)], dataType: .float32) + mask.withUnsafeMutableBytes { buffer, _ in + let values = buffer.bindMemory(to: Float32.self) + for column in 0.. position) ? -1e4 : 0 } + } + let cachePosition = try MLMultiArray(shape: [1], dataType: .int32) + cachePosition[0] = NSNumber(value: Int32(position)) + return try MLDictionaryFeatureProvider(dictionary: [ + "input_ids": inputIDs, "attention_mask": mask, "cache_position": cachePosition, + ]) + } + + /// Prefill returns `keys` / `values` as [layers, kvHeads, prompt, headDim] fp16; the decoder state holds + /// [1, kvHeads, cache, headDim] fp16 per layer. + private func copyCache(from output: MLFeatureProvider, into state: MLState, promptLength: Int) throws { + guard let keys = output.featureValue(for: "keys")?.multiArrayValue, + let values = output.featureValue(for: "values")?.multiArrayValue + else { throw ShortReplyError.invalidAsset("prefill output has no keys/values") } + let headDim = config.headDim + let kvHeads = config.kvHeads + let layerElements = kvHeads * promptLength * headDim + for layer in 0...size) + } + } + } + } + } + } + } + + private func lastLogits(_ output: MLFeatureProvider) throws -> MLMultiArray { + guard let logits = output.featureValue(for: "logits")?.multiArrayValue else { + throw ShortReplyError.invalidAsset("model output has no logits") + } + return logits + } + + private func argmax(_ logits: MLMultiArray) -> Int { + let count = logits.count + return logits.withUnsafeBytes { buffer in + switch logits.dataType { + case .float32: + let values = buffer.bindMemory(to: Float32.self) + var best = 0 + for index in 1.. values[best] { best = index } + return best + default: + let values = buffer.bindMemory(to: Float16.self) + var best = 0 + for index in 1.. values[best] { best = index } + return best + } + } + } + + /// Temperature + nucleus sampling over the vocabulary. + private func sample( + _ logits: MLMultiArray, temperature: Float, topP: Float, using rng: inout SeededGenerator + ) -> Int { + let count = logits.count + var scores = [Float](repeating: 0, count: count) + logits.withUnsafeBytes { buffer in + switch logits.dataType { + case .float32: + let values = buffer.bindMemory(to: Float32.self) + for index in 0.. probabilities[$1] }.prefix(256) + var kept: [(Int, Float)] = [] + var mass: Float = 0 + for index in candidates { + kept.append((index, probabilities[index])) + mass += probabilities[index] + if mass >= topP { break } + } + var draw = Float.random(in: 0.. Double { + let duration = start.duration(to: .now) + return Double(duration.components.seconds) + Double(duration.components.attoseconds) / 1e18 + } + + /// First non-empty line, stripped of a leading "Reply:" / bullet and surrounding quotes (mirrors `reply.py`). + static func cleanReply(_ raw: String) -> String { + guard + var line = raw.split(whereSeparator: \.isNewline).map({ $0.trimmingCharacters(in: .whitespaces) }) + .first(where: { !$0.isEmpty }) + else { return "" } + if let range = line.range(of: #"^(reply\s*:\s*|[-•]\s*)"#, options: [.regularExpression, .caseInsensitive]) { + line.removeSubrange(range) + } + return line.trimmingCharacters(in: CharacterSet(charactersIn: " \"'“”")) + } +} + +/// Small deterministic generator (SplitMix64) so a given variation always yields the same sampled reply. +struct SeededGenerator: RandomNumberGenerator { + private var state: UInt64 + + init(seed: UInt64) { state = seed } + + mutating func next() -> UInt64 { + state &+= 0x9E37_79B9_7F4A_7C15 + var z = state + z = (z ^ (z >> 30)) &* 0xBF58_476D_1CE4_E5B9 + z = (z ^ (z >> 27)) &* 0x94D0_49BB_1331_11EB + return z ^ (z >> 31) + } +} + +public enum ShortReplyError: Error, LocalizedError, Sendable { + case emptyPost + case postTooLong(Int) + case invalidAsset(String) + + public var errorDescription: String? { + switch self { + case .emptyPost: return "Select a post first." + case .postTooLong(let limit): return "The post is too long (\(limit) character limit)." + case .invalidAsset(let detail): return "Short-reply model asset problem: \(detail)" + } + } +} diff --git a/Sources/ShortReplyCheck/main.swift b/Sources/ShortReplyCheck/main.swift new file mode 100644 index 0000000..211ede3 --- /dev/null +++ b/Sources/ShortReplyCheck/main.swift @@ -0,0 +1,84 @@ +import FluidUse +import Foundation + +// Parity + latency check for the Swift host: replays a responses.jsonl (post, expected reply) through +// ShortReplyManager and reports exact matches and per-reply timing. +// swift run -c release ShortReplyCheck [column] [limit] + +let arguments = CommandLine.arguments +guard arguments.count >= 3 else { + FileHandle.standardError.write(Data("usage: ShortReplyCheck [column] [limit]\n".utf8)) + exit(2) +} +let directory = URL(fileURLWithPath: arguments[1]) +let column = arguments.count > 3 ? arguments[3] : "tuned" +let limit = arguments.count > 4 ? Int(arguments[4]) ?? .max : .max + +struct Row: Decodable { + let id: String + let post: String + let replies: [String: String] + + init(from decoder: Decoder) throws { + let container = try decoder.container(keyedBy: AnyKey.self) + id = try container.decode(String.self, forKey: AnyKey("id")) + post = try container.decode(String.self, forKey: AnyKey("post")) + var replies: [String: String] = [:] + for key in container.allKeys where key.stringValue != "id" && key.stringValue != "post" { + replies[key.stringValue] = try container.decode(String.self, forKey: key) + } + self.replies = replies + } +} + +struct AnyKey: CodingKey { + var stringValue: String + var intValue: Int? { nil } + init(_ string: String) { stringValue = string } + init?(stringValue: String) { self.stringValue = stringValue } + init?(intValue: Int) { nil } +} + +let rows = try String(contentsOf: URL(fileURLWithPath: arguments[2]), encoding: .utf8) + .split(whereSeparator: \.isNewline).prefix(limit) + .map { try JSONDecoder().decode(Row.self, from: Data($0.utf8)) } + +@available(macOS 15.0, *) +func run() async throws { + let manager = try await ShortReplyManager.load(from: directory) + try await manager.warmUp() + var matches = 0 + var totals: [Double] = [] + var prefills: [Double] = [] + var perToken: [Double] = [] + for (index, row) in rows.enumerated() { + let draft = try await manager.draft(for: row.post) + let expected = row.replies[column] ?? "" + let same = draft.reply == expected + matches += same ? 1 : 0 + totals.append(draft.timing.totalSeconds) + prefills.append(draft.timing.prefillSeconds) + if draft.timing.generatedTokens > 0 { + perToken.append(draft.timing.decodeSeconds / Double(draft.timing.generatedTokens)) + } + if !same { + print(" differs #\(index + 1): swift=\(draft.reply.debugDescription) python=\(expected.debugDescription)") + } + } + func median(_ values: [Double]) -> Double { + let sorted = values.sorted() + return sorted[sorted.count / 2] + } + print("exact reply match: \(matches)/\(rows.count)") + print( + String( + format: "prefill p50: %.1f ms decode p50: %.1f ms/token reply p50: %.0f ms", + median(prefills) * 1000, median(perToken) * 1000, median(totals) * 1000)) +} + +if #available(macOS 15.0, *) { + try await run() +} else { + FileHandle.standardError.write(Data("ShortReplyCheck needs macOS 15\n".utf8)) + exit(2) +} diff --git a/Sources/ShortReplyDemo/README.md b/Sources/ShortReplyDemo/README.md new file mode 100644 index 0000000..7ca87f9 --- /dev/null +++ b/Sources/ShortReplyDemo/README.md @@ -0,0 +1,48 @@ +# Short Reply demo + +Open a post's reply box in any app (X or Reddit in Chrome or Safari, Slack, Mail) and press **9**: the reply is drafted +on device and pasted into the box, no UI. Press **0** (or 9 again on the same post) to regenerate: the first draft is the +model's greedy reply (the benchmarked one); later ones are sampled (temperature 0.7, top-p 0.9, seeded), never +repeating an earlier draft, and they replace the box's contents. The panel has an **Again** button for the same. With text selected instead of a box focused, a floating panel shows the +draft with Copy / Insert (the menu-bar item can turn auto-insert off so the panel always shows). The model is the FluidUse short-reply Qwen3-0.6B fine-tune as one 724 MB Core ML package: the +prompt runs on the Neural Engine, the reply tokens on the GPU. The panel shows the timing split; **Copy** puts the +draft on the clipboard, **Insert** pastes it back into the app the post came from. + +```bash +chmod +x Sources/ShortReplyDemo/demo.sh +Sources/ShortReplyDemo/demo.sh --x # app + terminal (macmon on top, model log below) + x.com in Chrome +Sources/ShortReplyDemo/demo.sh --mock # same, with the local mock feed instead of x.com +``` + +The model directory holds `short_reply_0_6b.mlpackage` (functions `prefill` and `decode`), `tokenizer.json` and +`config.json`. Get it from Hugging Face, then point the launcher at it: + +```bash +hf download FluidInference/short-reply-0.6b-coreml --local-dir ~/Models/short-reply-0.6b-coreml +Sources/ShortReplyDemo/demo.sh --x ~/Models/short-reply-0.6b-coreml # or set SHORT_REPLY_MODEL_DIR +``` First launch compiles +the package (a few seconds) and runs one warm-up prompt; the log prints `model ready`. + +**Mock feed for recording.** `mock-feed/index.html` is a local feed of fictional posts with a reply box under each; +nothing on it is posted anywhere. Select a post, press 9, then **Insert**: the page routes the paste into that post's reply box, and "Reply" only +appends the text on the page. On the real site, click into the post's reply box before pressing Insert. + +Browsers: Safari exposes the selection through Accessibility directly; for Chrome the app switches on Chrome's +accessibility tree (`AXEnhancedUserInterface`) and otherwise falls back to a ⌘C round trip that restores your clipboard. + +The terminal uses macmon (`brew install macmon`, no sudo) for the GPU / ANE / power rows; asitop is the fallback but +its 0.0.24 release crashes on M5 chips. Permissions: the app reads the selection through Accessibility and listens for the global hotkey, so macOS asks for +**Accessibility** access on first use (System Settings › Privacy & Security). Without it, use the menu-bar item +"Draft reply from clipboard" after copying a post. + +What to expect: replies are short drafts for a person to review, not auto-posted. On the project's 100-post benchmark +about 1 in 8 drafts is generic or slightly off; the demo does not hide those. + +Measured on an M5 Pro (macOS 27): 89 of 100 benchmark replies identical to the PyTorch checkpoint; prefill 28 ms on the +Neural Engine, decode 24 ms per token on the GPU, about 190 ms per reply. + +Parity and latency of the Swift host against the Python export check: + +```bash +swift run -c release ShortReplyCheck tuned 100 +``` diff --git a/Sources/ShortReplyDemo/ReplyPanel.swift b/Sources/ShortReplyDemo/ReplyPanel.swift new file mode 100644 index 0000000..c9a2ab7 --- /dev/null +++ b/Sources/ShortReplyDemo/ReplyPanel.swift @@ -0,0 +1,110 @@ +import AppKit +import SwiftUI + +/// Floating, non-activating panel that shows the selected post, the draft, timing, and Copy / Insert. +@available(macOS 15.0, *) +@MainActor +final class ReplyPanel { + private let window: NSPanel + private let model: ReplyPanelModel + + init(model: ReplyPanelModel) { + self.model = model + window = NSPanel( + contentRect: NSRect(x: 0, y: 0, width: 440, height: 250), + styleMask: [.titled, .closable, .nonactivatingPanel, .utilityWindow, .fullSizeContentView], + backing: .buffered, defer: false) + window.title = "Short Reply" + window.titleVisibility = .hidden + window.titlebarAppearsTransparent = true + window.isFloatingPanel = true + window.level = .floating + window.hidesOnDeactivate = false + window.isReleasedWhenClosed = false + window.collectionBehavior = [.canJoinAllSpaces, .fullScreenAuxiliary] + window.contentView = NSHostingView( + rootView: ReplyPanelView(model: model, onClose: { [weak self] in self?.window.orderOut(nil) })) + } + + func hide() { window.orderOut(nil) } + + /// Shows the panel near the mouse without taking focus from the app the post was selected in. + func present() { + let mouse = NSEvent.mouseLocation + var origin = NSPoint(x: mouse.x + 16, y: mouse.y - window.frame.height - 16) + if let screen = NSScreen.screens.first(where: { $0.frame.contains(mouse) }) ?? NSScreen.main { + origin.x = min( + max(origin.x, screen.visibleFrame.minX + 8), screen.visibleFrame.maxX - window.frame.width - 8) + origin.y = min( + max(origin.y, screen.visibleFrame.minY + 8), screen.visibleFrame.maxY - window.frame.height - 8) + } + window.setFrameOrigin(origin) + window.orderFrontRegardless() + } +} + +@available(macOS 15.0, *) +struct ReplyPanelView: View { + @ObservedObject var model: ReplyPanelModel + var onClose: () -> Void + + var body: some View { + VStack(alignment: .leading, spacing: 10) { + HStack { + Text("Short Reply").font(.headline) + Spacer() + statusLabel + Button(action: onClose) { Image(systemName: "xmark.circle.fill") } + .buttonStyle(.plain).foregroundStyle(.secondary) + } + if !model.post.isEmpty { + Text(model.post) + .font(.callout).foregroundStyle(.secondary).lineLimit(3) + .frame(maxWidth: .infinity, alignment: .leading) + } + Group { + switch model.phase { + case .loading: + Text("Loading model…").foregroundStyle(.secondary) + case .idle: + Text("Select a post and press 9.").foregroundStyle(.secondary) + case .drafting: + HStack(spacing: 8) { + ProgressView().controlSize(.small) + Text("Drafting…") + } + case .done: + Text(model.reply).font(.title3.weight(.medium)).textSelection(.enabled) + case .failed(let message): + Text(message).foregroundStyle(.red) + } + } + .frame(maxWidth: .infinity, minHeight: 44, alignment: .leading) + HStack { + if let timing = model.timing { + Text( + String( + format: "%d ms · prefill %d ms on ANE · %d tokens on GPU%@", + Int(timing.totalSeconds * 1000), Int(timing.prefillSeconds * 1000), timing.generatedTokens, + model.trimmed ? " · long post, start used" : "") + ) + .font(.caption.monospacedDigit()).foregroundStyle(.secondary) + } + Spacer() + Button("Again") { Task { await model.draft(post: model.post) } }.disabled(model.phase != .done) + Button("Copy") { model.copyReply() }.disabled(model.phase != .done) + Button("Insert") { model.insertReply() }.disabled( + model.phase != .done || model.sourceApplication == nil + ) + .keyboardShortcut(.defaultAction) + } + } + .padding(16) + .frame(width: 440) + } + + private var statusLabel: some View { + Text(model.phase == .loading ? "loading" : "Qwen3-0.6B · 724 MB · on-device") + .font(.caption).foregroundStyle(.tertiary) + } +} diff --git a/Sources/ShortReplyDemo/ReplyPanelModel.swift b/Sources/ShortReplyDemo/ReplyPanelModel.swift new file mode 100644 index 0000000..fb308c8 --- /dev/null +++ b/Sources/ShortReplyDemo/ReplyPanelModel.swift @@ -0,0 +1,130 @@ +import AppKit +import FluidUse +import Foundation + +/// State behind the floating panel: model loading, the current post, the draft and its timing. +@available(macOS 15.0, *) +@MainActor +final class ReplyPanelModel: ObservableObject { + enum Phase: Equatable { + case loading + case idle + case drafting + case done + case failed(String) + } + + @Published var phase: Phase = .loading + @Published var post = "" + @Published var reply = "" + @Published var timing: ShortReplyManager.Timing? + @Published var drafts = 0 + @Published var trimmed = false + /// Same post again → sample a new reply instead of repeating the greedy one. + private var lastPost = "" + private var variation = 0 + private var previousReplies: Set = [] + var sourceApplication: NSRunningApplication? + + private var manager: ShortReplyManager? + + static var modelDirectory: URL { + if let path = ProcessInfo.processInfo.environment["SHORT_REPLY_MODEL_DIR"], !path.isEmpty { + return URL(fileURLWithPath: path) + } + return URL(fileURLWithPath: ".mobius/short-reply-lm/coreml-demo") + } + + func load() async { + let started = ContinuousClock.now + do { + let manager = try await ShortReplyManager.load(from: Self.modelDirectory) + try await manager.warmUp() + self.manager = manager + let elapsed = started.duration(to: .now) + print( + "model ready in \(elapsed.components.seconds) s · prefill on ANE, decode on GPU · \(Self.modelDirectory.lastPathComponent)" + ) + phase = .idle + } catch { + print("model load failed: \(error.localizedDescription)") + phase = .failed(error.localizedDescription) + } + } + + func draft(post text: String) async { + guard let manager else { + show(error: "Model is still loading.") + return + } + if text == lastPost { + variation += 1 + } else { + lastPost = text + variation = 0 + previousReplies = [] + } + post = text + reply = "" + timing = nil + trimmed = false + phase = .drafting + do { + let draft = try await manager.draft(for: text, variation: variation, avoiding: previousReplies) + previousReplies.insert(draft.reply) + reply = draft.reply + timing = draft.timing + trimmed = draft.trimmed + post = draft.post + drafts += 1 + phase = .done + let ms = Int(draft.timing.totalSeconds * 1000) + let prefill = Int(draft.timing.prefillSeconds * 1000) + print( + "#\(drafts) \(ms) ms (prefill \(prefill) ms ANE · \(draft.timing.generatedTokens) tokens GPU) · post \(text.count) chars\(variation > 0 ? " · regenerate #\(variation)" : "")" + ) + print(" post: \(text.prefix(110).replacingOccurrences(of: "\n", with: " "))") + print(" reply: \(draft.reply)") + } catch { + show(error: error.localizedDescription) + } + } + + func show(error: String) { + phase = .failed(error) + print("error: \(error)") + } + + func copyReply() { + NSPasteboard.general.clearContents() + NSPasteboard.general.setString(reply, forType: .string) + } + + /// Copies the reply, hands focus back to the source app, and sends it ⌘V directly (delivered to its process, + /// so it lands even if macOS declines the activation request). `replace` selects the field first (⌘A), for a + /// regenerated reply going into a box that already holds the previous one. + func insertReply(replace: Bool = false) { + copyReply() + guard let application = sourceApplication else { return } + if NSApp.isActive { + NSApp.yieldActivation(to: application) + } + application.activate() + let pid = application.processIdentifier + let name = application.localizedName ?? "app" + Task { @MainActor in + try? await Task.sleep(for: .milliseconds(200)) + if replace { + // Purge the box first: select all, delete, and give the editor a moment between each step. + KeystrokeSender.send(keyCode: 0, command: true, to: pid) // ⌘A + try? await Task.sleep(for: .milliseconds(80)) + KeystrokeSender.send(keyCode: 51, command: false, to: pid) // Delete + try? await Task.sleep(for: .milliseconds(80)) + } + KeystrokeSender.send(keyCode: 9, command: true, to: pid) // ⌘V + print("insert: \(replace ? "replaced in" : "pasted into") \(name) (pid \(pid))") + } + } + + var isRegenerate: Bool { variation > 0 } +} diff --git a/Sources/ShortReplyDemo/SelectionReader.swift b/Sources/ShortReplyDemo/SelectionReader.swift new file mode 100644 index 0000000..99e67a8 --- /dev/null +++ b/Sources/ShortReplyDemo/SelectionReader.swift @@ -0,0 +1,205 @@ +import AppKit +import ApplicationServices +import Carbon.HIToolbox + +/// Reads the text selected in the frontmost app: Accessibility first, then a ⌘C round trip through the pasteboard. +enum SelectionReader { + static var isTrusted: Bool { AXIsProcessTrusted() } + + /// Text to reply to. Order: the selection; then, when a text field has focus (a reply box), the post shown + /// above it in the same dialog; then a ⌘C round trip. `typedTrigger` removes the trigger key's character if it + /// landed in that text field. + struct Context { + let text: String? + /// A text field (reply box) had focus, so a draft can be inserted without any UI. + let fieldFocused: Bool + } + + static func context(typedTrigger: Bool = false) async -> Context { + if let text = accessibilitySelection(), !text.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty { + return Context(text: text, fieldFocused: false) + } + if let focused = focusedElement(), isEditable(focused) { + if typedTrigger { + KeystrokeSender.send( + keyCode: UInt16(kVK_Delete), command: false, + to: NSWorkspace.shared.frontmostApplication?.processIdentifier) + } + return Context(text: postAbove(focused), fieldFocused: true) + } + return Context(text: await pasteboardSelection(), fieldFocused: false) + } + + /// Removes the character a typed trigger key left in a focused text field, if any. + static func deleteTypedTriggerIfEditing() { + guard let focused = focusedElement(), isEditable(focused) else { return } + KeystrokeSender.send( + keyCode: UInt16(kVK_Delete), command: false, to: NSWorkspace.shared.frontmostApplication?.processIdentifier) + } + + private static func focusedElement() -> AXUIElement? { + guard let application = NSWorkspace.shared.frontmostApplication else { return nil } + let element = AXUIElementCreateApplication(application.processIdentifier) + AXUIElementSetAttributeValue(element, "AXEnhancedUserInterface" as CFString, kCFBooleanTrue) + var focused: CFTypeRef? + guard AXUIElementCopyAttributeValue(element, kAXFocusedUIElementAttribute as CFString, &focused) == .success, + let focused, CFGetTypeID(focused) == AXUIElementGetTypeID() + else { return nil } + return unsafeBitCast(focused, to: AXUIElement.self) + } + + private static func attribute(_ element: AXUIElement, _ name: String) -> CFTypeRef? { + var value: CFTypeRef? + return AXUIElementCopyAttributeValue(element, name as CFString, &value) == .success ? value : nil + } + + private static func isEditable(_ element: AXUIElement) -> Bool { + let role = attribute(element, kAXRoleAttribute) as? String ?? "" + if role == kAXTextAreaRole || role == kAXTextFieldRole || role == kAXComboBoxRole { return true } + var settable = DarwinBoolean(false) + AXUIElementIsAttributeSettable(element, kAXValueAttribute as CFString, &settable) + return settable.boolValue && attribute(element, kAXValueAttribute) is String + } + + /// Static text above the focused field within the nearest enclosing container that holds a real paragraph: + /// on a reply dialog that is the post being answered. Short fragments (names, times, "Replying to …") are dropped. + private static func postAbove(_ focused: AXUIElement) -> String? { + var node = focused + for _ in 0..<48 { // X nests the reply box dozens of AXGroups deep + guard let parentValue = attribute(node, kAXParentAttribute), + CFGetTypeID(parentValue) == AXUIElementGetTypeID() + else { return nil } + let parent = unsafeBitCast(parentValue, to: AXUIElement.self) + var paragraphs: [String] = [] + var budget = 2500 + var reachedFocus = false + collectText(parent, focused: focused, into: ¶graphs, budget: &budget, reachedFocus: &reachedFocus) + let kept = Self.postParagraphs(paragraphs) + if !kept.isEmpty { return kept.joined(separator: " ") } + node = parent + } + return nil + } + + /// The post body from a container's static texts in order: drops the header trio (display name, `@handle`, + /// time), the "Replying to" line and UI labels, and keeps the rest so a post split around mention links + /// ("… and", "@name", "…") comes back whole. Ignored when what remains is only short fragments. + static func postParagraphs(_ paragraphs: [String]) -> [String] { + var texts = paragraphs + if let handle = texts.indices.first(where: { texts[$0].hasPrefix("@") && !texts[$0].contains(" ") }), handle > 0 + { + var drop = [handle - 1, handle] + if handle + 1 < texts.count, isTimestamp(texts[handle + 1]) { drop.append(handle + 1) } + for index in drop.sorted(by: >) { texts.remove(at: index) } + } + let body = texts.filter { + !$0.hasPrefix("Replying to") && $0 != "Show more" && $0 != "Post your reply" && !isTimestamp($0) + } + return body.contains(where: { $0.count >= 20 }) ? body : [] + } + + private static func isTimestamp(_ text: String) -> Bool { + text.range(of: #"^(\d+[smhd]|[A-Z][a-z]{2} \d{1,2}(, \d{4})?|·)$"#, options: .regularExpression) != nil + } + + private static func collectText( + _ element: AXUIElement, focused: AXUIElement, into paragraphs: inout [String], budget: inout Int, + reachedFocus: inout Bool + ) { + guard budget > 0, !reachedFocus else { return } + budget -= 1 + if CFEqual(element, focused) { + reachedFocus = true + return + } + if attribute(element, kAXRoleAttribute) as? String == kAXStaticTextRole, + let text = attribute(element, kAXValueAttribute) as? String + { + let trimmed = text.trimmingCharacters(in: .whitespacesAndNewlines) + if !trimmed.isEmpty { paragraphs.append(trimmed) } + return + } + guard let children = attribute(element, kAXChildrenAttribute) as? [AXUIElement] else { return } + for child in children { + collectText(child, focused: focused, into: ¶graphs, budget: &budget, reachedFocus: &reachedFocus) + if reachedFocus { return } + } + } + + /// `AXSelectedText` of the focused element, which Safari, Slack and native text views all expose. + private static func accessibilitySelection() -> String? { + // Chrome only exposes web-content accessibility (incl. AXSelectedText) once an assistive client asks for it + // (focusedElement sets AXEnhancedUserInterface on the app). + guard let focused = focusedElement() else { return nil } + return attribute(focused, kAXSelectedTextAttribute) as? String + } + + /// Sends ⌘C to the frontmost app and reads the pasteboard, restoring its previous contents afterwards. + private static func pasteboardSelection() async -> String? { + let pasteboard = NSPasteboard.general + let previous = pasteboard.string(forType: .string) + let changeCount = pasteboard.changeCount + KeystrokeSender.send(keyCode: UInt16(kVK_ANSI_C), command: true) + for _ in 0..<20 { + try? await Task.sleep(for: .milliseconds(25)) + if pasteboard.changeCount != changeCount { break } + } + guard pasteboard.changeCount != changeCount else { return nil } + let text = pasteboard.string(forType: .string) + if let previous { + pasteboard.clearContents() + pasteboard.setString(previous, forType: .string) + } + return text + } +} + +enum KeystrokeSender { + /// Posts a key press system-wide, or straight to one process when `pid` is given (no need for it to be frontmost). + static func send(keyCode: UInt16, command: Bool, to pid: pid_t? = nil) { + let source = CGEventSource(stateID: .combinedSessionState) + guard let down = CGEvent(keyboardEventSource: source, virtualKey: keyCode, keyDown: true), + let up = CGEvent(keyboardEventSource: source, virtualKey: keyCode, keyDown: false) + else { return } + if command { + down.flags = .maskCommand + up.flags = .maskCommand + } + if let pid { + down.postToPid(pid) + up.postToPid(pid) + } else { + down.post(tap: .cghidEventTap) + up.post(tap: .cghidEventTap) + } + } +} + +/// Global keys: 9 (or ⌃⌥R) drafts, 0 regenerates the last draft. Needs Accessibility trust for global key events. A global +/// monitor observes only, so a typed 9 / 0 still reaches the app you're typing in; the handlers delete it again +/// when a text field had focus. +final class HotkeyMonitor { + enum Action: Sendable { + case draft(typedTrigger: Bool) + case regenerate(typedTrigger: Bool) + } + + private var monitor: Any? + + init(handler: @escaping @Sendable (Action) -> Void) { + monitor = NSEvent.addGlobalMonitorForEvents(matching: .keyDown) { event in + let modifiers = event.modifierFlags.intersection(.deviceIndependentFlagsMask) + if event.keyCode == UInt16(kVK_ANSI_9) && modifiers.isEmpty { return handler(.draft(typedTrigger: true)) } + if event.keyCode == UInt16(kVK_ANSI_0) && modifiers.isEmpty { + return handler(.regenerate(typedTrigger: true)) + } + if event.keyCode == UInt16(kVK_ANSI_R) && modifiers == [.control, .option] { + handler(.draft(typedTrigger: false)) + } + } + } + + deinit { + if let monitor { NSEvent.removeMonitor(monitor) } + } +} diff --git a/Sources/ShortReplyDemo/demo.sh b/Sources/ShortReplyDemo/demo.sh new file mode 100755 index 0000000..77b8394 --- /dev/null +++ b/Sources/ShortReplyDemo/demo.sh @@ -0,0 +1,44 @@ +#!/bin/zsh +# Short Reply demo: the menu-bar app, plus a terminal (Ghostty if installed) with macmon/asitop (GPU / ANE / power) on top +# and the live model log below. +# Sources/ShortReplyDemo/demo.sh [--x | --mock] [model dir] (or SHORT_REPLY_MODEL_DIR; default .mobius/short-reply-lm/coreml-demo) +# --x also opens x.com in Chrome (real feed; draft replies, don't press Post) +# --mock also opens the local mock feed page (fictional posts, nothing is posted) in Chrome +set -e +ROOT=$(cd "$(dirname "$0")/../.." && pwd) +BROWSER="Google Chrome" +OPEN="" +case "${1:-}" in + --x) OPEN="https://x.com/home"; shift ;; + --mock) OPEN="$ROOT/Sources/ShortReplyDemo/mock-feed/index.html"; shift ;; +esac +MODEL=${1:-${SHORT_REPLY_MODEL_DIR:-$ROOT/.mobius/short-reply-lm/coreml-demo}} +LOG=${TMPDIR:-/tmp}/short-reply-demo.log +# macmon (brew install macmon) needs no sudo and works on M5; asitop 0.0.24 crashes there (KeyError E0-Cluster_active). +if command -v macmon >/dev/null; then + MONITOR="$(command -v macmon)" +else + MONITOR="sudo sh -c 'pkill -x powermetrics; exec $(command -v asitop || echo "$HOME/.local/bin/asitop")'" +fi + +swift build -c release --product ShortReplyDemo --package-path "$ROOT" +: > "$LOG" +# One tmux session, reused across launches: the monitor on top, the model log below. +if ! tmux has-session -t short-reply 2>/dev/null; then + # The pane stays open if the monitor exits, so the reason is visible. + tmux new-session -d -s short-reply -x 160 -y 70 \ + "$MONITOR; echo; echo \"monitor exited (\$?) — press Enter to retry\"; read; $MONITOR" + tmux set-option -t short-reply remain-on-exit on + tmux split-window -v -t short-reply "tail -n 300 -F '$LOG'" + if [[ -d /Applications/Ghostty.app ]]; then + open -na Ghostty --args -e "$(command -v tmux)" attach -t short-reply + else + osascript -e 'tell application "Terminal" to do script "tmux attach -t short-reply"' \ + -e 'tell application "Terminal" to activate' >/dev/null + fi +fi +pkill -f "release/ShortReplyDemo" 2>/dev/null || true +cd "$ROOT" +SHORT_REPLY_MODEL_DIR="$MODEL" "$ROOT/.build/release/ShortReplyDemo" > "$LOG" 2> "$LOG.stderr" & +if [[ -n "$OPEN" ]]; then open -a "$BROWSER" "$OPEN"; fi +echo "demo pid $! · log $LOG · select a post anywhere and press 9 (or ⌃⌥R)" diff --git a/Sources/ShortReplyDemo/main.swift b/Sources/ShortReplyDemo/main.swift new file mode 100644 index 0000000..caf2ffc --- /dev/null +++ b/Sources/ShortReplyDemo/main.swift @@ -0,0 +1,122 @@ +import AppKit +import SwiftUI + +// Menu-bar app: select a post anywhere, press ⌃⌥R, get a drafted reply in a floating panel. +// Top-level code (not @main) so the macOS 15 Core ML APIs can be gated at runtime while the package targets macOS 14. + +@available(macOS 15.0, *) +@MainActor +final class AppDelegate: NSObject, NSApplicationDelegate { + private var statusItem: NSStatusItem? + private let model = ReplyPanelModel() + private var panel: ReplyPanel? + private var hotkey: HotkeyMonitor? + /// With a reply box focused, 9 drafts and pastes without showing the panel. + private var autoInsert = true + private var autoInsertItem: NSMenuItem? + + func applicationDidFinishLaunching(_ notification: Notification) { + let item = NSStatusBar.system.statusItem(withLength: NSStatusItem.variableLength) + item.button?.title = "↩︎" + item.button?.toolTip = "Short Reply — 9 drafts a reply into the focused reply box, 0 regenerates it" + let menu = NSMenu() + menu.addItem( + withTitle: "Draft reply for selection 9 / ⌃⌥R", action: #selector(draftFromMenu), keyEquivalent: "") + menu.addItem(withTitle: "Draft reply from clipboard", action: #selector(draftFromClipboard), keyEquivalent: "") + menu.addItem(.separator()) + let auto = NSMenuItem( + title: "Auto-insert into the focused reply box (no panel)", action: #selector(toggleAutoInsert), + keyEquivalent: "") + auto.state = .on + menu.addItem(auto) + autoInsertItem = auto + menu.addItem(.separator()) + menu.addItem(withTitle: "Quit", action: #selector(NSApplication.terminate(_:)), keyEquivalent: "q") + menu.items.forEach { $0.target = self } + item.menu = menu + statusItem = item + + panel = ReplyPanel(model: model) + hotkey = HotkeyMonitor { [weak self] action in + Task { @MainActor in + switch action { + case .draft(let typed): self?.draftFromSelection(typedTrigger: typed) + case .regenerate(let typed): + if typed { SelectionReader.deleteTypedTriggerIfEditing() } + self?.regenerate() + } + } + } + if !SelectionReader.isTrusted { + print( + "accessibility: not trusted yet — grant access in System Settings > Privacy & Security > Accessibility") + } + Task { await model.load() } + } + + @objc func draftFromMenu() { draftFromSelection(typedTrigger: false) } + + /// 0: a new sampled reply for the last post, replacing the box contents (or a first draft if there is none yet). + func regenerate() { + guard !model.post.isEmpty, model.phase != .drafting else { return draftFromSelection(typedTrigger: false) } + let previous = model.sourceApplication + Task { @MainActor in + await model.draft(post: model.post) + guard case .done = model.phase else { return panel?.present() ?? () } + if autoInsert, previous != nil { + model.sourceApplication = previous + model.insertReply(replace: true) + } else { + panel?.present() + } + } + } + + @objc func toggleAutoInsert() { + autoInsert.toggle() + autoInsertItem?.state = autoInsert ? .on : .off + } + + func draftFromSelection(typedTrigger: Bool) { + let previous = NSWorkspace.shared.frontmostApplication + Task { @MainActor in + let context = await SelectionReader.context(typedTrigger: typedTrigger) + await draft(context.text, source: previous, insertDirectly: autoInsert && context.fieldFocused) + } + } + + @objc func draftFromClipboard() { + Task { @MainActor in + await draft(NSPasteboard.general.string(forType: .string), source: nil, insertDirectly: false) + } + } + + private func draft(_ text: String?, source: NSRunningApplication?, insertDirectly: Bool) async { + guard let text, !text.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { + model.show(error: "Select a post, or open its reply box, then press 9.") + panel?.present() + return + } + model.sourceApplication = source + if !insertDirectly { panel?.present() } + await model.draft(post: text) + guard insertDirectly else { return } + if case .done = model.phase { + model.insertReply(replace: model.isRegenerate) + } else { + panel?.present() // an error: show it + } + } +} + +setvbuf(stdout, nil, _IOLBF, 0) +if #available(macOS 15.0, *) { + let application = NSApplication.shared + let delegate = AppDelegate() + application.delegate = delegate + application.setActivationPolicy(.accessory) + application.run() +} else { + FileHandle.standardError.write(Data("ShortReplyDemo needs macOS 15\n".utf8)) + exit(2) +} diff --git a/Sources/ShortReplyDemo/mock-feed/index.html b/Sources/ShortReplyDemo/mock-feed/index.html new file mode 100644 index 0000000..f088a79 --- /dev/null +++ b/Sources/ShortReplyDemo/mock-feed/index.html @@ -0,0 +1,166 @@ + + + + +Home / Feed (mock) + + + + +
+ +
+
For you
Following
+
Y
What is happening?!
+
+
Mock page for a demo. Every post is fictional; “Reply” only appends text on this page. Nothing is sent anywhere.
+
+ +
+ + + diff --git a/Tests/FluidUseTests/ShortReplyTests.swift b/Tests/FluidUseTests/ShortReplyTests.swift new file mode 100644 index 0000000..562d04a --- /dev/null +++ b/Tests/FluidUseTests/ShortReplyTests.swift @@ -0,0 +1,49 @@ +import XCTest + +@testable import FluidUse + +final class ShortReplyTests: XCTestCase { + func testCleanReplyMirrorsThePythonCleanup() throws { + guard #available(macOS 15.0, iOS 18.0, *) else { throw XCTSkip("ShortReplyManager needs macOS 15") } + XCTAssertEqual(ShortReplyManager.cleanReply("Reply: Nice job!\nExtra text"), "Nice job!") + XCTAssertEqual(ShortReplyManager.cleanReply(" \n- “Congrats on the 5K!” "), "Congrats on the 5K!") + XCTAssertEqual(ShortReplyManager.cleanReply(" \n "), "") + } + + func testLinksAreStrippedFromPosts() throws { + guard #available(macOS 15.0, iOS 18.0, *) else { throw XCTSkip("ShortReplyManager needs macOS 15") } + XCTAssertEqual( + ShortReplyManager.stripLinks("Translation with pic.x.com/2eKrZU3eEr and https://example.com/x done"), + "Translation with and done") + } + + /// The Swift prompt must tokenize exactly as `reply.py` + `apply_chat_template(enable_thinking=False)`. + /// Reference ids come from the Python tokenizer for the post "I finished my first 5K today." (78 tokens). + func testPromptTokensMatchPythonReference() throws { + guard let directory = ProcessInfo.processInfo.environment["SHORT_REPLY_MODEL_DIR"], !directory.isEmpty else { + throw XCTSkip("Set SHORT_REPLY_MODEL_DIR (folder with tokenizer.json) to run") + } + let tokenizer = try QwenBPETokenizer( + tokenizerJsonURL: URL(fileURLWithPath: directory).appendingPathComponent("tokenizer.json")) + let system = + "Write one natural, very short reply to the social post as a separate person. React to a concrete detail in " + + "the post when possible. Do not invent facts or personal experiences. Use at most 12 words. Output only the reply." + let text = + "<|im_start|>system\n\(system)<|im_end|>\n<|im_start|>user\nPost: I finished my first 5K today.\nReply:<|im_end|>\n" + + "<|im_start|>assistant\n\n\n\n\n" + let reference = [ + 151644, 8948, 198, 7985, 825, 5810, 11, 1602, 2805, 9851, 311, 279, 3590, 1736, 438, 264, 8651, 1697, 13, + 3592, + 311, 264, 14175, 7716, 304, 279, 1736, 979, 3204, 13, 3155, 537, 17023, 13064, 476, 4345, 11449, 13, 5443, + 518, + 1429, 220, 16, 17, 4244, 13, 9258, 1172, 279, 9851, 13, 151645, 198, 151644, 872, 198, 4133, 25, 358, 8060, + 847, + 1156, 220, 20, 42, 3351, 624, 20841, 25, 151645, 198, 151644, 77091, 198, 151667, 271, 151668, 271, + ] + XCTAssertEqual(try tokenizer.encode(text), reference) + XCTAssertEqual( + tokenizer.decode(try tokenizer.encode("Congrats on finishing your first 5K!")), + "Congrats on finishing your first 5K!") + XCTAssertEqual(tokenizer.decode([151645, 8948]), "system", "special tokens are dropped when decoding") + } +} From 9c993b1929ae68e6a4d6c02946ebce91bf57de50 Mon Sep 17 00:00:00 2001 From: Alex-Wengg Date: Thu, 1 Oct 2026 22:05:03 -0400 Subject: [PATCH 2/2] Short reply: review fixes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Package.swift / README: keep main's Intern-Decision target and section (the branch rebuild had taken older copies). Host: compile the package once via KevManager.compiled; echo guard on greedy replies (seeded resample, as reply.py); long-post trimming re-checks the tokenized prompt and flags the character cut; empty replies are an error; prefill K/V copied by the arrays' strides with layout checks; NaN-safe sampling; regenerate returns the last attempt instead of re-decoding a known duplicate. Demo: regenerate keys off the raw post text (normalized text no longer resets sampling), ignores keys while a draft is in flight, Quit keeps its responder-chain target, bare 9/0 keys can be switched off (⌃⌥R stays), Chromium-only AXEnhancedUserInterface, post lookup visits each element once. README fence fixed; tokenizer decode reuses the special-id set. --- Package.swift | 5 +- README.md | 34 ++++++ Sources/FluidUse/Qwen/QwenBPETokenizer.swift | 9 +- .../ShortReply/ShortReplyManager.swift | 113 +++++++++++++----- Sources/ShortReplyCheck/main.swift | 2 +- Sources/ShortReplyDemo/README.md | 10 +- Sources/ShortReplyDemo/ReplyPanelModel.swift | 8 +- Sources/ShortReplyDemo/SelectionReader.swift | 46 +++---- Sources/ShortReplyDemo/main.swift | 42 +++++-- 9 files changed, 200 insertions(+), 69 deletions(-) diff --git a/Package.swift b/Package.swift index 8866260..2c6f496 100644 --- a/Package.swift +++ b/Package.swift @@ -54,11 +54,12 @@ let package = Package( .executableTarget(name: "ImageSortCheck", dependencies: ["ImageSort", "FluidUse"]), .executableTarget(name: "ImageSortDemo", dependencies: ["ImageSort"], exclude: ["README.md"]), .executableTarget(name: "KevCheck", dependencies: ["FluidUse"]), + .executableTarget(name: "InternDecisionCheck", dependencies: ["FluidUse"]), + .executableTarget(name: "ShortReplyCheck", dependencies: ["FluidUse"]), + .executableTarget(name: "ShortReplyDemo", dependencies: ["FluidUse"], exclude: ["README.md", "demo.sh", "mock-feed"]), .executableTarget(name: "GLiClassServe", dependencies: ["FluidUse"]), .executableTarget( name: "KevGuessWhoDemo", dependencies: ["FluidUse", "SortAnything"], exclude: ["README.md", "demo.sh"]), - .executableTarget(name: "ShortReplyCheck", dependencies: ["FluidUse"]), - .executableTarget(name: "ShortReplyDemo", dependencies: ["FluidUse"], exclude: ["README.md", "demo.sh", "mock-feed"]), .testTarget( name: "FluidUseTests", dependencies: ["FluidUse", "LayaTetris"], resources: [.copy("Fixtures")] diff --git a/README.md b/README.md index 0c99dc3..0f874b4 100644 --- a/README.md +++ b/README.md @@ -153,6 +153,40 @@ On an M5 Pro a short ticket with two questions takes 18 ms and a Wikipedia bio w `swift run -c release KevGuessWhoDemo` plays Guess Who over 80 Wikipedia people with it ([Sources/KevGuessWhoDemo](Sources/KevGuessWhoDemo/README.md)). +## Intern-Decision + +`InternDecisionManager` runs [Intern-Decision-0.8B](https://huggingface.co/internlm/Intern-Decision-0.8B) (Shanghai AI +Laboratory, Qwen3.5 backbone, Apache-2.0) on the GPU through Core ML. A request is a JSON state and up to 16 named +questions (multiple choice, yes/no, or a score); every question is answered in one call from the logits before its +`` marker, with the checkpoint's calibration temperature. The prompt is byte for byte the checkpoint's own +compiler and chat template, so answers match its reference engine (0 of 60 differ on its published suites, max +probability delta 0.004). The pinned snapshot downloads from +[FluidInference/intern-decision-0.8b-coreml](https://huggingface.co/FluidInference/intern-decision-0.8b-coreml) on first +use (~3.4 GB, macOS 14 / iOS 17). Text only; images are not exported. + +```swift +let model = try await InternDecisionManager.load(from: try await InternDecisionModelStore.ensure()) +let result = try await model.decide( + state: ["channel": "email", "message": "Charged twice for my annual renewal. Refund the duplicate before Friday."], + questions: [ + ("team", .choice("Which team should handle this?", options: [("billing", "Refunds."), ("technical", "Bugs.")])), + ("frustrated", .noul("Is the customer frustrated?")), + ("urgency", .score("How urgent is this?", levels: ["Low", "Medium", "High"])), + ]) +print(result.answers.map { "\($0.field): \($0.decision) \($0.confidence)" }) +``` + +On an M5 Pro that request (319 tokens, three fields) takes 61 ms in the 320-token bucket; the checkpoint's own PyTorch +path on the same Mac takes 150 ms (bf16). Buckets are 320, 512 and 1,024 tokens and the pass costs the bucket, not the +request. `swift run -c release InternDecisionCheck bench ` reproduces the number. + +A second snapshot, `InternDecisionModelStore.ensure(.showdown)`, is the same model fine-tuned to pick Pokémon Showdown +battle actions ([FluidInference/intern-decision-0.8b-showdown-coreml](https://huggingface.co/FluidInference/intern-decision-0.8b-showdown-coreml)), +distilled from Intern-Decision-4B on a MacBook. It answers the same typed questions; given a battle state and the legal +moves and switches as `choice` options it matches its 4B teacher (24-6 vs poke-env's max-power player, 9-21 vs its +heuristic player over 30 battles). Buckets are 512, 640 and 1,024 tokens with int8 weights: about 0.85 GB in memory +with one bucket in use, 1.9 GB to download, the same 90 ms per decision as fp16. + ## Short replies `ShortReplyManager` drafts one short reply to a social post with a sub-1B model: diff --git a/Sources/FluidUse/Qwen/QwenBPETokenizer.swift b/Sources/FluidUse/Qwen/QwenBPETokenizer.swift index e6eb45a..81df1ae 100644 --- a/Sources/FluidUse/Qwen/QwenBPETokenizer.swift +++ b/Sources/FluidUse/Qwen/QwenBPETokenizer.swift @@ -2,8 +2,8 @@ import Foundation import os /// Byte-level BPE tokenizer for Qwen `tokenizer.json` files (Qwen2 through Qwen3.5): NFC normalization, added tokens -/// matched literally, Qwen's split regex, GPT-2 byte-to-unicode mapping, then merges by rank. Encoding only; no special -/// tokens are added. +/// matched literally, Qwen's split regex, GPT-2 byte-to-unicode mapping, then merges by rank. `encode` adds no +/// special tokens; `decode` drops them. public final class QwenBPETokenizer: Sendable { private let vocabulary: [String: Int] private let mergeRanks: [String: Int] @@ -13,6 +13,7 @@ public final class QwenBPETokenizer: Sendable { /// Reverse tables for `decode`. private let tokenForID: [Int: String] private let byteForCharacter: [Character: UInt8] + private let specialIDs: Set /// Encoded pieces; ordinary text repeats the same words, so this saves most of the merge loops. private let cache = OSAllocatedUnfairLock<[String: [Int]]>(initialState: [:]) @@ -53,6 +54,7 @@ public final class QwenBPETokenizer: Sendable { var byteForCharacter = [Character: UInt8](minimumCapacity: 256) for (byte, character) in byteToCharacter.enumerated() { byteForCharacter[character] = UInt8(byte) } self.byteForCharacter = byteForCharacter + specialIDs = Set(addedTokens.map(\.id)) _ = try Self.splitter() } @@ -83,9 +85,8 @@ public final class QwenBPETokenizer: Sendable { /// Text for `ids`, dropping added (special) tokens; invalid byte sequences decode lossily. public func decode(_ ids: [Int]) -> String { - let special = Set(addedTokens.map(\.id)) var bytes: [UInt8] = [] - for id in ids where !special.contains(id) { + for id in ids where !specialIDs.contains(id) { guard let token = tokenForID[id] else { continue } for character in token { if let byte = byteForCharacter[character] { bytes.append(byte) } diff --git a/Sources/FluidUse/ShortReply/ShortReplyManager.swift b/Sources/FluidUse/ShortReply/ShortReplyManager.swift index bf67791..ffb0718 100644 --- a/Sources/FluidUse/ShortReply/ShortReplyManager.swift +++ b/Sources/FluidUse/ShortReply/ShortReplyManager.swift @@ -56,7 +56,7 @@ public actor ShortReplyManager { Config.self, from: Data(contentsOf: directory.appendingPathComponent("config.json"))) var package = directory.appendingPathComponent(config.package) if package.pathExtension == "mlpackage" { - package = try await MLModel.compileModel(at: package) + package = try await KevManager.compiled(package) // compile once, keep the .mlmodelc beside the package } return try ShortReplyManager( config: config, compiled: package, tokenizer: directory.appendingPathComponent("tokenizer.json"), @@ -94,18 +94,41 @@ public actor ShortReplyManager { return try tokenizer.encode(text) } - /// Whitespace-normalized, links dropped, and cut (by tokens, from the end) so the prompt fits the prefill length. + /// Whitespace-normalized, links dropped, and cut (by tokens, from the end) until the full prompt fits the prefill + /// length. The fit is checked on the re-tokenized prompt, since a cut can land mid-word or mid-merge. func preparePost(_ rawPost: String) throws -> (post: String, trimmed: Bool) { var post = Self.stripLinks(rawPost).split(whereSeparator: \.isWhitespace).joined(separator: " ") guard !post.isEmpty else { throw ShortReplyError.emptyPost } - if post.count > config.maxPostCharacters { post = String(post.prefix(config.maxPostCharacters)) } - let overhead = try promptTokens(for: "").count - let budget = config.prefillLength - overhead - guard budget > 8 else { throw ShortReplyError.invalidAsset("prefill length too short for the system prompt") } - let postTokens = try tokenizer.encode(post) - guard postTokens.count > budget else { return (post, false) } - let cut = tokenizer.decode(Array(postTokens.prefix(budget - 1))).trimmingCharacters(in: .whitespaces) - return (cut + "…", true) + var trimmed = false + if post.count > config.maxPostCharacters { + post = String(post.prefix(config.maxPostCharacters)) + trimmed = true + } + let length = config.prefillLength + if try promptTokens(for: post).count <= length { return (post, trimmed) } + var keep = max(1, try tokenizer.encode(post).count - (try promptTokens(for: post).count - length) - 1) + while keep > 0 { + let cut = + tokenizer.decode(Array(try tokenizer.encode(post).prefix(keep))) + .trimmingCharacters(in: .whitespaces).replacingOccurrences(of: "\u{FFFD}", with: "") + "…" + if try promptTokens(for: cut).count <= length { return (cut, true) } + keep -= 2 + } + throw ShortReplyError.invalidAsset("prefill length too short for the system prompt") + } + + /// True when the reply repeats a run of five or more consecutive words from the post (mirrors `reply.py`). + static func echoes(_ post: String, _ reply: String, run: Int = 5) -> Bool { + let postWords = post.lowercased().split { !$0.isLetter && !$0.isNumber && $0 != "'" }.map(String.init) + let replyWords = reply.lowercased().split { !$0.isLetter && !$0.isNumber && $0 != "'" }.map(String.init) + guard replyWords.count >= run, postWords.count >= run else { return false } + for start in 0...(replyWords.count - run) { + let window = Array(replyWords[start.. String { @@ -116,15 +139,27 @@ public actor ShortReplyManager { /// `variation` 0 decodes greedily (the benchmarked behavior); higher values sample (temperature 0.7, top-p 0.9) /// with a seed derived from the value, so "regenerate" gives a different, repeatable reply. `avoiding` rejects - /// replies equal to earlier drafts (up to a few attempts). + /// replies equal to earlier drafts (up to a few attempts). A greedy reply that parrots the post is replaced by the + /// first seeded sample that does not, as `reply.py` does. public func draft( for rawPost: String, variation: Int = 0, avoiding previous: Set = [] ) async throws -> Draft { - for attempt in 0..<(variation == 0 ? 1 : 4) { - let draft = try await decodeDraft(for: rawPost, variation: variation == 0 ? 0 : variation + attempt * 1000) - if variation == 0 || !previous.contains(draft.reply) { return draft } + if variation == 0 { + let greedy = try await decodeDraft(for: rawPost, variation: 0) + guard Self.echoes(greedy.post, greedy.reply) else { return greedy } + for attempt in 1...6 { + let sampled = try await decodeDraft(for: rawPost, variation: attempt) + if !Self.echoes(sampled.post, sampled.reply) { return sampled } + } + return greedy + } + var last: Draft? + for attempt in 0..<4 { + let draft = try await decodeDraft(for: rawPost, variation: variation + attempt * 1000) + if !previous.contains(draft.reply) { return draft } + last = draft } - return try await decodeDraft(for: rawPost, variation: variation) + return last! } private func decodeDraft(for rawPost: String, variation: Int) async throws -> Draft { @@ -158,6 +193,7 @@ public actor ShortReplyManager { } let decodeSeconds = seconds(since: decodeStarted) let reply = ShortReplyManager.cleanReply(tokenizer.decode(generated)) + guard !reply.isEmpty else { throw ShortReplyError.emptyReply } let timing = Timing( prefillSeconds: prefillSeconds, decodeSeconds: decodeSeconds, generatedTokens: generated.count) logger.info( @@ -220,25 +256,43 @@ public actor ShortReplyManager { else { throw ShortReplyError.invalidAsset("prefill output has no keys/values") } let headDim = config.headDim let kvHeads = config.kvHeads - let layerElements = kvHeads * promptLength * headDim - for layer in 0...size) + dst.advanced(by: head * headStride), + src.baseAddress!.advanced(by: head * rowElements), + rowElements * MemoryLayout.size) } } } @@ -301,6 +355,7 @@ public actor ShortReplyManager { mass += probabilities[index] if mass >= topP { break } } + guard mass.isFinite, mass > 0 else { return argmax(logits) } var draw = Float.random(in: 0.. Double { let sorted = values.sorted() - return sorted[sorted.count / 2] + return sorted.isEmpty ? 0 : sorted[sorted.count / 2] } print("exact reply match: \(matches)/\(rows.count)") print( diff --git a/Sources/ShortReplyDemo/README.md b/Sources/ShortReplyDemo/README.md index 7ca87f9..5191bb8 100644 --- a/Sources/ShortReplyDemo/README.md +++ b/Sources/ShortReplyDemo/README.md @@ -20,8 +20,14 @@ The model directory holds `short_reply_0_6b.mlpackage` (functions `prefill` and ```bash hf download FluidInference/short-reply-0.6b-coreml --local-dir ~/Models/short-reply-0.6b-coreml Sources/ShortReplyDemo/demo.sh --x ~/Models/short-reply-0.6b-coreml # or set SHORT_REPLY_MODEL_DIR -``` First launch compiles -the package (a few seconds) and runs one warm-up prompt; the log prints `model ready`. +``` + +The first launch compiles the package once (kept as `.mlmodelc` beside it) and runs one warm-up prompt; the log +prints `model ready`. + +**Keys.** The bare **9** and **0** keys are observed system-wide while the app runs (a global monitor can't swallow +them, so a 9 typed into a focused field is deleted again by the app). That is convenient for a demo and wrong for +daily use: turn off "Bare 9 / 0 keys (demo mode)" in the menu-bar item and use ⌃⌥R instead. **Mock feed for recording.** `mock-feed/index.html` is a local feed of fictional posts with a reply box under each; nothing on it is posted anywhere. Select a post, press 9, then **Insert**: the page routes the paste into that post's reply box, and "Reply" only diff --git a/Sources/ShortReplyDemo/ReplyPanelModel.swift b/Sources/ShortReplyDemo/ReplyPanelModel.swift index fb308c8..c5c9d8a 100644 --- a/Sources/ShortReplyDemo/ReplyPanelModel.swift +++ b/Sources/ShortReplyDemo/ReplyPanelModel.swift @@ -20,8 +20,8 @@ final class ReplyPanelModel: ObservableObject { @Published var timing: ShortReplyManager.Timing? @Published var drafts = 0 @Published var trimmed = false - /// Same post again → sample a new reply instead of repeating the greedy one. - private var lastPost = "" + /// The post text as read from the app; drafts of the same raw text sample a new reply instead of repeating. + private(set) var rawPost = "" private var variation = 0 private var previousReplies: Set = [] var sourceApplication: NSRunningApplication? @@ -57,10 +57,10 @@ final class ReplyPanelModel: ObservableObject { show(error: "Model is still loading.") return } - if text == lastPost { + if text == rawPost { variation += 1 } else { - lastPost = text + rawPost = text variation = 0 previousReplies = [] } diff --git a/Sources/ShortReplyDemo/SelectionReader.swift b/Sources/ShortReplyDemo/SelectionReader.swift index 99e67a8..943104e 100644 --- a/Sources/ShortReplyDemo/SelectionReader.swift +++ b/Sources/ShortReplyDemo/SelectionReader.swift @@ -40,7 +40,10 @@ enum SelectionReader { private static func focusedElement() -> AXUIElement? { guard let application = NSWorkspace.shared.frontmostApplication else { return nil } let element = AXUIElementCreateApplication(application.processIdentifier) - AXUIElementSetAttributeValue(element, "AXEnhancedUserInterface" as CFString, kCFBooleanTrue) + if Self.isChromium(application) { + // Chromium browsers expose web-content accessibility only once an assistive client asks for it. + AXUIElementSetAttributeValue(element, "AXEnhancedUserInterface" as CFString, kCFBooleanTrue) + } var focused: CFTypeRef? guard AXUIElementCopyAttributeValue(element, kAXFocusedUIElementAttribute as CFString, &focused) == .success, let focused, CFGetTypeID(focused) == AXUIElementGetTypeID() @@ -48,6 +51,14 @@ enum SelectionReader { return unsafeBitCast(focused, to: AXUIElement.self) } + private static func isChromium(_ application: NSRunningApplication) -> Bool { + let id = application.bundleIdentifier?.lowercased() ?? "" + return [ + "com.google.chrome", "org.chromium", "com.brave.browser", "com.microsoft.edgemac", "company.thebrowser", + ] + .contains { id.hasPrefix($0) } + } + private static func attribute(_ element: AXUIElement, _ name: String) -> CFTypeRef? { var value: CFTypeRef? return AXUIElementCopyAttributeValue(element, name as CFString, &value) == .success ? value : nil @@ -62,18 +73,23 @@ enum SelectionReader { } /// Static text above the focused field within the nearest enclosing container that holds a real paragraph: - /// on a reply dialog that is the post being answered. Short fragments (names, times, "Replying to …") are dropped. + /// on a reply dialog that is the post being answered. Walks up one level at a time, scanning only the siblings + /// that precede the path to the field, so every element is visited once. private static func postAbove(_ focused: AXUIElement) -> String? { var node = focused + var paragraphs: [String] = [] // document order, everything before the field so far + var budget = 4000 for _ in 0..<48 { // X nests the reply box dozens of AXGroups deep - guard let parentValue = attribute(node, kAXParentAttribute), + guard budget > 0, let parentValue = attribute(node, kAXParentAttribute), CFGetTypeID(parentValue) == AXUIElementGetTypeID() else { return nil } let parent = unsafeBitCast(parentValue, to: AXUIElement.self) - var paragraphs: [String] = [] - var budget = 2500 - var reachedFocus = false - collectText(parent, focused: focused, into: ¶graphs, budget: &budget, reachedFocus: &reachedFocus) + var before: [String] = [] + for child in attribute(parent, kAXChildrenAttribute) as? [AXUIElement] ?? [] { + if CFEqual(child, node) { break } + collectText(child, into: &before, budget: &budget) + } + paragraphs = before + paragraphs let kept = Self.postParagraphs(paragraphs) if !kept.isEmpty { return kept.joined(separator: " ") } node = parent @@ -102,16 +118,9 @@ enum SelectionReader { text.range(of: #"^(\d+[smhd]|[A-Z][a-z]{2} \d{1,2}(, \d{4})?|·)$"#, options: .regularExpression) != nil } - private static func collectText( - _ element: AXUIElement, focused: AXUIElement, into paragraphs: inout [String], budget: inout Int, - reachedFocus: inout Bool - ) { - guard budget > 0, !reachedFocus else { return } + private static func collectText(_ element: AXUIElement, into paragraphs: inout [String], budget: inout Int) { + guard budget > 0 else { return } budget -= 1 - if CFEqual(element, focused) { - reachedFocus = true - return - } if attribute(element, kAXRoleAttribute) as? String == kAXStaticTextRole, let text = attribute(element, kAXValueAttribute) as? String { @@ -120,10 +129,7 @@ enum SelectionReader { return } guard let children = attribute(element, kAXChildrenAttribute) as? [AXUIElement] else { return } - for child in children { - collectText(child, focused: focused, into: ¶graphs, budget: &budget, reachedFocus: &reachedFocus) - if reachedFocus { return } - } + for child in children { collectText(child, into: ¶graphs, budget: &budget) } } /// `AXSelectedText` of the focused element, which Safari, Slack and native text views all expose. diff --git a/Sources/ShortReplyDemo/main.swift b/Sources/ShortReplyDemo/main.swift index caf2ffc..9358507 100644 --- a/Sources/ShortReplyDemo/main.swift +++ b/Sources/ShortReplyDemo/main.swift @@ -14,36 +14,55 @@ final class AppDelegate: NSObject, NSApplicationDelegate { /// With a reply box focused, 9 drafts and pastes without showing the panel. private var autoInsert = true private var autoInsertItem: NSMenuItem? + /// Bare 9 / 0 keys are observed system-wide; switch them off outside a demo (⌃⌥R always works). + private var bareKeys = true + private var bareKeysItem: NSMenuItem? func applicationDidFinishLaunching(_ notification: Notification) { let item = NSStatusBar.system.statusItem(withLength: NSStatusItem.variableLength) item.button?.title = "↩︎" item.button?.toolTip = "Short Reply — 9 drafts a reply into the focused reply box, 0 regenerates it" let menu = NSMenu() - menu.addItem( - withTitle: "Draft reply for selection 9 / ⌃⌥R", action: #selector(draftFromMenu), keyEquivalent: "") - menu.addItem(withTitle: "Draft reply from clipboard", action: #selector(draftFromClipboard), keyEquivalent: "") + for (title, action) in [ + ("Draft reply for selection 9 / ⌃⌥R", #selector(draftFromMenu)), + ("Draft reply from clipboard", #selector(draftFromClipboard)), + ] { + let item = NSMenuItem(title: title, action: action, keyEquivalent: "") + item.target = self + menu.addItem(item) + } menu.addItem(.separator()) let auto = NSMenuItem( title: "Auto-insert into the focused reply box (no panel)", action: #selector(toggleAutoInsert), keyEquivalent: "") + auto.target = self auto.state = .on menu.addItem(auto) autoInsertItem = auto + let bare = NSMenuItem( + title: "Bare 9 / 0 keys (demo mode)", action: #selector(toggleBareKeys), keyEquivalent: "") + bare.target = self + bare.state = .on + menu.addItem(bare) + bareKeysItem = bare menu.addItem(.separator()) + // Quit keeps a nil target so the responder chain reaches NSApplication. menu.addItem(withTitle: "Quit", action: #selector(NSApplication.terminate(_:)), keyEquivalent: "q") - menu.items.forEach { $0.target = self } item.menu = menu statusItem = item panel = ReplyPanel(model: model) hotkey = HotkeyMonitor { [weak self] action in Task { @MainActor in + guard let self else { return } switch action { - case .draft(let typed): self?.draftFromSelection(typedTrigger: typed) + case .draft(let typed): + if typed && !self.bareKeys { return } + self.draftFromSelection(typedTrigger: typed) case .regenerate(let typed): + if typed && !self.bareKeys { return } if typed { SelectionReader.deleteTypedTriggerIfEditing() } - self?.regenerate() + self.regenerate() } } } @@ -58,10 +77,11 @@ final class AppDelegate: NSObject, NSApplicationDelegate { /// 0: a new sampled reply for the last post, replacing the box contents (or a first draft if there is none yet). func regenerate() { - guard !model.post.isEmpty, model.phase != .drafting else { return draftFromSelection(typedTrigger: false) } + if model.phase == .drafting { return } // a draft is in flight; ignore the key + guard !model.rawPost.isEmpty else { return draftFromSelection(typedTrigger: false) } let previous = model.sourceApplication Task { @MainActor in - await model.draft(post: model.post) + await model.draft(post: model.rawPost) guard case .done = model.phase else { return panel?.present() ?? () } if autoInsert, previous != nil { model.sourceApplication = previous @@ -77,7 +97,13 @@ final class AppDelegate: NSObject, NSApplicationDelegate { autoInsertItem?.state = autoInsert ? .on : .off } + @objc func toggleBareKeys() { + bareKeys.toggle() + bareKeysItem?.state = bareKeys ? .on : .off + } + func draftFromSelection(typedTrigger: Bool) { + if model.phase == .drafting { return } let previous = NSWorkspace.shared.frontmostApplication Task { @MainActor in let context = await SelectionReader.context(typedTrigger: typedTrigger)