From 4a62d4fbeba3afc2b6715f02c0051a5d2b29792d Mon Sep 17 00:00:00 2001 From: Jaedon Munton Date: Wed, 5 Aug 2026 10:35:49 +0100 Subject: [PATCH 1/2] fix:adds an amended intro message and contact info to the models page --- src/lib/model-artifacts.ts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/lib/model-artifacts.ts b/src/lib/model-artifacts.ts index ad289b7..7b64c43 100644 --- a/src/lib/model-artifacts.ts +++ b/src/lib/model-artifacts.ts @@ -207,6 +207,8 @@ export async function getModelsIndexMarkdown(): Promise { export function renderModelsIndexIntroMarkdown(): string { return `Doubleword Batch API is priced per model based on token usage. Costs are calculated separately for input tokens (the content you send) and output tokens (the content generated by the model). +We can offer custom pricing for bulk discounts, large workloads, and dedicated deployments - reach out to [hello@doubleword.ai](mailto:hello@doubleword.ai). + The table below outlines the models we have available and their pricing per 1M tokens. If you are interested in understanding pricing for a model not listed below or if you'd like to request a new model - please reach out to support@doubleword.ai. :::info{title="Prompt caching"} From 284bf419e755264284a31b286991b75b11778e8a Mon Sep 17 00:00:00 2001 From: Jaedon Munton Date: Thu, 6 Aug 2026 18:10:51 +0100 Subject: [PATCH 2/2] fix: add async TTFT guarantees warning to models index (hardcoded) --- src/lib/model-artifacts.ts | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/src/lib/model-artifacts.ts b/src/lib/model-artifacts.ts index 7b64c43..5c2960e 100644 --- a/src/lib/model-artifacts.ts +++ b/src/lib/model-artifacts.ts @@ -214,6 +214,13 @@ The table below outlines the models we have available and their pricing per 1M t :::info{title="Prompt caching"} Prompt-caching availability and rates are model-specific. Use **Cache read** to compare each supported model's reduced cached-input price with its standard input price. See the [prompt caching guide](/inference-api/prompt-caching) for setup, TTLs, and write pricing. ::: + +:::warning{title="Async TTFT guarantees"} +We target a Time to First Token (TTFT) of under one minute for individual model calls, with two exceptions: + +- **Tier availability:** the one-minute target does not apply to the flex tier for models that aren't also available on the realtime tier — currently mostly OCR models (as of Aug 2026), and subject to change as we expand the catalog. +- **High-volume bursts:** during large concurrent bursts to the same model and tier, the target applies to the first call, not the whole batch; remaining start times scale with queue depth. +::: `; }