import type { FeatureRouting, SwitchboardOptions } from "./switchboard-types" /** * Applied to every request. * * Both fields are set defensively rather than left to the key's defaults. An API key * minted for a different tool can carry its own `category`/`prefer_free` defaults, * and anything this app leaves unset silently inherits them. Verified against the * gateway: an unset request inherited `category: "complex_coding"` and free-model * routing from the key, which sent drink prompts to a free coding model. * * `prefer_free` is off because every call site here parses JSON out of the response * and free models are the least reliable at emitting it, and because the recommend * and bartender features send personal drink history and home bar inventory - the * gateway guide notes free endpoints may log or train on prompts. */ const BASE: SwitchboardOptions = { prefer_free: false, peer_review: false, } /** * Timeouts are generous for the same reason token budgets are: a routed reasoning * model is slow. A plain drink search measured ~42s end to end, and latency varies * with which model the router picks, so these are sized well above the typical case. * * Token budgets are deliberately generous. The router may pick a reasoning model, * and reasoning tokens are drawn from the same `max_tokens` budget as the answer. * Verified: an identical request returned `content: null` at max_tokens 512 (the * whole budget went to reasoning) and correct JSON at 4096. Treat ~2048 as the floor * for anything that must return content, not as a cost lever. */ export const FEATURE_ROUTING = { // Vision. The only place `tier` earns its keep: these run once per deliberate user // action and their output prefills a form, so a miss costs the user typing. On a // test label, `frontier` read name/type/subType/abv correctly (~2.8s, $0.003) while // unconstrained routing picked a small model that got the name but missed the ABV, // and free routing returned only the name. menuExtraction: { feature: "menu.extract", switchboard: { ...BASE, category: "general", tier: "frontier" }, timeoutMs: 240_000, maxTokens: 4096, }, labelExtraction: { feature: "label.extract", switchboard: { ...BASE, category: "general", tier: "frontier" }, timeoutMs: 180_000, maxTokens: 4096, }, // Short interactive lookups. Deliberately no `tier` - letting the classifier choose // beat both alternatives by a wide margin when measured. `tier: "cheap"` pinned a // slow reasoning model (42-180s, timed out twice in five trials and truncated its // JSON once), and `tier: "frontier"` escalated as far as Claude Opus at $0.02 a // call. Unconstrained, the same prompts landed on a small fast model in well under // a second for a few hundredths of a cent. drinkSearch: { feature: "drink.search", switchboard: { ...BASE, category: "simple" }, timeoutMs: 180_000, maxTokens: 3072, }, barcodeLookup: { feature: "bar.barcode", switchboard: { ...BASE, category: "simple" }, timeoutMs: 120_000, maxTokens: 2048, }, // General text. menuRecommend: { feature: "menu.recommend", switchboard: { ...BASE, category: "general" }, timeoutMs: 180_000, maxTokens: 4096, }, bartenderSuggest: { feature: "bartender.suggest", switchboard: { ...BASE, category: "general" }, timeoutMs: 240_000, maxTokens: 4096, }, bartenderRecreate: { feature: "bartender.recreate", switchboard: { ...BASE, category: "general" }, timeoutMs: 180_000, maxTokens: 3072, }, recommendSuggest: { feature: "recommend.suggest", switchboard: { ...BASE, category: "general" }, timeoutMs: 180_000, maxTokens: 4096, }, recommendSimilar: { feature: "recommend.similar", switchboard: { ...BASE, category: "general" }, timeoutMs: 180_000, maxTokens: 4096, }, // Sends the user's whole rating history, and the result is persisted and then // re-read by recommend/suggest and recommend/similar - a bad profile poisons both // until it is regenerated, so this one does not get a cost lever. flavorProfile: { feature: "recommend.profile", switchboard: { ...BASE, category: "business" }, timeoutMs: 240_000, maxTokens: 4096, }, } satisfies Record