[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"sanity-p2W5_RLV1ZzlAphdE-zajUTNVVOUYD5I2wlAyTijVCc":3,"sanity-2aSN8Gzmus9I-AQTZFf57dSrGxCF2NWbbsL2t6VFbeU":1572},{"data":4,"sourceMap":-1},{"latestPodcast":5,"latestReleases":14,"post":39,"recent":1547},[6],{"_id":7,"publishedAt":8,"slug":9,"sponsored":12,"title":13},"8e6c3a8a-3d27-44c8-be90-0b74d1090a4a","2026-10-06T17:00:00.000Z",{"_type":10,"current":11},"slug","tales-from-the-2026-developer-survey-results",null,"Tales from the 2026 Developer Survey results",[15,21,27,33],{"_id":16,"publishedAt":17,"slug":18,"title":20},"7a2d88e5-ee53-4f7c-a46e-47bed1cebadf","2026-09-30T16:00:00.000Z",{"_type":10,"current":19},"anyone-can-start-building-verified-knowledge-with-stack-internal","Anyone can start building verified knowledge with Stack Internal",{"_id":22,"publishedAt":23,"slug":24,"title":26},"12c6a8a7-135f-401f-ac1a-1c26c33e69c0","2026-09-03T16:00:00.000Z",{"_type":10,"current":25},"security-control-and-accessibility-si-2026-6","Elevating security, control, and accessibility: Stack Internal 2026.6",{"_id":28,"publishedAt":29,"slug":30,"title":32},"adcf1bca-3295-4ac5-9b3c-23f337974190","2026-07-30T15:10:00.000Z",{"_type":10,"current":31},"introducing-stack-internal-new-platform-experience","Your trusted knowledge layer: Introducing Stack Internal's new platform experience",{"_id":34,"publishedAt":35,"slug":36,"title":38},"eb5b66eb-9410-4329-83bb-22bbff39402a","2026-04-28T13:00:00.000Z",{"_type":10,"current":37},"turn-scattered-knowledge-into-trusted-intelligence","Turning scattered knowledge into trusted intelligence: Stack Internal 2026.3",{"_createdAt":40,"_id":41,"_rev":42,"_type":43,"_updatedAt":44,"author":45,"body":56,"comments":1456,"dateUrl":1457,"excerpt":1458,"image":1459,"product":12,"publishedAt":1465,"seo":1466,"slug":1469,"sponsored":12,"tags":1471,"title":1546,"visible":1456},"2026-10-08T20:08:46Z","004e0789-dc2d-496b-a8af-b6e6aefde5f7","OFkij2n5IGXx7uXmd9AtlL","blogPost","2026-10-08T20:16:38Z",[46],{"_createdAt":47,"_id":48,"_rev":49,"_type":50,"_updatedAt":51,"employee":52,"name":53,"slug":54},"2026-10-07T20:10:11Z","c7434ebf-211a-49d5-a564-22a91d63ed1e","9fZfA2bz90M2vN1VVZP5Ov","blogAuthor","2026-10-07T20:12:01Z","none","Varun Jindal",{"_type":10,"current":55},"varun-jindal",[57,94,123,132,156,164,204,207,263,271,279,321,349,369,392,400,415,418,426,434,442,445,492,515,523,531,539,562,566,582,590,593,601,609,625,628,676,684,692,712,720,733,745,757,769,785,788,796,804,812,820,823,826,834,837,897,921,929,937,940,1004,1012,1052,1060,1063,1071,1079,1087,1095,1135,1155,1158,1166,1190,1198,1214,1262,1278,1281,1304,1312,1320,1323,1347,1394,1402,1432,1448],{"_key":58,"_type":59,"children":60,"markDefs":92,"style":93},"8c3674bfd56a","block",[61,66,71,75,79,83,88],{"_key":62,"_type":63,"marks":64,"text":65},"eac00e21cfc1","span",[],"This is Level 5 of a six-level maturity model for running LLM systems in production. The earlier levels got the system ",{"_key":67,"_type":63,"marks":68,"text":70},"6d32fe0f5e6b",[69],"em","working",{"_key":72,"_type":63,"marks":73,"text":74},"a396df0c2fbd",[]," and ",{"_key":76,"_type":63,"marks":77,"text":78},"89a5f7400fe6",[69],"correct",{"_key":80,"_type":63,"marks":81,"text":82},"c66709b28750",[],". Level 5 is about ",{"_key":84,"_type":63,"marks":85,"text":87},"bdb67b7938c4",[86],"strong","operability",{"_key":89,"_type":63,"marks":90,"text":91},"50db93d813d9",[],": can you actually run this thing day to day, see what it’s deciding, control what it costs, survive a provider outage, stop it in seconds when it misbehaves — and is the platform underneath shaped to support all of that?",[],"normal",{"_key":95,"_type":59,"children":96,"markDefs":122,"style":93},"740b6d7fa4e7",[97,101,105,109,113,118],{"_key":98,"_type":63,"marks":99,"text":100},"88f232e86489",[],"A traditional service that goes wrong returns a 500. An LLM system that goes wrong can take ",{"_key":102,"_type":63,"marks":103,"text":104},"5448c566818e",[69],"wrong actions at scale, fast",{"_key":106,"_type":63,"marks":107,"text":108},"50dd216dc3e0",[],". That asymmetry is why operability isn’t a nice-to-have here — it’s the difference between a system you can run and one you have to hope about. This post is the concrete spec for the operability layer: the observability signals, the cost controls, the routing and failover, the kill switch, the identity model, and the infrastructure spine that holds it together. It’s the longest post in the series, because it’s the layer with the most moving parts. The unifying idea, though, is small",{"_key":110,"_type":63,"marks":111,"text":112},"03ef36b60949",[86],": one chokepoint your model traffic flows through, one ",{"_key":114,"_type":63,"marks":115,"text":117},"0660fc349fde",[116,86],"code","decision_id",{"_key":119,"_type":63,"marks":120,"text":121},"78af9a224315",[86]," that threads everything, and a domain that never knows which vendor answered.",[],{"_key":124,"_type":59,"children":125,"markDefs":130,"style":131},"a3c14d4029ac",[126],{"_key":127,"_type":63,"marks":128,"text":129},"0dfd427568ec",[],"Observe the decision, not just the service",[],"h3",{"_key":133,"_type":59,"children":134,"markDefs":155,"style":93},"30915fe57ffb",[135,139,143,147,151],{"_key":136,"_type":63,"marks":137,"text":138},"e59cdf683f4c",[],"Standard observability — request rate, latency, error rate, CPU — tells you the ",{"_key":140,"_type":63,"marks":141,"text":142},"606c76ec23f8",[69],"service",{"_key":144,"_type":63,"marks":145,"text":146},"a82d48f411ee",[]," is up. It tells you nothing about whether the ",{"_key":148,"_type":63,"marks":149,"text":150},"3fb7e24d7ee6",[69],"agent",{"_key":152,"_type":63,"marks":153,"text":154},"33cc29459ed6",[]," is doing its job well. You need a second layer of signals specific to AI, plus the schemas and thresholds to make them actionable.",[],{"_key":157,"_type":59,"children":158,"markDefs":163,"style":93},"05e2d3700ee0",[159],{"_key":160,"_type":63,"marks":161,"text":162},"61266c793cc9",[86],"The model: RED + a decision layer",[],{"_key":165,"_type":59,"children":166,"markDefs":203,"style":93},"a80863ce119b",[167,171,175,179,183,187,191,195,199],{"_key":168,"_type":63,"marks":169,"text":170},"e0212bf2f319",[],"Keep your usual RED metrics (Rate, Errors, Duration) for the service. Add a ",{"_key":172,"_type":63,"marks":173,"text":174},"9977681b1d50",[86],"decision layer",{"_key":176,"_type":63,"marks":177,"text":178},"b82a7e21bb76",[]," that treats every agent decision as a first-class, measurable event. Names below follow Prometheus conventions (`_total` counters, ",{"_key":180,"_type":63,"marks":181,"text":182},"695facaf3c06",[116],"_bucket",{"_key":184,"_type":63,"marks":185,"text":186},"a74d3bf15084",[]," histograms), a",{"_key":188,"_type":63,"marks":189,"text":190},"deafdb56d1aa",[86],"nd every series carries ",{"_key":192,"_type":63,"marks":193,"text":194},"d1722742005e",[116,86],"tenant_id and ",{"_key":196,"_type":63,"marks":197,"text":198},"9efed5f23a59",[86],"capability`",{"_key":200,"_type":63,"marks":201,"text":202},"7c287243570e",[]," as labels (omitted in the table for brevity — assume them everywhere). One caveat: a per-`tenant_id` label on histograms is a cardinality bomb past a few hundred tenants — keep tenant counts bounded, or move per-tenant rollups to recording rules \u002F exemplars beyond that.",[],{"_key":205,"_type":116,"code":206,"markDefs":12},"3a6403eee331","metric                             type       key labels                                                              what it answers\n---------------------------------  ---------  ----------------------------------------------------------------------  -------------------------------------------------\n`agent_decisions_total`            counter    `outcome` (auto \u002F hitl_recommended \u002F hitl_required \u002F reject \u002F abstain)  volume + the auto\u002Fhuman split\n`agent_auto_execution_ratio`       gauge      —                                                                       % handled without a human\n`agent_decision_confidence`        histogram  —                                                                       distribution of composed confidence\n`agent_decision_duration_seconds`  histogram  `node`                                                                  latency per graph node\n`guardrail_blocks_total`           counter    `layer` (input \u002F output \u002F pii), `rule`                                  what's being blocked, where\n`judge_invocations_total`          counter    —                                                                       how often the judge ran (the ratio's denominator)\n`judge_disagreements_total`        counter    —                                                                       judge overruled the primary\n`llm_tokens_total`                 counter    `direction` (in \u002F out), `model`                                         token consumption\n`llm_cost_usd_total`               counter    `model`                                                                 spend, rolled up by tenant\u002Fcapability\n`llm_call_duration_seconds`        histogram  `model`, `outcome`                                                      model latency, separate from service\n`shadow_eval_pass_ratio`           gauge      `slice`                                                                 live quality on sampled prod traffic\n`human_override_rate`              gauge      —                                ",{"_key":208,"_type":59,"children":209,"markDefs":262,"style":93},"544502d28294",[210,214,218,222,226,230,234,238,242,246,250,254,258],{"_key":211,"_type":63,"marks":212,"text":213},"aa4a69ac1abe",[],"Four families fall out of that: ",{"_key":215,"_type":63,"marks":216,"text":217},"b80e056f69b0",[86],"decision",{"_key":219,"_type":63,"marks":220,"text":221},"973047bf2473",[]," (volume, split, confidence, latency), ",{"_key":223,"_type":63,"marks":224,"text":225},"be8aa47dd703",[86],"safety",{"_key":227,"_type":63,"marks":228,"text":229},"30210b16b50f",[]," (guardrail blocks, judge disagreement), ",{"_key":231,"_type":63,"marks":232,"text":233},"e4ae5df9bcc8",[86],"cost\u002Fperf",{"_key":235,"_type":63,"marks":236,"text":237},"177ed34951c1",[]," (tokens, USD, model latency), ",{"_key":239,"_type":63,"marks":240,"text":241},"c4c447b96b63",[86],"quality",{"_key":243,"_type":63,"marks":244,"text":245},"3fefe091e467",[]," (shadow eval, overrides). If you only instrument four things, make them ",{"_key":247,"_type":63,"marks":248,"text":249},"aab2e3e481d9",[116],"auto_execution_ratio, ",{"_key":251,"_type":63,"marks":252,"text":253},"4e85b701bf65",[],"guardrail_blocks_total`, ",{"_key":255,"_type":63,"marks":256,"text":257},"3be830917939",[116],"llm_cost_usd_total, and ",{"_key":259,"_type":63,"marks":260,"text":261},"dfe3af7ab00d",[],"human_override_rate` — they cover behavior, safety, money, and quality.",[],{"_key":264,"_type":59,"children":265,"markDefs":270,"style":93},"7e703b3f8c8b",[266],{"_key":267,"_type":63,"marks":268,"text":269},"9f2fb374bec4",[86],"Structured logging: three tiers",[],{"_key":272,"_type":59,"children":273,"markDefs":278,"style":93},"dbc93f8ddf02",[274],{"_key":275,"_type":63,"marks":276,"text":277},"7e2d2dcfe56b",[],"Don’t dump everything into one log stream — tier by sensitivity, because some of this is regulated data and most alerts only need tier 1.",[],{"_key":280,"_type":59,"children":281,"level":318,"listItem":319,"markDefs":320,"style":93},"14c23bbf9d8b",[282,286,290,294,298,302,306,310,314],{"_key":283,"_type":63,"marks":284,"text":285},"67ec5b5ac1f6",[86],"Tier 1 — operational",{"_key":287,"_type":63,"marks":288,"text":289},"5cf69d28c3cc",[]," (safe anywhere, high volume): timestamp, level, ",{"_key":291,"_type":63,"marks":292,"text":293},"ecb9729c80b0",[116],"decision_id, ",{"_key":295,"_type":63,"marks":296,"text":297},"7d96c1727996",[],"tenant_id`, ",{"_key":299,"_type":63,"marks":300,"text":301},"5e726cacee70",[116],"capability, ",{"_key":303,"_type":63,"marks":304,"text":305},"57db7f0f0582",[],"node`, ",{"_key":307,"_type":63,"marks":308,"text":309},"718f5248e8dc",[116],"duration_ms",{"_key":311,"_type":63,"marks":312,"text":313},"584894c0bf47",[],", ",{"_key":315,"_type":63,"marks":316,"text":317},"c6ad45b5e997",[116],"outcome. This is what your alerts query.",1,"bullet",[],{"_key":322,"_type":59,"children":323,"level":318,"listItem":319,"markDefs":348,"style":93},"6e0806614dda",[324,328,332,336,340,344],{"_key":325,"_type":63,"marks":326,"text":327},"c7f425b9accc",[86],"Tier 2 — decision metadata",{"_key":329,"_type":63,"marks":330,"text":331},"ca24dc031389",[]," (inside your trust boundary; the audit-adjacent record): ",{"_key":333,"_type":63,"marks":334,"text":335},"d561872d8d94",[116],"model, ",{"_key":337,"_type":63,"marks":338,"text":339},"2d7e676dd360",[],"prompt_version`, ",{"_key":341,"_type":63,"marks":342,"text":343},"816f62bd1cbc",[116],"composed_confidence, routing decision, judge result, guardrail actions, an ",{"_key":345,"_type":63,"marks":346,"text":347},"ea0dbd661347",[],"inputs_hash`, and a redacted PII-free summary.",[],{"_key":350,"_type":59,"children":351,"level":318,"listItem":319,"markDefs":368,"style":93},"348f17cadc85",[352,356,360,364],{"_key":353,"_type":63,"marks":354,"text":355},"4f0bc763d5cc",[86],"Tier 3 — regulated\u002Fraw",{"_key":357,"_type":63,"marks":358,"text":359},"22a40950201b",[]," (encrypted, access-controlled, ",{"_key":361,"_type":63,"marks":362,"text":363},"a953c7c00277",[86],"never",{"_key":365,"_type":63,"marks":366,"text":367},"1b7445f37c44",[]," in your general log store): the sensitive payload, if you retain it at all — usually you store a hash + redacted summary in tier 2 and skip tier 3 entirely.",[],{"_key":370,"_type":59,"children":371,"markDefs":391,"style":93},"aa36e44f56f6",[372,376,380,384,387],{"_key":373,"_type":63,"marks":374,"text":375},"d75cabfa98e5",[],"The rule: ",{"_key":377,"_type":63,"marks":378,"text":379},"238048540721",[86],"tier 1 and 2 are queryable by engineers; tier 3 is a vault.",{"_key":381,"_type":63,"marks":382,"text":383},"dc5446416d8b",[]," A ",{"_key":385,"_type":63,"marks":386,"text":117},"68f9538fd62c",[116],{"_key":388,"_type":63,"marks":389,"text":390},"ba918c5cc3a5",[]," threads all three — and the trace below — so you can reconstruct any single decision end-to-end.",[],{"_key":393,"_type":59,"children":394,"markDefs":399,"style":93},"dcf3ab5cb513",[395],{"_key":396,"_type":63,"marks":397,"text":398},"d17697e7b92b",[86],"Trace the decision graph",[],{"_key":401,"_type":59,"children":402,"markDefs":414,"style":93},"e99b25483683",[403,407,410],{"_key":404,"_type":63,"marks":405,"text":406},"86c9fd97dcb0",[],"A normal trace shows the HTTP request across services. An ",{"_key":408,"_type":63,"marks":409,"text":150},"ca0b4f3cb589",[69],{"_key":411,"_type":63,"marks":412,"text":413},"9dcbf000acc0",[]," trace should also span the internal graph so you can see which step failed and how long each took:",[],{"_key":416,"_type":116,"code":417,"markDefs":12},"fdc2039fe61f","span: agent.decision            attrs: decision_id, tenant_id, capability, outcome, confidence\n ├─ span: entry                 attrs: identity, auth_type\n ├─ span: context_load          attrs: slices_loaded, cache_hit\n ├─ span: llm_decision          attrs: model, prompt_version, tokens_in, tokens_out, cost_usd\n ├─ span: output_guardrail      attrs: blocks[], pii_actions[]\n ├─ span: judge                 attrs: ran, model, agreed     (only when sampled in)\n ├─ span: routing               attrs: decision=auto|hitl|reject, threshold\n └─ span: exit                  attrs: ledger_entry_id",{"_key":419,"_type":59,"children":420,"markDefs":425,"style":93},"f62af3b71d96",[421],{"_key":422,"_type":63,"marks":423,"text":424},"89f4e7d7c377",[],"Now “why did decision X take 4 seconds \u002F get rejected?” is a single trace lookup.",[],{"_key":427,"_type":59,"children":428,"markDefs":433,"style":93},"d09b86e30672",[429],{"_key":430,"_type":63,"marks":431,"text":432},"9f163f185adc",[86],"Alert on symptoms, with real thresholds",[],{"_key":435,"_type":59,"children":436,"markDefs":441,"style":93},"4b14b3fb77cc",[437],{"_key":438,"_type":63,"marks":439,"text":440},"083fba7563fa",[],"Alert on what hurts the user or business, not on causes. A starter set:",[],{"_key":443,"_type":116,"code":444,"markDefs":12},"887714ce5c44","alert                  condition (example)                                                             severity\n---------------------  ------------------------------------------------------------------------------  ------------\nAutoExecutionSwing     `auto_execution_ratio` moves >15% vs 7-day baseline                             warning\nGuardrailBlockSpike    `rate(guardrail_blocks_total[5m])` > 3× trailing hr                             critical\nJudgeDisagreementHigh  rate(judge_disagreements_total[1h]) \u002F rate(judge_invocations_total[1h]) > 0.15  critical\nCostCeiling            `increase(llm_cost_usd_total[1h])` > budget\u002F24                                  warning→page\nShadowEvalDrop         `shadow_eval_pass_ratio` \u003C 0.95                                                 critical\nModelLatencyP99        `llm_call_duration_seconds` p99 > 8s for 10m         ",{"_key":446,"_type":59,"children":447,"markDefs":491,"style":93},"4b51afac42a8",[448,452,456,460,464,468,471,475,479,483,487],{"_key":449,"_type":63,"marks":450,"text":451},"165521ef5daf",[],"Page on the safety and quality ones; the rest open a ticket. And set SLOs that mean something for a decision system: ",{"_key":453,"_type":63,"marks":454,"text":455},"be14eed6e98d",[86],"decision availability",{"_key":457,"_type":63,"marks":458,"text":459},"c3a4feb222cd",[]," (≥ 99.9% of decisions return a proposal — ",{"_key":461,"_type":63,"marks":462,"text":463},"12a4a1074f27",[69],"failing to decide",{"_key":465,"_type":63,"marks":466,"text":467},"9350423497a3",[]," is the real outage, not a 5xx), ",{"_key":469,"_type":63,"marks":470,"text":241},"faa514a9633f",[86],{"_key":472,"_type":63,"marks":473,"text":474},"3f7a9c5e11b6",[]," (shadow-eval ≥ baseline − 2%, rolling 7d), ",{"_key":476,"_type":63,"marks":477,"text":478},"1a5e31c90a34",[86],"latency",{"_key":480,"_type":63,"marks":481,"text":482},"65ffd2a0eada",[]," (p95 excluding human time under your interaction budget), and ",{"_key":484,"_type":63,"marks":485,"text":486},"9967be20083c",[86],"cost",{"_key":488,"_type":63,"marks":489,"text":490},"19e56f974b0d",[]," (USD per 1k decisions within ±20% of plan).",[],{"_key":493,"_type":59,"children":494,"markDefs":514,"style":93},"c9768f0206e1",[495,499,503,507,510],{"_key":496,"_type":63,"marks":497,"text":498},"c68dec9ed1eb",[],"Three observability anti-patterns worth naming: one log stream for everything (tier-3 leaks into searchable logs — a compliance problem), self-reported model confidence as a metric (it’s miscalibrated; track ",{"_key":500,"_type":63,"marks":501,"text":502},"7401102eb14b",[69],"composed",{"_key":504,"_type":63,"marks":505,"text":506},"9b2c6de83d25",[]," confidence and validate against outcomes), and no ",{"_key":508,"_type":63,"marks":509,"text":117},"a65ffbb77349",[116],{"_key":511,"_type":63,"marks":512,"text":513},"c779d1df12d6",[]," (every investigation becomes archaeology).",[],{"_key":516,"_type":59,"children":517,"markDefs":522,"style":131},"669da971e454",[518],{"_key":519,"_type":63,"marks":520,"text":521},"a8f4486cf091",[],"Control the cost at the chokepoint",[],{"_key":524,"_type":59,"children":525,"markDefs":530,"style":93},"fd6248488374",[526],{"_key":527,"_type":63,"marks":528,"text":529},"5f2afcc17bc5",[],"LLM cost has a nasty property: it’s invisible until the invoice, and it scales with things you’re not watching — tokens per call, calls per request, retries, a chatty prompt someone added. Teams discover their unit economics are upside down only after they’ve shipped. The fix isn’t a cheaper model; it’s a handful of structural controls that make cost observable and bounded — and they all hang off the same chokepoint.",[],{"_key":532,"_type":59,"children":533,"markDefs":538,"style":93},"b9ff0274b813",[534],{"_key":535,"_type":63,"marks":536,"text":537},"d982e99d96f9",[86],"Measure where you enforce",[],{"_key":540,"_type":59,"children":541,"markDefs":561,"style":93},"17fbb4c3a753",[542,546,550,553,557],{"_key":543,"_type":63,"marks":544,"text":545},"f09fc6bf9ed9",[],"You can’t control what you can’t see, and you can’t see spend when model calls happen all over the codebase. Route every call through one gateway. That component is where you ",{"_key":547,"_type":63,"marks":548,"text":549},"ba8a22ce5fc2",[69],"measure",{"_key":551,"_type":63,"marks":552,"text":74},"64ce1666d137",[],{"_key":554,"_type":63,"marks":555,"text":556},"042d0cec887a",[69],"enforce",{"_key":558,"_type":63,"marks":559,"text":560},"a748a7905943",[],":",[],{"_key":563,"_type":116,"code":564,"language":565,"markDefs":12},"946ac034b05d","def complete(self, req: ChatRequest) -> ChatResponse:\n    rate_limiter.check(req.tenant_id, est_tokens(req))     # enforce (estimate now; settle actuals below)\n    resp = self._adapter.complete(req)\n    metrics.incr(\"llm_tokens_total\", resp.tokens_in,  direction=\"in\",  model=req.model, tenant_id=req.tenant_id)\n    metrics.incr(\"llm_tokens_total\", resp.tokens_out, direction=\"out\", model=req.model, tenant_id=req.tenant_id)\n    metrics.incr(\"llm_cost_usd_total\", cost(req.model, resp), model=req.model, tenant_id=req.tenant_id, capability=req.capability)\n    return resp","python",{"_key":567,"_type":59,"children":568,"markDefs":581,"style":93},"20accc7c1373",[569,573,577],{"_key":570,"_type":63,"marks":571,"text":572},"30533e5cb4bc",[],"Now cost is attributable per tenant × capability × model — which is how you find ",{"_key":574,"_type":63,"marks":575,"text":576},"747429228a5b",[69],"where",{"_key":578,"_type":63,"marks":579,"text":580},"55782a3ff16e",[]," the money goes (usually one chatty capability or one oversized prompt) instead of vaguely “using less AI.”",[],{"_key":583,"_type":59,"children":584,"markDefs":589,"style":93},"86304a0eef26",[585],{"_key":586,"_type":63,"marks":587,"text":588},"3844f9ea73be",[86],"The levers, in order of payoff",[],{"_key":591,"_type":116,"code":592,"markDefs":12},"97bc896a8adc","lever                     mechanism                                                     typical impact\n------------------------  ------------------------------------------------------------  ----------------------------------------------------------------------\n**don't call the model**  route easy\u002Fdeterministic cases through code                   often the biggest — a lot of \"AI cost\" is the model doing a rule's job\n**batch**                 one call for many items vs N calls                            fewer round-trips, cheaper per item\n**cache**                 memoize deterministic results (embeddings, repeated lookups)  the cheapest call is the one you skip\n**right-size the model**  cheap model for simple decisions, capable for hard ones       big — most traffic is simple, most cost is the premium model\n**trim the prompt**       load only needed context; kill \"just in case\" preamble        recurring tax paid on *every* call",{"_key":594,"_type":59,"children":595,"markDefs":600,"style":93},"53eb90c470ea",[596],{"_key":597,"_type":63,"marks":598,"text":599},"e8b4bff54abf",[],"Right-sizing is worth making explicit, because it feeds directly into routing (next section): if a cheap model passes evals for a slice, route it there — 5–20× cheaper per call is common. Make the math visible (`cost = tokens_in\u002F1k × price_in + tokens_out\u002F1k × price_out`) rather than guessing.",[],{"_key":602,"_type":59,"children":603,"markDefs":608,"style":93},"bed141b4cc60",[604],{"_key":605,"_type":63,"marks":606,"text":607},"d10fc26002a4",[86],"Budgets, rate limits, and the silent multipliers",[],{"_key":610,"_type":59,"children":611,"markDefs":624,"style":93},"d45097b10efb",[612,616,620],{"_key":613,"_type":63,"marks":614,"text":615},"1ddd96e6b742",[],"Cap spend and rate per tenant, with state ",{"_key":617,"_type":63,"marks":618,"text":619},"27dbe86a0e23",[86],"shared across replicas",{"_key":621,"_type":63,"marks":622,"text":623},"74a58ff105ce",[]," — a stateless worker pool can’t enforce a per-tenant limit from local memory, since each replica would allow its own full fraction:",[],{"_key":626,"_type":116,"code":627,"language":565,"markDefs":12},"db05caea6f3c","RATE = { \"default\": { \"tokens_per_min\": 200_000, \"usd_per_day\": 50 } }  # derive from real prices\n\ndef check(tenant_id, est_tokens):\n    limits = RATE_FOR(tenant_id)\n    # add the ESTIMATED token count, not 1 — counting calls against a token budget never trips\n    if redis.incr_window(f\"tok:{tenant_id}\", by=est_tokens, ttl=60) > limits[\"tokens_per_min\"]:\n        raise RateLimited(tenant_id)\n    spent = redis.get_float(f\"usd:{tenant_id}:{today()}\")    # written by the gateway's post-call cost path\n    if spent > limits[\"usd_per_day\"]:        raise BudgetExceeded(tenant_id)         # 100%: hard stop\n    if spent > 0.8 * limits[\"usd_per_day\"]:  alert(tenant_id, \"80% of daily budget\") # 80%: warn",{"_key":629,"_type":59,"children":630,"markDefs":675,"style":93},"6daa6226c409",[631,635,639,643,647,651,655,659,663,667,671],{"_key":632,"_type":63,"marks":633,"text":634},"993800a1bbff",[],"Also set a ",{"_key":636,"_type":63,"marks":637,"text":638},"ade14f0deccb",[86],"platform-wide ceiling",{"_key":640,"_type":63,"marks":641,"text":642},"cbb24889f8c3",[]," — N tenants × per-tenant cap has no aggregate bound otherwise. And watch the two things that quietly 10× a bill: ",{"_key":644,"_type":63,"marks":645,"text":646},"5ba18a69ef5b",[86],"retry storms",{"_key":648,"_type":63,"marks":649,"text":650},"39e62088c6f0",[]," (a flaky validation re-calling the model) and ",{"_key":652,"_type":63,"marks":653,"text":654},"405f97abf481",[86],"unbounded tool loops",{"_key":656,"_type":63,"marks":657,"text":658},"faae5ade1883",[]," (an agent that keeps going). Cap both, and emit ",{"_key":660,"_type":63,"marks":661,"text":662},"0eb03fc55a76",[116],"llm_retries_total{reason}",{"_key":664,"_type":63,"marks":665,"text":666},"354a4515fb28",[]," so a storm is visible. LLM cost isn’t fundamentally high; it’s fundamentally ",{"_key":668,"_type":63,"marks":669,"text":670},"9e2911d51af3",[69],"unmonitored",{"_key":672,"_type":63,"marks":673,"text":674},"a89759bb4532",[],". Monitor it at the chokepoint and the bill stays proportional to value.",[],{"_key":677,"_type":59,"children":678,"markDefs":683,"style":131},"155681b289fb",[679],{"_key":680,"_type":63,"marks":681,"text":682},"b7663896dd7c",[],"Make it fast — without losing quality",[],{"_key":685,"_type":59,"children":686,"markDefs":691,"style":93},"cf8898f77928",[687],{"_key":688,"_type":63,"marks":689,"text":690},"c1ab94d579f8",[],"Cost and latency share a root cause: a pipeline that takes minutes usually isn’t slow because of the model. It’s an O(n²) loop, a per-record round-trip, or serial stages — the same performance bugs that always plagued data pipelines, now wrapped around an LLM.",[],{"_key":693,"_type":59,"children":694,"markDefs":711,"style":93},"71307a9a3187",[695,699,703,707],{"_key":696,"_type":63,"marks":697,"text":698},"b71df1ea8f53",[86],"Rule 0: lock a quality baseline before you touch anything.",{"_key":700,"_type":63,"marks":701,"text":702},"87663a837def",[]," Performance work is dangerous because the fastest version is often subtly less correct. Capture a baseline — representative dataset, current outputs, key quality metrics — and re-run it after every change. No optimization is accepted that regresses the baseline. ",{"_key":704,"_type":63,"marks":705,"text":706},"853501ef166a",[86],"Rule 1: profile, don’t guess.",{"_key":708,"_type":63,"marks":709,"text":710},"08f5486174ea",[]," A pipeline matching 10,000 × 10,000 items “feels” model-bound, but the profile shows 90% of the time in a nested loop doing 100,000,000 comparisons. The model was never the problem.",[],{"_key":713,"_type":59,"children":714,"markDefs":719,"style":93},"a52c9b813abe",[715],{"_key":716,"_type":63,"marks":717,"text":718},"20b9efcdbb8d",[],"The usual suspects, in order of payoff:",[],{"_key":721,"_type":59,"children":722,"level":318,"listItem":731,"markDefs":732,"style":93},"fc8ce2440f52",[723,727],{"_key":724,"_type":63,"marks":725,"text":726},"bb2b17bcb7a0",[86],"The O(n²) match loop",{"_key":728,"_type":63,"marks":729,"text":730},"8cb62c025375",[]," → index once into a hash\u002Fkeyed join, then look up: ~100M comparisons become ~20k operations.","number",[],{"_key":734,"_type":59,"children":735,"level":318,"listItem":731,"markDefs":744,"style":93},"d62efee9e822",[736,740],{"_key":737,"_type":63,"marks":738,"text":739},"f9b78e7a69eb",[86],"Per-record model\u002Fnetwork round-trips",{"_key":741,"_type":63,"marks":742,"text":743},"a03d9f147761",[]," → batch, or skip the model entirely for easy cases (`partition(rows, is_deterministic)`, rules for the easy ones, one batched model call for the hard ones). This is the same “don’t call the model” lever from cost, paying off twice.",[],{"_key":746,"_type":59,"children":747,"level":318,"listItem":731,"markDefs":756,"style":93},"35a6e2d67193",[748,752],{"_key":749,"_type":63,"marks":750,"text":751},"176a9821fc26",[86],"Serial stages that could be parallel",{"_key":753,"_type":63,"marks":754,"text":755},"463eddfe00de",[]," → bounded concurrency, sized to the real constraint (provider rate limit, CPU, memory) — not unbounded, which just moves the bottleneck and blows limits.",[],{"_key":758,"_type":59,"children":759,"level":318,"listItem":731,"markDefs":768,"style":93},"6c7b1fc82cab",[760,764],{"_key":761,"_type":63,"marks":762,"text":763},"4985cef77ba9",[86],"Recomputation",{"_key":765,"_type":63,"marks":766,"text":767},"bf8eb3e2f916",[]," → cache deterministic work (embeddings, parsed inputs, reference lookups).",[],{"_key":770,"_type":59,"children":771,"markDefs":784,"style":93},"aeebf9618716",[772,776,780],{"_key":773,"_type":63,"marks":774,"text":775},"e9c90cafe1a0",[],"After every change, re-run both axes — latency benchmark ",{"_key":777,"_type":63,"marks":778,"text":779},"29c7d73964b1",[69],"and",{"_key":781,"_type":63,"marks":782,"text":783},"3ddfa2c5fcc2",[]," eval vs baseline — and revert anything that regressed quality no matter how fast it is:",[],{"_key":786,"_type":116,"code":787,"markDefs":12},"8840e77cfbdf","change                  latency     quality vs baseline\nhash-join (was O(n²))   180s → 12s   = baseline ✓\nbatch model calls       12s → 6s     = baseline ✓\nparallel stages (×8)    6s → 1.8s    = baseline ✓",{"_key":789,"_type":59,"children":790,"markDefs":795,"style":93},"47820eba4b0b",[791],{"_key":792,"_type":63,"marks":793,"text":794},"b261295db1e5",[],"The model is rarely the bottleneck — and “fast” should never be a guess about whether it’s still correct.",[],{"_key":797,"_type":59,"children":798,"markDefs":803,"style":131},"a05b850552cb",[799],{"_key":800,"_type":63,"marks":801,"text":802},"f320f266936e",[],"Routing, fallback, and the off switch",[],{"_key":805,"_type":59,"children":806,"markDefs":811,"style":93},"2b855918a1ce",[807],{"_key":808,"_type":63,"marks":809,"text":810},"4defdf01395f",[],"You don’t have “a model.” You have a fleet — cheap and capable, primary and judge, this provider and that — and you need a layer that picks the right one, survives when one goes down, and can be stopped in seconds when it misbehaves. All three live behind the same gateway, and all three only work if the domain doesn’t care which model answered.",[],{"_key":813,"_type":59,"children":814,"markDefs":819,"style":93},"aa8c46f92059",[815],{"_key":816,"_type":63,"marks":817,"text":818},"86ede0d5a050",[86],"Route by what the decision needs",[],{"_key":821,"_type":116,"code":822,"markDefs":12},"64d5e7da2f00","condition                        route to                           why\n-------------------------------  ---------------------------------  ----------------------------------\n`is_judge`                       a *different* family from primary  independent blind spots\n`stakes == low` \u002F high volume    cheap\u002Fsmall model                  most traffic; biggest cost lever\nambiguous \u002F high stakes          capable model                      accuracy where it matters\nlong-context \u002F extraction-heavy  the model measured best at it      task fit (only if you've measured)",{"_key":824,"_type":116,"code":825,"language":565,"markDefs":12},"7fe6425adc54","def route(d) -> str:\n    if d.is_judge:               return JUDGE_MODEL        # independent from primary\n    if d.stakes == \"low\":        return CHEAP_MODEL\n    if d.needs_long_context:     return LONG_CTX_MODEL\n    return CAPABLE_MODEL\n# keep routing rules in ONE place (the gateway), readable + testable — not per-call-site strings",{"_key":827,"_type":59,"children":828,"markDefs":833,"style":93},"4b6355e74ff6",[829],{"_key":830,"_type":63,"marks":831,"text":832},"d53932b93ddd",[86],"Fallback: survive a provider going down",[],{"_key":835,"_type":116,"code":836,"language":565,"markDefs":12},"5f54c06546a7","def route_chain(req) -> list[str]:                  # the ordered fallback chain\n    if req.is_judge:        return [JUDGE_MODEL]     # no silent fallback for a judge\n    if req.stakes == \"low\": return [CHEAP_MODEL, CAPABLE_MODEL]   # fall UP on failure\n    return [CAPABLE_MODEL, FALLBACK_MODEL]\n\ndef complete_with_fallback(req, deadline):\n    for model in route_chain(req):\n        if breaker[model].is_open():                # CHECK BEFORE calling — skip a known-dead provider\n            continue\n        remaining = deadline - now()                # per-ATTEMPT budget vs an overall deadline\n        if remaining \u003C= 0:\n            break\n        try:\n            resp = call(model, req, timeout=remaining, idem_key=req.idem_key)\n            breaker[model].record_success()\n            return resp\n        except (Timeout, ProviderError):\n            breaker[model].record_failure()         # record on EVERY failure → the breaker can open\n    return route_to_human(req, reason=\"all_models_unavailable\")   # degrade deliberately, don't error",{"_key":838,"_type":59,"children":839,"markDefs":896,"style":93},"93759bacc935",[840,844,848,852,856,860,864,868,872,876,880,884,888,892],{"_key":841,"_type":63,"marks":842,"text":843},"0314010f93bd",[],"The non-obvious bits that make this correct: ",{"_key":845,"_type":63,"marks":846,"text":847},"9d040323f329",[86],"check the breaker before calling",{"_key":849,"_type":63,"marks":850,"text":851},"184c887638a1",[]," and record a failure on ",{"_key":853,"_type":63,"marks":854,"text":855},"e110b4bca09c",[69],"every",{"_key":857,"_type":63,"marks":858,"text":859},"36a1d1ed8589",[]," caught error (the common bug — doing both in the ",{"_key":861,"_type":63,"marks":862,"text":863},"760b06dd02b5",[116],"except",{"_key":865,"_type":63,"marks":866,"text":867},"d73df5433111",[]," means the breaker never proactively skips a dead provider). Use ",{"_key":869,"_type":63,"marks":870,"text":871},"23040fea98de",[86],"one overall deadline with per-attempt timeouts",{"_key":873,"_type":63,"marks":874,"text":875},"539da32ce441",[69]," — ",{"_key":877,"_type":63,"marks":878,"text":879},"15f028370c25",[],"the naive alternative, a fixed timeout reused per model, makes total latency N × timeout and blows the upstream budget. ",{"_key":881,"_type":63,"marks":882,"text":883},"ef0f30ecbbd6",[86],"Idempotency is mandatory",{"_key":885,"_type":63,"marks":886,"text":887},"0318ef15eb5d",[],": a primary that timed out but actually completed (and wrote a ledger entry) must not be double-processed. And ",{"_key":889,"_type":63,"marks":890,"text":891},"543d5514b966",[86],"degrade deliberately",{"_key":893,"_type":63,"marks":894,"text":895},"c87345a58f46",[]," — decide per capability whether a weaker fallback’s answer is acceptable or it should route to a human.",[],{"_key":898,"_type":59,"children":899,"markDefs":920,"style":93},"aeec91dd9761",[900,904,908,912,916],{"_key":901,"_type":63,"marks":902,"text":903},"21303456f616",[],"One discipline ties routing and fallback together: ",{"_key":905,"_type":63,"marks":906,"text":907},"30e5ed9eff4d",[86],"eval every model on the path.",{"_key":909,"_type":63,"marks":910,"text":911},"ca396f41896b",[]," A routed-to or fallen-back-to model is a ",{"_key":913,"_type":63,"marks":914,"text":915},"e7a9540eae31",[69],"different",{"_key":917,"_type":63,"marks":918,"text":919},"2cf716edbe0b",[]," model, so potentially different quality. Your golden set should pass on the cheap model and the fallback model for the capabilities that use them — a fallback that quietly tanks quality is a worse outage than the one it covers. Record which model decided (in the ledger), so outcome analysis can see whether the cheap or fallback model underperformed.",[],{"_key":922,"_type":59,"children":923,"markDefs":928,"style":93},"692490251c47",[924],{"_key":925,"_type":63,"marks":926,"text":927},"74ec9c569bc3",[86],"The kill switch: an off that takes effect in seconds",[],{"_key":930,"_type":59,"children":931,"markDefs":936,"style":93},"25eb34b9459c",[932],{"_key":933,"_type":63,"marks":934,"text":935},"2f5b86feb410",[],"A deploy takes minutes you may not have. The switch must be runtime state every node checks:",[],{"_key":938,"_type":116,"code":939,"language":565,"markDefs":12},"dbf23b5e74fa","class Mode(Enum):\n    LIVE       = \"live\"          # normal\n    HUMAN_ONLY = \"human_only\"    # stop auto-execute; still propose to humans\n    HALTED     = \"halted\"        # stop deciding entirely\n\ndef get_mode(switch, tenant_id, capability) -> Mode:\n    try:\n        raw = (switch.read(f\"killswitch:{tenant_id}:{capability}\")  # most specific\n               or switch.read(f\"killswitch:{tenant_id}\")            # whole tenant\n               or switch.read(\"killswitch:global\"))                 # global flip lands everywhere\n        return Mode(raw) if raw else Mode.LIVE\n    except SwitchUnavailable:\n        return Mode.HUMAN_ONLY   # fail toward safe — NOT live, NOT halted (a blip shouldn't self-DoS)",{"_key":941,"_type":59,"children":942,"markDefs":1003,"style":93},"5f991c1728e3",[943,947,951,955,959,963,967,971,975,979,983,987,991,995,999],{"_key":944,"_type":63,"marks":945,"text":946},"e152f4bea202",[],"Four design choices make it trustworthy: ",{"_key":948,"_type":63,"marks":949,"text":950},"02295562d3b5",[86],"fast propagation",{"_key":952,"_type":63,"marks":953,"text":954},"2f9ddb36b99f",[]," (back it with shared state \u002F pub-sub so a flip lands across all instances in ",{"_key":956,"_type":63,"marks":957,"text":958},"d2a372543e35",[69],"seconds",{"_key":960,"_type":63,"marks":961,"text":962},"8daa04cd4d21",[]," — a 30s local TTL is not “seconds”); ",{"_key":964,"_type":63,"marks":965,"text":966},"45dd37ed8231",[86],"granular",{"_key":968,"_type":63,"marks":969,"text":970},"3872bc5224a7",[]," (per-tenant and per-capability, so the blast radius of “off” matches the blast radius of the problem); a ",{"_key":972,"_type":63,"marks":973,"text":974},"ff6eec16e4f3",[86],"middle gear",{"_key":976,"_type":63,"marks":977,"text":978},"b602a5505712",[]," (`HUMAN_ONLY` keeps proposing while stopping auto-execution — often you don’t need ",{"_key":980,"_type":63,"marks":981,"text":982},"ee8ee6c99c63",[69],"off",{"_key":984,"_type":63,"marks":985,"text":986},"549525716f25",[],", you need ",{"_key":988,"_type":63,"marks":989,"text":990},"f9dc2424c1fb",[69],"humans back in the loop",{"_key":992,"_type":63,"marks":993,"text":994},"817ec8d5673b",[],"); and ",{"_key":996,"_type":63,"marks":997,"text":998},"02fda9148a41",[86],"fail toward safe",{"_key":1000,"_type":63,"marks":1001,"text":1002},"d7d48c6b8f40",[]," (if a node can’t read the switch, assume the conservative mode).",[],{"_key":1005,"_type":59,"children":1006,"markDefs":1011,"style":93},"92bae8b74fa5",[1007],{"_key":1008,"_type":63,"marks":1009,"text":1010},"facf2348caee",[86],"Degrade by design, and prove it",[],{"_key":1013,"_type":59,"children":1014,"markDefs":1051,"style":93},"35ab503311cd",[1015,1019,1023,1027,1031,1035,1039,1043,1047],{"_key":1016,"_type":63,"marks":1017,"text":1018},"8cab777f6a64",[],"Decide ",{"_key":1020,"_type":63,"marks":1021,"text":1022},"11b6a0fd291d",[69],"in advance",{"_key":1024,"_type":63,"marks":1025,"text":1026},"453ec2078656",[]," how the system bends so failure isn’t a cascade. A ",{"_key":1028,"_type":63,"marks":1029,"text":1030},"3ea6aa0c4c38",[86],"circuit breaker",{"_key":1032,"_type":63,"marks":1033,"text":1034},"8f74515fe08d",[]," stops you hammering a dead dependency — failing in milliseconds instead of behind 60s timeouts, which is exactly how a dependency outage becomes ",{"_key":1036,"_type":63,"marks":1037,"text":1038},"d651f3668d9d",[69],"your",{"_key":1040,"_type":63,"marks":1041,"text":1042},"3c35af7a5f52",[]," thread-exhaustion outage. Under overload, ",{"_key":1044,"_type":63,"marks":1045,"text":1046},"88b645648773",[86],"shed load deliberately",{"_key":1048,"_type":63,"marks":1049,"text":1050},"16e9fa9d07f6",[],": slow down or route-to-human rather than crash. A system that degrades to “a human handles it” is still serving its purpose.",[],{"_key":1053,"_type":59,"children":1054,"markDefs":1059,"style":93},"961fd1816dea",[1055],{"_key":1056,"_type":63,"marks":1057,"text":1058},"02b352655ebc",[],"These mechanisms are worthless if they only work in theory. Run chaos tests against agents:",[],{"_key":1061,"_type":116,"code":1062,"markDefs":12},"193f7903887d","chaos test                   assert\n---------------------------  -----------------------------------------------------\nkill the model provider      decisions route to humans, not error out\nflip the kill switch         auto-execution stops, fast, across all instances\noverload \u002F burst             sheds load \u002F degrades, doesn't crash\nrestart a node mid-decision  in-flight decisions recover or fail safe (idempotent)\ndependency latency spike     circuit breaker trips; no thread pileup",{"_key":1064,"_type":59,"children":1065,"markDefs":1070,"style":93},"f1bbbb9582c4",[1066],{"_key":1067,"_type":63,"marks":1068,"text":1069},"079b481b0b2f",[],"If you haven’t exercised the kill switch and fallbacks under real failure, you don’t know they work — you’re hoping. The kill switch is the thing that lets you sleep.",[],{"_key":1072,"_type":59,"children":1073,"markDefs":1078,"style":131},"042a19c35085",[1074],{"_key":1075,"_type":63,"marks":1076,"text":1077},"0eb06701716a",[],"The platform underneath",[],{"_key":1080,"_type":59,"children":1081,"markDefs":1086,"style":93},"aa2b2dfe42bd",[1082],{"_key":1083,"_type":63,"marks":1084,"text":1085},"a416861f0aff",[],"All of the above assumes a place to stand: a structure that lets you swap models without a refactor, an identity model that keeps each action attributable, and a right-sized set of infrastructure.",[],{"_key":1088,"_type":59,"children":1089,"markDefs":1094,"style":93},"1763230b14cf",[1090],{"_key":1091,"_type":63,"marks":1092,"text":1093},"be9a8a83430a",[86],"Hexagonal architecture: keep the vendor out of your domain",[],{"_key":1096,"_type":59,"children":1097,"markDefs":1134,"style":93},"43894f4484b9",[1098,1102,1106,1110,1114,1118,1122,1126,1130],{"_key":1099,"_type":63,"marks":1100,"text":1101},"f0709932ac4e",[],"The model you ship on won’t be the model you started with. Providers leapfrog every few months; pricing changes; a region or compliance rule forces a switch. If your business logic is littered with vendor SDK imports and provider-shaped request objects, every one of those is a refactor. The fix is an old idea applied to a new problem: ",{"_key":1103,"_type":63,"marks":1104,"text":1105},"3b7edb83525b",[86],"ports and adapters.",{"_key":1107,"_type":63,"marks":1108,"text":1109},"4a5b306d4747",[]," Your ",{"_key":1111,"_type":63,"marks":1112,"text":1113},"0440134cb376",[86],"domain",{"_key":1115,"_type":63,"marks":1116,"text":1117},"bc7469da6602",[]," depends only on ",{"_key":1119,"_type":63,"marks":1120,"text":1121},"d18869bbaa3d",[86],"ports",{"_key":1123,"_type":63,"marks":1124,"text":1125},"84d9133f0d7d",[]," — interfaces you define, in your terms. The messy outside (model providers, datastores, queues) lives in ",{"_key":1127,"_type":63,"marks":1128,"text":1129},"f353e1cd9634",[86],"adapters",{"_key":1131,"_type":63,"marks":1132,"text":1133},"632e35bb87de",[]," that implement those ports. The domain never imports a vendor SDK.",[],{"_key":1136,"_type":59,"children":1137,"markDefs":1154,"style":93},"e26dcfb34e48",[1138,1142,1146,1150],{"_key":1139,"_type":63,"marks":1140,"text":1141},"f2832d2babae",[],"Define one gateway through which ",{"_key":1143,"_type":63,"marks":1144,"text":1145},"89b86b1d77a4",[69],"all",{"_key":1147,"_type":63,"marks":1148,"text":1149},"3ed7cbf32546",[]," model traffic flows, in ",{"_key":1151,"_type":63,"marks":1152,"text":1153},"def3ed156de5",[69],"your vocabulary:",[],{"_key":1156,"_type":116,"code":1157,"language":565,"markDefs":12},"1d34f10393f5","@dataclass(frozen=True)\nclass ChatRequest:                  # YOUR vocabulary — note what you deliberately DON'T expose\n    messages: list[dict]\n    schema: dict | None = None      # structured output (your concept, not a provider's response_format)\n    max_tokens: int = 1024\n    # no provider-specific knobs (logit_bias, etc.) — they leak the vendor into the domain.\n\nclass LLMGateway(Protocol):         # shown sync for clarity; a real gateway is async\u002Fstreaming\n    def complete(self, request: ChatRequest) -> ChatResponse: ...   # your types, not a provider's",{"_key":1159,"_type":59,"children":1160,"markDefs":1165,"style":93},"8b6b2ca54a8e",[1161],{"_key":1162,"_type":63,"marks":1163,"text":1164},"7765450c4001",[],"This is the same gateway that does cost measurement, rate limiting, routing, fallback, and the kill switch — every cross-cutting concern lives here once, not scattered. That’s not a coincidence: the chokepoint that makes the domain portable is the chokepoint that makes the system operable. Because the domain depends on a Protocol, tests inject a fake — no network, no spend, no flakiness — while adapters get their own integration tests against the real thing.",[],{"_key":1167,"_type":59,"children":1168,"markDefs":1189,"style":93},"5fc635cdd9cf",[1169,1173,1177,1181,1185],{"_key":1170,"_type":63,"marks":1171,"text":1172},"6f060225bccf",[],"Don’t rely on vigilance to keep the boundary clean; ",{"_key":1174,"_type":63,"marks":1175,"text":1176},"5509ceb763a6",[86],"enforce it with a linter.",{"_key":1178,"_type":63,"marks":1179,"text":1180},"28d2f0aa7a64",[]," An import-contract rule (“the domain may not import vendor SDKs”) turns the build red the moment someone slips a vendor import into domain code. Pair it with a naming convention — anything under ",{"_key":1182,"_type":63,"marks":1183,"text":1184},"235be2c3f642",[116],"adapters\u002F",{"_key":1186,"_type":63,"marks":1187,"text":1188},"3ad6ffff8f74",[]," or suffixed per-provider may import that provider, nothing else may. Most ecosystems have an equivalent (module-boundary lint in JS\u002FTS, ArchUnit in Java, depguard in Go). It costs an extra translation layer and some upfront interface design; it buys one-file provider swaps, a testable domain, and a boundary the build maintains forever. Then \"can we switch models?\" stops being a project and becomes an afternoon.",[],{"_key":1191,"_type":59,"children":1192,"markDefs":1197,"style":93},"b386ce90d437",[1193],{"_key":1194,"_type":63,"marks":1195,"text":1196},"35ddfd0a5a2f",[86],"Identity: whose authority is each action carried with?",[],{"_key":1199,"_type":59,"children":1200,"markDefs":1213,"style":93},"bae2d6d78f0e",[1201,1205,1209],{"_key":1202,"_type":63,"marks":1203,"text":1204},"63adac4de9a3",[],"An agent that acts on a user’s behalf has an identity problem a stateless API doesn’t. A request arrives as a user; the agent reasons, calls downstream services, maybe runs a while, maybe does work after the user has gone. At every hop: ",{"_key":1206,"_type":63,"marks":1207,"text":1208},"c149c8b67ccd",[69],"whose authority is this running with, and is it allowed to do this, for this user, in this tenant?",{"_key":1210,"_type":63,"marks":1211,"text":1212},"721e0fc21064",[]," Mishandle it and you’ve built the classic confused deputy — the agent using its own broad privileges to do something the user couldn’t.",[],{"_key":1215,"_type":59,"children":1216,"markDefs":1261,"style":93},"3dbd19a2a49a",[1217,1221,1225,1229,1233,1237,1241,1245,1249,1253,1257],{"_key":1218,"_type":63,"marks":1219,"text":1220},"ddf8b8ca8f5f",[],"Two flows, two strategies. ",{"_key":1222,"_type":63,"marks":1223,"text":1224},"d6f65875d3a3",[86],"Short, synchronous",{"_key":1226,"_type":63,"marks":1227,"text":1228},"e539ed41052f",[]," (\u003C ~60s): propagate the ",{"_key":1230,"_type":63,"marks":1231,"text":1232},"f5f61f14c6fc",[69],"user’s",{"_key":1234,"_type":63,"marks":1235,"text":1236},"0aaea692aa18",[]," credential — the inbound token rides along to downstream calls, so every action runs with exactly the user’s authority (safe only when every downstream does its own per-action authz). ",{"_key":1238,"_type":63,"marks":1239,"text":1240},"3de7ca89d739",[86],"Long-running \u002F deferred: ",{"_key":1242,"_type":63,"marks":1243,"text":1244},"43da6fbe9d59",[],"you can’t hold the user’s token, so at the entry node mint a",{"_key":1246,"_type":63,"marks":1247,"text":1248},"342f528d92f6",[86]," short-lived delegated grant ",{"_key":1250,"_type":63,"marks":1251,"text":1252},"9b0c81faa80f",[],"— an asymmetrically-signed token scoped to this workflow, acting ",{"_key":1254,"_type":63,"marks":1255,"text":1256},"07d8cfa97d38",[69],"for",{"_key":1258,"_type":63,"marks":1259,"text":1260},"593874f7bdb6",[]," the user, bound to one audience and one tenant, with least-privilege exact-match scopes, a revocable id, and a TTL in hours with a hard cap.",[],{"_key":1263,"_type":59,"children":1264,"markDefs":1277,"style":93},"56630a3f0d2a",[1265,1269,1273],{"_key":1266,"_type":63,"marks":1267,"text":1268},"1d70c47f55a0",[],"The single most important check, on ",{"_key":1270,"_type":63,"marks":1271,"text":1272},"58c0c989e4c9",[86],"every hop",{"_key":1274,"_type":63,"marks":1275,"text":1276},"101a207fe267",[],", is that the identity’s tenant matches the request’s tenant:",[],{"_key":1279,"_type":116,"code":1280,"language":565,"markDefs":12},"b6e89c7821e9","def authorize(token: str, request):\n    # AUTHENTICATE first: asymmetric signature (downstream verifies but can't mint), reject\n    # alg=none, check exp\u002Fiat, aud == this service, jti not denylisted. Never trust a parsed identity.\n    identity = verify_grant(token, expected_aud=THIS_SERVICE, leeway_s=30)\n    if identity.tenant_id != request.tenant_id:\n        audit.violation(\"cross_tenant\", identity=identity, request=request)\n        raise Forbidden()                              # never proceed\n    if request.action not in identity.scope:           # EXACT membership — no \"write:*\" wildcard\n        raise Forbidden(f\"out of scope: {request.action}\")\n    return identity",{"_key":1282,"_type":59,"children":1283,"markDefs":1303,"style":93},"09631f2b4a8d",[1284,1288,1292,1296,1299],{"_key":1285,"_type":63,"marks":1286,"text":1287},"356ff8be1f02",[],"This is the check that stops one tenant’s agent from ever touching another’s data — even through a bug or a prompt injection. ",{"_key":1289,"_type":63,"marks":1290,"text":1291},"2442572c747d",[86],"Sign asymmetrically",{"_key":1293,"_type":63,"marks":1294,"text":1295},"848b95a8af79",[]," (a shared HMAC secret lets any holder mint grants); use least privilege and short TTLs (a leaked short-lived grant is a small problem, a long-lived one is a breach); validate at ",{"_key":1297,"_type":63,"marks":1298,"text":855},"bd88c562a489",[69],{"_key":1300,"_type":63,"marks":1301,"text":1302},"886417feff05",[]," hop (injection happens after the edge); and log cross-tenant denials as audit violations — they’re attack signal, never a silent 403. The point of all of it: every single thing your agents do is attributable, scoped, and bounded to the person and tenant it was meant for.",[],{"_key":1305,"_type":59,"children":1306,"markDefs":1311,"style":93},"83aa6a943fab",[1307],{"_key":1308,"_type":63,"marks":1309,"text":1310},"ef8b703d4ce8",[86],"Infrastructure: provision the spine, resist the cargo cult",[],{"_key":1313,"_type":59,"children":1314,"markDefs":1319,"style":93},"e42e922af978",[1315],{"_key":1316,"_type":63,"marks":1317,"text":1318},"e4c2190ec201",[],"Standing up an agent platform swings between two failure modes: under-provisioning (no audit store, no secrets management, one shared god-credential) or over-provisioning (a queue, three databases, and a service mesh for what is really a handful of stateless services). The actual shortlist:",[],{"_key":1321,"_type":116,"code":1322,"markDefs":12},"99d3d76dfcca","component        why                                                          start with\n---------------  -----------------------------------------------------------  ------------------------------------\nagent services   the capabilities                                             a few stateless services, autoscaled\nmodel gateway    one chokepoint; only thing holding provider creds            1\naudit datastore  the append-only decision ledger (canonical record)           a relational DB\ncache            session\u002Fworking state; backing for rate-limit + kill switch  1 (e.g. Redis)\nsecrets store    model keys, guardrail config, signing keys                   managed secrets manager\nobject storage   large inputs\u002Fartifacts + tiered (regulated) logs, encrypted  1 bucket",{"_key":1324,"_type":59,"children":1325,"markDefs":1346,"style":93},"4632268ed859",[1326,1330,1334,1338,1342],{"_key":1327,"_type":63,"marks":1328,"text":1329},"0559af3b5252",[],"Agent endpoints are structurally ",{"_key":1331,"_type":63,"marks":1332,"text":1333},"ab0f49abba36",[86],"stateless HTTP services",{"_key":1335,"_type":63,"marks":1336,"text":1337},"1eae8e58a809",[],". If they propose decisions and don’t run long internal loops, you usually do ",{"_key":1339,"_type":63,"marks":1340,"text":1341},"78fc5801cf13",[69],"not",{"_key":1343,"_type":63,"marks":1344,"text":1345},"2b0a789e356f",[]," need a queue or event bus to start — sync request\u002Fresponse covers it. Notice what’s deliberately absent: a queue, a vector store, a second database. Add each only when a concrete workload demands it.",[],{"_key":1348,"_type":59,"children":1349,"markDefs":1393,"style":93},"3873058828d3",[1350,1354,1358,1362,1366,1370,1374,1378,1382,1386,1389],{"_key":1351,"_type":63,"marks":1352,"text":1353},"cdf61327f5ce",[],"Two disciplines matter more than the component list. ",{"_key":1355,"_type":63,"marks":1356,"text":1357},"df0ade3b7dd3",[86],"Least-privilege IAM",{"_key":1359,"_type":63,"marks":1360,"text":1361},"35cfdf52b915",[],": don’t hand every service one broad credential — the gateway gets ",{"_key":1363,"_type":63,"marks":1364,"text":1365},"40d7433278d8",[116],"model:invoke",{"_key":1367,"_type":63,"marks":1368,"text":1369},"1ffa30699e40",[]," and the model-keys secret; agent services get the audit store and their own object-store prefix but ",{"_key":1371,"_type":63,"marks":1372,"text":1373},"9d03be903f17",[69],"no",{"_key":1375,"_type":63,"marks":1376,"text":1377},"bd3e58ec11df",[]," model creds (they call the gateway); the audit DB user is scoped to its schema. Use per-service workload identities, never shared static keys, so a compromise’s blast radius is one component’s narrow permissions, not the platform. And ",{"_key":1379,"_type":63,"marks":1380,"text":1381},"2903ea0ba2f6",[86],"encryption",{"_key":1383,"_type":63,"marks":1384,"text":1385},"8488ffebade4",[],": a managed key for data at rest from day one, a dedicated signing key if the ledger is cryptographically signed, and per-tenant keys designed for even if you start single-tenant. The discipline isn’t adding exotic infrastructure — it’s ",{"_key":1387,"_type":63,"marks":1388,"text":1341},"6d06ddfd8194",[69],{"_key":1390,"_type":63,"marks":1391,"text":1392},"90669e88bd89",[]," adding it, scoping every credential tightly, and putting the audit store and the gateway chokepoint in from day one.",[],{"_key":1395,"_type":59,"children":1396,"markDefs":1401,"style":131},"0cbd94efb798",[1397],{"_key":1398,"_type":63,"marks":1399,"text":1400},"4aa62d06ea9b",[],"The takeaway",[],{"_key":1403,"_type":59,"children":1404,"markDefs":1431,"style":93},"f044410c7f16",[1405,1409,1412,1416,1419,1423,1427],{"_key":1406,"_type":63,"marks":1407,"text":1408},"71cfdcdfd66b",[],"Operability is what turns a working LLM system into one you can actually run. Instrument the ",{"_key":1410,"_type":63,"marks":1411,"text":217},"49fbd91494c3",[69],{"_key":1413,"_type":63,"marks":1414,"text":1415},"72564886a583",[],", not just the service — a metric catalog labeled by tenant and capability, tiered logs, a ",{"_key":1417,"_type":63,"marks":1418,"text":117},"bcaa3f2cd588",[116],{"_key":1420,"_type":63,"marks":1421,"text":1422},"f01fa8293cf5",[]," threading metrics, logs, traces, and ledger, and alerts on symptoms with real thresholds. Funnel all model traffic through one gateway so cost is visible and bounded, then pull the levers in order: skip the model, batch, cache, right-size, trim — and profile before you optimize, baselining quality at every step. Treat the model as a ",{"_key":1424,"_type":63,"marks":1425,"text":1426},"0f51766307e6",[69],"routing decision",{"_key":1428,"_type":63,"marks":1429,"text":1430},"f8122285c5f1",[],", not a constant: cheap to cheap, hard to capable, an independent model to judge, failover so one provider's outage isn't yours — with an instant, granular kill switch and designed degradation you've actually proven under chaos. And stand all of it on a platform that keeps the vendor behind ports, threads tenant-checked identity through every hop, and provisions the spine without the cargo cult.",[],{"_key":1433,"_type":59,"children":1434,"markDefs":1447,"style":93},"9b64ced64bab",[1435,1439,1443],{"_key":1436,"_type":63,"marks":1437,"text":1438},"eedf538db013",[],"The thread running through every section is the same: ",{"_key":1440,"_type":63,"marks":1441,"text":1442},"35c5219c704f",[86],"one chokepoint.",{"_key":1444,"_type":63,"marks":1445,"text":1446},"64bbb88f4b76",[]," The gateway that makes your domain portable is the gateway where you measure cost, route models, fail over, and flip the kill switch. Build that one seam well and operability stops being a scramble during incidents and becomes a property of the system.",[],{"_key":1449,"_type":59,"children":1450,"markDefs":1455,"style":93},"4ded0f107028",[1451],{"_key":1452,"_type":63,"marks":1453,"text":1454},"b7b4cc538a4c",[69],"Series: Running LLM systems in production — Level 5 of 6: Operability.",[],true,"2026\u002F10\u002F08","Your service can be 100% up and still quietly approving the wrong things, burning its budget, or failing over into untested quality. Level 5 is the infrastructure that lets you see your decisions, bound your spend, route and fail over between models, kill bad behavior in seconds — and the platform that makes all of it possible.",{"_type":1460,"alt":1461,"asset":1462},"image","A diagram titled \"Operating an LLM System,\" showing components like a gateway, kill switch, and processing units (Cheap, Capable, Long-Context, Judge), with monitoring gauges for behavior, safety, cost, and quality.",{"_ref":1463,"_type":1464},"image-667f7a3f8613f4689cd750f08fa181812c248a4e-3000x1500-png","reference","2026-10-08T20:08:53.629Z",{"_type":1467,"canonicalUrl":1468},"seo","https:\u002F\u002Fmedium.com\u002F@varunjindal9\u002Foperating-an-llm-system-observability-cost-routing-and-the-platform-underneath-12403b8e4689",{"_type":10,"current":1470},"part-5-operating-an-llm-system-observability-cost-routing-and-the-platform-underneath",[1472,1481,1492,1515],{"_createdAt":1473,"_id":1474,"_rev":1475,"_type":1476,"_updatedAt":1477,"slug":1478,"title":1480},"2023-05-23T16:43:21Z","wp-tagcat-ai","fpDTFQqIDjNJIbHDKPBGpV","blogTag","2025-01-30T16:19:01Z",{"current":1479},"ai","AI",{"_createdAt":1482,"_id":1483,"_rev":1484,"_system":1485,"_type":1476,"_updatedAt":1488,"slug":1489,"title":1491},"2026-06-12T16:16:20Z","51c761d7-73f7-42f4-aa49-8484e3849e7c","P0qLqkXH0zpkT6RRZ9Iwel",{"base":1486},{"id":1483,"rev":1487},"MwgZb85ftkde1TTvQsHYa6","2026-09-28T16:40:45Z",{"_type":10,"current":1490},"building-software","Building software",{"_createdAt":1493,"_id":1494,"_rev":1495,"_system":1496,"_type":1476,"_updatedAt":1499,"description":1500,"featuredPosts":1509,"slug":1512,"title":1514},"2025-04-24T16:28:57Z","797b8797-6e65-4723-b53f-8bc005305384","46s78gX2DRxVswzX0kQ1Ty",{"base":1497},{"id":1494,"rev":1498},"IpfPEqg1c3Byvj9RrB3Xaj","2026-10-07T20:13:00Z",[1501],{"_key":1502,"_type":59,"children":1503,"markDefs":1508,"style":93},"bb32f75814b4",[1504],{"_key":1505,"_type":63,"marks":1506,"text":1507},"dbcf27ef29b3",[],"Community-generated articles submitted for your reading pleasure. If you’re interested in seeing your work here, log in with your Stack Overflow account and click the link below. Articles will be licensed under a CC BY-SA 4.0 grant. ",[],[1510],{"_key":1511,"_type":1464},"9d9ea8c4082d",{"_type":10,"current":1513},"contributed","The Heap",{"_createdAt":1516,"_id":1517,"_rev":1518,"_system":1519,"_type":1476,"_updatedAt":1522,"description":1523,"slug":1543,"title":1545},"2025-08-08T15:49:22Z","39391cf4-6f9a-4238-8670-c1e44b66db9e","09X6HDzCi2VfMov6gSLf7H",{"base":1520},{"id":1517,"rev":1521},"TdCcmC7LyfLVwjB8GEXoh6","2025-12-10T19:34:33Z",[1524,1532],{"_key":1525,"_type":59,"children":1526,"markDefs":1531,"style":93},"a4b1a37cbbcc",[1527],{"_key":1528,"_type":63,"marks":1529,"text":1530},"d8e8f3e0fd9c",[],"These articles are licensed under a Creative Commons Attribution-ShareAlike 4.0 International license. ",[],{"_key":1533,"_type":59,"children":1534,"markDefs":1540,"style":93},"7effd489c71f",[1535],{"_key":1536,"_type":63,"marks":1537,"text":1539},"538808bb5325",[1538],"fd643b288690","creativecommons.org\u002Flicenses\u002Fby-sa\u002F4.0\u002Fdeed.en",[1541],{"_key":1538,"_type":1542},"link",{"_type":10,"current":1544},"cc-by-sa","CC BY-SA 4.0","Part 5: Operating an LLM system: observability, cost, routing, and the platform underneath",[1548,1554,1560,1566],{"_id":1549,"publishedAt":1550,"slug":1551,"sponsored":12,"title":1553},"f13e8883-6d06-432d-9f93-10ae8b0fd048","2026-10-08T20:22:56.685Z",{"_type":10,"current":1552},"production-grade-llms-and-agents-a-field-guide","Production-grade LLMs and agents: a field guide",{"_id":1555,"publishedAt":1556,"slug":1557,"sponsored":12,"title":1559},"358fa8ac-2467-4ef4-893c-dad3490046df","2026-10-08T14:00:00.000Z",{"_type":10,"current":1558},"a-green-exit-code-is-not-evidence-that-the-work-happened","A green exit code is not evidence that the work happened",{"_id":1561,"publishedAt":1562,"slug":1563,"sponsored":12,"title":1565},"4ae1a4e4-cb6e-4110-b6f3-50b8afa09f49","2026-10-07T20:48:55.325Z",{"_type":10,"current":1564},"part-4-safety-and-governance-for-llm-systems-guardrails-pii-audit-and-memory","Part 4: Safety and governance for LLM systems: guardrails, PII, audit, and memory",{"_id":1567,"publishedAt":1568,"slug":1569,"sponsored":12,"title":1571},"ce1fd642-fe2b-4d83-95ad-67b7645d7959","2026-10-07T20:40:29.066Z",{"_type":10,"current":1570},"part-3-knowing-when-your-agent-doesn-t-know-the-confidence-layer","Part 3: Knowing when your agent doesn’t know: the confidence layer",{"data":1573,"sourceMap":-1},{"count":1574,"lastTimestamp":12},0]