[{"data":1,"prerenderedAt":4280},["ShallowReactive",2],{"docs-nav":3,"docs-article-engineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fobservability-and-operations":797},[4,17,27,44,55,67,75,82,94,106,114,122,129,135,144,153,161,169,177,189,202,211,218,229,240,248,260,268,276,286,296,305,314,323,331,337,343,350,356,364,371,378,383,393,401,410,415,422,432,439,444,451,458,462,467,475,487,499,509,516,525,533,539,545,551,557,563,567,579,593,603,614,621,626,633,640,646,653,658,665,673,678,686,692,699,704,710,721,730,740,747,753,761,767,776,782,791],{"path":5,"title":6,"description":7,"group":8,"section":6,"order":9,"tags":10,"lastUpdated":16},"\u002Fagents\u002Fagentic-crm","Agentic CRM","Research brief and build plan for an AgencyCore agentic CRM layer, rendered as an interactive page — the core operating loop, the target architecture, the typed-tool risk gateway, the proposed-actions review queue, and the four-slice MVP.","Agents",0,[11,12,13,14,15],"crm","agents","ai","architecture","research","2026-06-12",{"path":18,"title":19,"description":20,"group":8,"section":21,"order":22,"tags":23,"lastUpdated":26},"\u002Fagents\u002Fchat","Chat agent","High-level system design of the AgencyCore chat agent — core components, data flow, and the two abstractions that hold it together.","Reference",1,[12,14,24,25],"chat","system-design","2026-05-13",{"path":28,"title":29,"description":30,"group":8,"section":31,"order":32,"tags":33,"lastUpdated":43},"\u002Fagents\u002Fcompany-enrichment","Company Enrichment","The company enrichment workflow - a cache-first read in front of the company intelligence database that fills firmographic, contact and technographic facts via a fixed-order provider waterfall, and writes every resolved fact back with provenance so the first org pays once and every later search rides free.","Enrichment",2,[12,34,35,36,37,38,39,40,41,42],"workflow","enrichment","companies","waterfall","cache","intelligence-database","firmographics","provenance","sonar","2026-06-10",{"path":45,"title":46,"description":47,"group":8,"section":48,"order":9,"tags":49,"lastUpdated":54},"\u002Fagents\u002Fcompany-sonar","Company Signals","Signal-first company discovery for marketing agencies, on the Claude Agent SDK, with a global intelligence cache and deterministic composite scoring.","Company Sonar",[12,34,42,50,51,52,35,53,14],"company-search","signals","agent-sdk","scoring","2026-06-08",{"path":56,"title":57,"description":58,"group":8,"section":48,"order":22,"tags":59,"lastUpdated":66},"\u002Fagents\u002Fcompany-sonar\u002Fsignal-monitoring","Company Signals Monitoring","Realtime signal capture layer on top of the data graph. Detects hot events, scores them with a Claude managed agent against each agency's ICP, fans out alerts.",[14,51,60,61,62,63,64,65],"intel","icp","alerts","monitoring","sse","managed-agents","2026-06-09",{"path":68,"title":69,"description":70,"group":8,"section":71,"order":22,"tags":72,"lastUpdated":74},"\u002Fagents\u002Fconcepts\u002Fchat-agent-design-principles","Designing chat agents","The 2026 playbook for production chat agents that reach into internal systems via tools — context engineering, memory, tool design, when to add complexity.","Concepts",[12,14,24,73],"context-engineering","2026-05-14",{"path":76,"title":77,"description":78,"group":8,"section":71,"order":32,"tags":79,"lastUpdated":74},"\u002Fagents\u002Fconcepts\u002Fsystem-prompt-architecture","System prompt architecture","How to structure a production chat agent system prompt — eight sections, what each one does, and the rules vendors converge on.",[12,80,81],"prompt-engineering","system-prompt",{"path":83,"title":84,"description":85,"group":8,"section":84,"order":9,"tags":86,"lastUpdated":54},"\u002Fagents\u002Fenvoy","Envoy","High-level system design for the AI outreach engine — the sequence step state machine, the human-in-the-loop draft approval gate, multi-source context enrichment, and the inbox sentiment flow, rendered as an interactive page.",[12,87,88,89,90,91,92,93,14],"envoy","outreach","sales-engagement","sequences","state-machine","human-in-the-loop","nylas",{"path":95,"title":96,"description":97,"group":8,"section":98,"order":9,"tags":99,"lastUpdated":16},"\u002Fagents\u002Fheadhunter","Headhunter","The AI talent-search pipeline on one page - the production six-step design with its current-title relevance gate, and the 2.0 system design with internal-first waterfall sourcing, a pluggable source registry, automatic entity resolution, and a people intelligence graph that compounds every run.","General Search",[12,34,100,101,14,25,102,37,103,104,105],"headhunter","recruiting","multi-source","entity-resolution","people-intelligence","flywheel",{"path":107,"title":108,"description":109,"group":8,"section":21,"order":32,"tags":110,"lastUpdated":113},"\u002Fagents\u002Fpaperclip","Paperclip","Architecture deep dive into the Paperclip orchestration system.",[12,14,111,112],"orchestration","paperclip","2026-04-20",{"path":115,"title":116,"description":117,"group":8,"section":31,"order":22,"tags":118,"lastUpdated":16},"\u002Fagents\u002Fpeople-enrichment","People Enrichment","The people enrichment workflow - a cache-first read in front of the people intelligence database that fills profile, contact and employment facts via a fixed-order provider waterfall, keyed on the LinkedIn URL, and writes every resolved fact back with provenance so the first org pays once and every later search rides free. The fill step Headhunter and People Signals both call.",[12,34,35,119,37,38,39,120,41,100,121],"people","linkedin","people-sonar",{"path":123,"title":124,"description":125,"group":8,"section":126,"order":9,"tags":127,"lastUpdated":54},"\u002Fagents\u002Fpeople-sonar","People Signals","Signal-first people discovery for marketing agencies, built on the headhunter pipeline, with a composite score weighted by signal strength, source reputation, recency, and ICP fit.","People Sonar",[12,34,121,128,51,100,35,53,14],"people-search",{"path":130,"title":131,"description":132,"group":8,"section":126,"order":22,"tags":133,"lastUpdated":54},"\u002Fagents\u002Fpeople-sonar\u002Fpeople-signal-monitoring","People Signals Monitoring","Forward-looking design for the push layer that tracks known people - champions, past contacts, target-company decision-makers - and fires a warm lead the moment they change jobs, get promoted, or their company has an event.",[14,51,60,119,63,134],"warm-leads",{"path":136,"title":137,"description":138,"group":139,"section":140,"order":22,"tags":141,"lastUpdated":143},"\u002Fengineering\u002Fguides\u002Fagent-execution-stack","The Agent Execution Stack","Durable workflows over pluggable agent backends — how AgencyCore runs AI agents on Inngest over a webhook-driven Claude Managed Agents backend.","Engineering","Guides",[12,142,14,25],"inngest","2026-06-25",{"path":145,"title":146,"description":147,"group":139,"section":140,"order":9,"tags":148,"lastUpdated":143},"\u002Fengineering\u002Fguides\u002Fagent-runtime","Agent runtime","How AgencyCore runs AI agents on a provider-neutral runtime — the abstraction layer that lets us swap the agent backend, with Claude managed agents as the current provider.",[12,149,14,150,151,152,25],"runtime","anthropic","claude","providers",{"path":154,"title":155,"description":156,"group":139,"section":21,"order":157,"tags":158,"lastUpdated":160},"\u002Fengineering\u002Freference\u002Fagno-to-agent-sdk-migration","Agno → Claude Agent SDK migration","System-design spec for moving the ac-python-api workflow engine off Agno onto Anthropic's Claude Agent SDK \u002F Managed Agents, tiered by control-flow shape.",10,[12,14,159,52,65],"migration","2026-06-06",{"path":162,"title":163,"description":164,"group":139,"section":21,"order":22,"tags":165,"lastUpdated":54},"\u002Fengineering\u002Freference\u002Fcloudflare-agent-sandbox","Cloudflare agent sandbox","Cloudflare's Workers-based agent platform, evaluated as an alternative sandbox for our Agno workflows.",[12,166,167,168,159],"sandbox","cloudflare","workers",{"path":170,"title":171,"description":172,"group":139,"section":21,"order":32,"tags":173,"lastUpdated":176},"\u002Fengineering\u002Freference\u002Fvirtual-filesystem-rag","Virtual filesystem for AI assistants","How ChromaFs provides AI agents with structured file access.",[12,174,14,175],"rag","chromafs","2026-04-18",{"path":178,"title":179,"description":180,"group":139,"section":181,"order":182,"tags":183,"lastUpdated":188},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fcapabilities\u002Fstate-and-knowledge","State and knowledge","What a run may know. One deterministic context builder over application state, knowledge and memory, one owner for every fact, and memory that is written through a tool.","Agentic platform",11,[184,185,186,11,187],"context","memory","knowledge","pgvector","2026-08-31",{"path":190,"title":191,"description":192,"group":139,"section":181,"order":157,"tags":193,"lastUpdated":201},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fcapabilities\u002Ftools-and-integrations","Tools and integrations","A tool is the one way an agent reaches the world. AgencyCore owns the model facing contract, the invoke path, the credentials and the result boundary.",[194,195,196,197,198,199,200],"tools","integrations","mcp","agno","policy","security","idempotency","2026-09-04",{"path":203,"title":204,"description":205,"group":139,"section":181,"order":22,"tags":206,"lastUpdated":210},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fcontract","Platform contract","One platform behind chat, interactive channels, triggers, approvals and background runs, with one Agno runtime, one tool layer, one state layer, and three cross-cutting planes.",[12,14,197,142,194,207,149,208,198,209],"skills","channels","observability","2026-09-02",{"path":212,"title":181,"description":213,"group":139,"section":214,"order":22,"tags":215,"lastUpdated":201},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform","The whole agentic platform on one page - who starts a run, the one boundary every run passes, how the work executes, and what comes back.","System design",[12,14,216,197,142,217,198],"overview","runs",{"path":219,"title":220,"description":221,"group":139,"section":181,"order":222,"tags":223,"lastUpdated":228},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Finterfaces\u002Fagent-access","Agent access (CLI and MCP)","How an outside AI agent reaches AgencyCore. The ac CLI works today as a user seat. An MCP server is planned and not designed.",6,[224,196,12,151,225,226,227],"cli","access","auth","todo","2026-08-18",{"path":230,"title":231,"description":232,"group":139,"section":181,"order":233,"tags":234,"lastUpdated":239},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Finterfaces\u002Fchannel-gateway","Channel gateway","The only layer that knows both an interactive channel and the platform. One message shape converges inbound, one intent shape diverges outbound, and no model call happens here.",3,[208,235,236,237,238,199],"slack","web","identity","sessions","2026-08-30",{"path":241,"title":242,"description":243,"group":139,"section":181,"order":244,"tags":245,"lastUpdated":210},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Finterfaces\u002Ffront-door","Front door","The conversational control layer. It turns a request into one structured decision, then deterministic application code answers or hands work to RunManager.",4,[246,247,197,184,198,217],"front-door","routing",{"path":249,"title":250,"description":251,"group":139,"section":181,"order":32,"tags":252,"lastUpdated":259},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Finterfaces\u002Fsurfaces","Surfaces","Every product surface and its API contract. Web chat goes through the gateway; every schema-native surface calls the domain API.",[253,254,24,255,256,257,258,217,64],"surfaces","api","approvals","prospects","saved-searches","builder","2026-09-03",{"path":261,"title":262,"description":263,"group":139,"section":181,"order":264,"tags":265,"lastUpdated":188},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Finterfaces\u002Ftriggers","Triggers","A Run with no person. Every producer emits one Event, matching is deterministic, and dispatch reuses RunManager, Policy and Inngest.",5,[266,267,142,200],"triggers","events",{"path":269,"title":270,"description":271,"group":139,"section":181,"order":272,"tags":273,"lastUpdated":259},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fidempotency","Idempotency","One durable PostgreSQL key service prevents duplicate effects and freezes mutable input before selected Run starts. A Run start is guarded by a unique index on the Run row.",14,[200,217,194,274,275],"webhooks","reliability",{"path":277,"title":278,"description":279,"group":139,"section":181,"order":280,"tags":281,"lastUpdated":285},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fobservability-and-operations","Observability and operations","One run row, one span tree and one usage meter. Sentry reports system failure; AgencyCore spans explain what the agent did.",13,[209,217,282,283,64,284],"spans","usage","sentry","2026-08-26",{"path":287,"title":288,"description":289,"group":139,"section":181,"order":290,"tags":291,"lastUpdated":295},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fpolicy-and-governance","Policy and governance","One deterministic plane answers may this happen, at three checkpoints, with one grant model, one approval model and one decision log.",12,[198,292,255,293,294],"permissions","limits","governance","2026-08-25",{"path":297,"title":6,"description":298,"group":139,"section":299,"order":22,"tags":300,"lastUpdated":210},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fagentic-crm","The AgencyCore CRM loop for turning signals and discovery into qualified organization prospects, CRM relationships and outreach.","Agentic products",[11,301,51,302,256,35,303,304,87],"lead-generation","intelligence","signals-search","email-sequence",{"path":306,"title":307,"description":308,"group":139,"section":299,"order":264,"tags":309,"lastUpdated":188},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fbuilder-chat","Front door builder chat","Conversational authoring for organization-specific Agent and Workflow definitions, entered through the normal Front Door and backed by the existing DefinitionService.",[310,311,246,12,312,313,198],"authoring","definitions","workflows","templates",{"path":315,"title":316,"description":317,"group":139,"section":181,"order":318,"tags":319,"lastUpdated":201},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fcapability-contracts","Company, People and Signals contracts","The five Phase 7 product capabilities, their bounded inputs, stable references, permissions and results.",21,[320,321,119,51,322],"capabilities","company","contracts",{"path":324,"title":325,"description":326,"group":139,"section":181,"order":327,"tags":328,"lastUpdated":330},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fcapability-scenarios","Capability design scenarios","Normal, failure and recovery cases for the Phase 7 capability contracts, with implementation owners.",22,[320,329,321,119,51],"validation","2026-09-05",{"path":332,"title":333,"description":334,"group":139,"section":299,"order":233,"tags":335,"lastUpdated":210},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Femail-sequence","Email sequence workflow","Envoy durable outreach for one or many people, with fresh context, approvals, reply waits, follow-ups and Nylas transport.",[336,87,34,142,93,255],"email",{"path":338,"title":339,"description":340,"group":139,"section":299,"order":244,"tags":341,"lastUpdated":188},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fgeneral-chat","Front door general chat","The default conversational answer path for AgencyCore. It answers from supplied context, cites what it used, asks when context is insufficient, and delegates real work through the normal Front Door.",[24,246,186,184,247,342],"citations",{"path":344,"title":345,"description":346,"group":139,"section":299,"order":222,"tags":347,"lastUpdated":188},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fhuman-review","Human review inbox","One product page for every agentic action that is paused because a person must authorize an exact proposal. It is a view over the shared approval primitive, not a second review system.",[348,255,349,198,12],"human-review","inbox",{"path":351,"title":352,"description":353,"group":139,"section":299,"order":32,"tags":354,"lastUpdated":259},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fsignals-search","Signals Search","One bounded discovery workflow that finds companies, verifies signals, finds relevant people, and produces evidence-backed organization prospects without prematurely creating CRM records.",[303,355,36,119,51,302,256,11,35],"discovery",{"path":357,"title":358,"description":359,"group":139,"section":299,"order":360,"tags":361,"lastUpdated":188},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fproducts\u002Fworkflow-visualizer","Workflow visualizer","One constrained workflow graph, reused to author a draft, read a published definition, and watch a Run. Build mode edits the draft; run mode overlays Run and span state on the frozen snapshot.",7,[312,362,258,311,217,282,363,255],"visualizer","graph",{"path":365,"title":366,"description":367,"group":139,"section":181,"order":368,"tags":369,"lastUpdated":201},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fruntime\u002Fdefinitions","Runtime definitions","Editable drafts, one published configuration per definition, template forks, deterministic validation, and the Run snapshot that keeps in flight work stable.",8,[149,311,329,370],"publishing",{"path":372,"title":373,"description":374,"group":139,"section":181,"order":375,"tags":376,"lastUpdated":259},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fruntime\u002Fexecution","Runtime execution","The Run record, the Inngest step boundaries, agent segments, workflow nodes, approvals, cancellation, failure handling and live events.",9,[149,217,197,142,255,377,64],"cancellation",{"path":379,"title":380,"description":381,"group":139,"section":181,"order":360,"tags":382,"lastUpdated":285},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fruntime","Agentic runtime","One Run contract, one Agno agent runtime, one deterministic workflow model, and the component boundaries that keep the framework replaceable.",[149,217,197,312,207,142],{"path":384,"title":385,"description":386,"group":139,"section":387,"order":244,"tags":388,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Fcompany-context","Company context","L3. Company state, knowledge and memory are three different things. One deterministic builder turns them into one brief.","Mission Control",[389,390,186,185,184,391,11],"mission-control","company-state","retrieval","2026-08-12",{"path":394,"title":395,"description":396,"group":139,"section":387,"order":22,"tags":397,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Fexperience","Experience","L6. Where a person observes and controls the company, and the one rule that keeps the UI out of the business.",[389,398,399,255,400],"ui","control-plane","activity",{"path":402,"title":403,"description":404,"group":139,"section":387,"order":222,"tags":405,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Ffoundation","Foundation","L1. Generic infrastructure with no business logic in it. The test is that another product could run on it unchanged.",[389,406,407,408,267,409,226,209],"infrastructure","database","queue","storage",{"path":411,"title":387,"description":412,"group":139,"section":214,"order":233,"tags":413,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control","The internal Company OS. Six layers and one policy plane put a person in control of company state and of autonomous execution.",[389,414,14,12,312,198,399],"company-os",{"path":416,"title":417,"description":418,"group":139,"section":387,"order":32,"tags":419,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Fintelligence","Intelligence","L5. The agent is the primitive. A skill is how it works, a tool is how it reaches the world, and the two are never the same thing.",[389,12,207,420,421],"planning","reasoning",{"path":423,"title":424,"description":425,"group":139,"section":387,"order":368,"tags":426,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Fmetrics-and-connectors","Metrics and connectors","A worked example across every layer. Three vendors, one metric pipeline, three views, and the rule that decides what we store.",[389,427,195,428,429,284,430,431],"metrics","stripe","posthog","ingest","dashboards",{"path":433,"title":434,"description":435,"group":139,"section":387,"order":233,"tags":436,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Forchestration","Orchestration","L4. Workflow, run, step, trigger and event. Five nouns that turn a decision into durable execution.",[389,312,217,266,267,437,438],"durability","retry",{"path":440,"title":288,"description":441,"group":139,"section":387,"order":360,"tags":442,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Fpolicy-and-governance","A plane, not a layer. One place decides what an agent may do, under what conditions, and how much. Human approval is one of its three answers.",[389,198,294,255,292,293,443],"audit",{"path":445,"title":191,"description":446,"group":139,"section":387,"order":264,"tags":447,"lastUpdated":392},"\u002Fengineering\u002Fsystem-design\u002Fmission-control\u002Ftools-and-integrations","L2. One contract for every capability. The tool is the only route to the world, and it is where policy, audit and tenancy meet.",[389,194,195,448,449,450],"adapters","registry","credentials",{"path":452,"title":453,"description":454,"group":139,"section":455,"order":22,"tags":456,"lastUpdated":210},"\u002Fengineering\u002Fsystem-design\u002Fworkflows\u002Fcompany-search","Company search","Implementation notes for company.search. Search resolves and gates company identities; enrichment is a separate capability.","Workflows",[321,457,142,42],"search",{"path":459,"title":31,"description":460,"group":139,"section":455,"order":233,"tags":461,"lastUpdated":210},"\u002Fengineering\u002Fsystem-design\u002Fworkflows\u002Fenrichment","Reusable company and people enrichment workflows with canonical Intelligence write-back, existing tier freshness and bounded asynchronous email.",[35,321,119,142],{"path":463,"title":464,"description":465,"group":139,"section":455,"order":32,"tags":466,"lastUpdated":259},"\u002Fengineering\u002Fsystem-design\u002Fworkflows\u002Fpeople-search","People search","Implementation notes for people.search. Bounded company scope and persona gates return selectable person identities without enrichment.",[119,457,142,100],{"path":468,"title":469,"description":470,"group":139,"section":455,"order":244,"tags":471,"lastUpdated":474},"\u002Fengineering\u002Fsystem-design\u002Fworkflows\u002Fsignals-search","Signals search","Superseded. The earlier on-demand buying-signal search component, kept as a record of the design that the agentic platform Signals Search workflow replaces.",[51,12,142,472,473],"intelligence-databases","superseded","2026-08-28",{"path":476,"title":477,"description":478,"group":479,"section":480,"order":481,"tags":482,"lastUpdated":66},"\u002Flearnings\u002Fagentic-sdlc","The agentic SDLC","How AI agents move from autocomplete to owning the loop across the software lifecycle, and why that shifts the bottleneck from coding to verification.","Learnings",null,30,[12,483,484,485,486],"sdlc","engineering","verification","review",{"path":488,"title":489,"description":490,"group":479,"section":480,"order":491,"tags":492,"lastUpdated":498},"\u002Flearnings\u002Fagi-to-asi","From AGI to ASI","What lies beyond human-level AI. The four technological pathways from AGI to artificial superintelligence, the formal ceiling that bounds them, and the six bottlenecks that could stall the climb - distilled from the DeepMind report.",50,[493,494,495,496,497],"ai-futures","asi","agi","scaling","recursive-self-improvement","2026-06-19",{"path":500,"title":501,"description":502,"group":479,"section":480,"order":503,"tags":504,"lastUpdated":66},"\u002Flearnings\u002Fai-native-company-playbook","AI native company playbook","Why AI should be the operating system your company runs on, not a tool it uses, and the concrete practices that follow - closed loops, a queryable org, software factories, and token maxing.",40,[505,506,12,507,508],"ai-native","company-building","gtm","founders",{"path":510,"title":511,"description":512,"group":479,"section":480,"order":157,"tags":513,"lastUpdated":54},"\u002Flearnings\u002Fbuying-intent-signals","Buying intent signals","How buyers leak their intent before they ever fill in a form, and how to read those signals before the window closes.",[514,51,507,515],"intent","sales",{"path":517,"title":518,"description":519,"group":479,"section":480,"order":520,"tags":521,"lastUpdated":54},"\u002Flearnings\u002Fcold-outbound-system","Cold outbound system","A high-level study of an open-source 29-skill cold email system, organized into five sequential tracks from ICP to iteration.",20,[522,523,507,524],"outbound","cold-email","systems",{"path":526,"title":527,"description":528,"group":479,"section":480,"order":529,"tags":530,"lastUpdated":532},"\u002Flearnings\u002Fswan-gtm-skills-architecture","Swan GTM skills architecture","A research note on Swan AI's foundations and maps model for GTM agents, with ASCII diagrams and ideas AgencyCore can borrow.",60,[507,12,73,531,14],"swan","2026-07-01",{"path":534,"title":535,"description":536,"group":387,"section":480,"order":272,"tags":537,"lastUpdated":43},"\u002Fmission-control\u002Fciops-agent","CIOps agent","High-level system architecture and design notes for the Mission Control CIOps agent.",[389,12,538,14],"ciops",{"path":540,"title":541,"description":542,"group":387,"section":480,"order":182,"tags":543,"lastUpdated":43},"\u002Fmission-control\u002Fcostops-agent","CostOps agent","High-level system architecture and design notes for the Mission Control CostOps agent.",[389,12,544,14],"finops",{"path":546,"title":547,"description":548,"group":387,"section":480,"order":520,"tags":549,"lastUpdated":54},"\u002Fmission-control\u002Fdashboard","Dashboard","The Mission Control product UI - a dark cockpit with a fleet-nav rail, company-state grid, a working escalation queue, live ledger and a global kill switch.",[389,12,550,398],"dashboard",{"path":552,"title":553,"description":554,"group":387,"section":480,"order":280,"tags":555,"lastUpdated":43},"\u002Fmission-control\u002Fproduct-analytics-agent","ProductAnalytics agent","High-level system architecture and design notes for the Mission Control ProductAnalytics agent.",[389,12,556,14],"product-analytics",{"path":558,"title":559,"description":560,"group":387,"section":480,"order":290,"tags":561,"lastUpdated":43},"\u002Fmission-control\u002Frevenueops-agent","RevenueOps agent","High-level system architecture and design notes for the Mission Control RevenueOps agent.",[389,12,562,14],"revops",{"path":564,"title":214,"description":565,"group":387,"section":480,"order":157,"tags":566,"lastUpdated":54},"\u002Fmission-control\u002Fsystem-design","One screen for the whole company, watched by a guardrailed fleet of ops agents that explain, propose, act and learn overnight.",[389,12,544,14],{"path":568,"title":569,"description":570,"group":571,"section":480,"order":32,"tags":572,"lastUpdated":578},"\u002Fproduct-design\u002Fonboarding-flow","Onboarding flow","Product design for the signup wizard and how TAM building folds into it. Analyzes the flow today (account, profile, company), the gap (no ICP, empty dashboard), and the integration of a new \"who you sell to\" ICP step plus a build-and-reveal screen that lands the user on a populated, ranked list.","Product Design",[573,61,574,575,576,577],"onboarding","tam","activation","ux","user-journey","2026-06-11",{"path":580,"title":581,"description":582,"group":571,"section":480,"order":233,"tags":583,"lastUpdated":592},"\u002Fproduct-design\u002Fpricing-entitlements","Pricing tiers, entitlements and usage credits","Specification for subscription tiers with gated platform access: composable plan entitlements, a unified usage-credit currency, plan-sourced limits, per-module trials and a two-ticket delivery plan built on the Stripe billing foundation. Written for discussion; the Linear document is the canonical copy with ticket links.",[584,585,586,587,588,589,590,591],"pricing","entitlements","billing","credits","subscriptions","plans","seats","trials","2026-07-06",{"path":594,"title":595,"description":596,"group":571,"section":480,"order":233,"tags":597,"lastUpdated":578},"\u002Fproduct-design\u002Fsales-signals-ux","Designing Signals","Product design for the sales-signals experience in ac-frontend: the 14-type taxonomy and its color system, the anatomy of a signal card across four densities, the 0-10 lead score scale, the origin tag (sonar pull vs proactive push), the seven surfaces where signals render (launchpad, sonar app, company detail, timeline, activities, data layer, Envoy), and the interaction rules that keep them consistent.",[51,576,598,11,42,599,600,601,602],"design-system","lead-score","origin","pull","push",{"path":604,"title":605,"description":606,"group":607,"section":608,"order":244,"tags":609,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Factivities","Activities","Deep dive on crm_activities, the interaction + task log of the CRM — where it is served from, how a row is born and read, and its full schema, relationships and rules.","Proprietary data","CRM",[11,610,611,612,613],"activities","tasks","data-model","schema",{"path":615,"title":616,"description":617,"group":607,"section":608,"order":264,"tags":618,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Fcommunications","Communications","Deep dive on crm_communications and crm_communication_events, the unified email\u002Fcall\u002Fmessage log and its per-message engagement tracking — where it is served from, the outbound message lifecycle, and the full schema, relationships and rules.",[11,619,336,620,612],"communications","engagement",{"path":622,"title":623,"description":624,"group":607,"section":608,"order":22,"tags":625,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Fcompanies","Companies","Deep dive on crm_companies, the account record at the centre of the CRM — where it is served from, how a row is born and read, and its full schema, relationships and rules.",[11,36,612,613,14],{"path":627,"title":628,"description":629,"group":607,"section":608,"order":233,"tags":630,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Fdeals","Deals","Deep dive on the deal pipeline — crm_deals, crm_pipeline_stages and crm_pipeline_config. Where it is served from, the life of a deal, and its full schema, relationships and rules.",[11,631,632,612,613],"deals","pipeline",{"path":634,"title":635,"description":636,"group":607,"section":608,"order":222,"tags":637,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Flists","Lists","Deep dive on crm_lists and crm_list_members, the static or dynamic member collections of the CRM — where they are served from, how a list and its members come to be and are read, and their schema, relationships and rules.",[11,638,639,612,613],"lists","segments",{"path":641,"title":642,"description":643,"group":607,"section":608,"order":32,"tags":644,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Fpeople","People","Deep dive on crm_people, the contact record of the CRM — where it is served from, how a row is born and read, and its full schema, relationships and rules.",[11,119,645,612,613],"contacts",{"path":647,"title":648,"description":649,"group":607,"section":608,"order":368,"tags":650,"lastUpdated":43},"\u002Fproprietary-data\u002Fcrm\u002Fsaved-filters","Saved filters","Deep dive on crm_saved_filters, the named reusable filter snapshots over the company, person and signal list views — where it is served from, how a saved view is born and applied, and its full schema, relationships and rules.",[11,651,652,612,613],"saved-filters","views",{"path":654,"title":655,"description":656,"group":607,"section":608,"order":360,"tags":657,"lastUpdated":578},"\u002Fproprietary-data\u002Fcrm\u002Fsignals","Signals","Deep dive on the signals tables - signals, company_signals and person_signals, the CRM's sales-intelligence layer. Where signals are served from, how one is born and attached, and the full schema, relationships and rules.",[11,51,302,612,613],{"path":659,"title":660,"description":661,"group":607,"section":662,"order":22,"tags":663,"lastUpdated":43},"\u002Fproprietary-data\u002Fintelligence-databases\u002Fcompany-intelligence-database","Company Intelligence Database","Decided architecture for ENG-669, the cross-org company intelligence layer that acts as a read-through cache in front of enrichment providers, with public-facts-only privacy and provenance-tracked write-back.","Intelligence databases",[14,60,36,51,38,664],"eng-669",{"path":666,"title":667,"description":668,"group":607,"section":662,"order":244,"tags":669,"lastUpdated":578},"\u002Fproprietary-data\u002Fintelligence-databases\u002Forg-signal-feed","Org Signal Feed","The per-org activation layer on top of the shared signals store. One immutable intel_signals row fans out to many orgs through scoring (signal-type weight times ICP fit times recency decay) and materializes as ranked, tiered rows in intel_org_signal_feed - the only org-scoped, RLS-per-org table of the signal stack, the door the launchpad, inbox and digest all read through. Signals enter by two ingest classes - a user's sonar pull (ungated) or an automated push (gated by threshold plus an optional competitor-ICP check) - logged in intel_signal_ingests, and each feed row records its origin.",[14,60,51,670,53,671,672,575,430,601,602,600],"feed","decay","rls",{"path":674,"title":675,"description":676,"group":607,"section":662,"order":32,"tags":677,"lastUpdated":578},"\u002Fproprietary-data\u002Fintelligence-databases\u002Fpeople-intelligence-database","People Intelligence Database","Decided architecture for the cross-org people intelligence layer - a read-through cache in front of headhunter research and Hunter email lookups, with LinkedIn-URL identity, append-only employment edges, per-tier freshness stamps on the flat profile, shared intel_sources provenance, unified intel_signals, and a GDPR erasure path.",[14,60,119,51,38,100],{"path":679,"title":680,"description":681,"group":607,"section":662,"order":233,"tags":682,"lastUpdated":578},"\u002Fproprietary-data\u002Fintelligence-databases\u002Fsignals-intelligence-database","Signals Intelligence Database","Decided v1 architecture for the unified signal store - one polymorphic append-only intel_signals table that holds both company and person signals, with a shared taxonomy, source-ranked provenance, an intel_signal_ingests log that records which pipeline found each signal, decay at read time, and a person-to-company rollup so a champion job change surfaces on the company feed.",[14,60,51,683,671,684,670,685,41,601,602],"polymorphic","taxonomy","ingests",{"path":687,"title":688,"description":689,"group":607,"section":480,"order":9,"tags":690,"lastUpdated":16},"\u002Fproprietary-data\u002Foverview","Data Layer Overview","The AgencyCore data layer in one map - the org-scoped CRM plane in production today and the global intelligence plane designed to sit in front of it, with interactive diagrams of both, the end-to-end data flow, freshness and precedence rules, the privacy seam, and the rollout path.",[691,14,60,11,51,38,25,216],"data-layer",{"path":693,"title":694,"description":695,"group":696,"section":480,"order":9,"tags":697,"lastUpdated":54},"\u002Froadmap","Roadmap - June 2026","June 2026 product plan across four themes. The spine is moving our agents onto an isolated sandbox runtime and rebuilding the core agents and workflows on it, then standing up a read-through intelligence data store and shipping the Stripe billing system. Knowledge base, assistant, and credit tracking carry into the July roadmap.","Roadmap",[698,420],"roadmap",{"path":700,"title":701,"description":702,"group":696,"section":480,"order":22,"tags":703,"lastUpdated":54},"\u002Froadmap\u002Fjuly-2026","Roadmap - July 2026","July 2026 product plan across three themes, all carried over from June. Building on June's sandbox runtime, July grounds the agents in a knowledge base, launches the AI chat assistant, and meters every action with per-action credit tracking that reconciles into the Stripe billing system shipped in June.",[698,420],{"path":705,"title":706,"description":707,"group":696,"section":480,"order":32,"tags":708,"lastUpdated":532},"\u002Froadmap\u002Fjune-2026-slides","Roadmap slides - June 2026","Board-review slide deck for the June 2026 product roadmap, rendered directly from the original PPTX in the docs site.",[698,420,709],"slides",{"path":711,"title":712,"description":713,"group":714,"section":8,"order":520,"tags":715,"lastUpdated":16},"\u002Fsymphony\u002Fagents\u002Fdevops-agent","DevOps agent","Interactive design for a Slack-first Symphony DevOps agent that wraps production promotion, rollback, audit, and operational jobs behind policy gates, typed runbooks, and an auditable ledger.","Symphony",[716,235,717,718,719,720],"symphony","devops","production","runbooks","operations",{"path":722,"title":723,"description":724,"group":714,"section":8,"order":157,"tags":725,"lastUpdated":16},"\u002Fsymphony\u002Fagents\u002Foncall-agent","Oncall agent","Interactive design for a Symphony oncall agent that turns Sentry incidents into rich Linear tickets, investigates with Codex, opens fix PRs, and resolves Sentry after merge.",[716,284,726,727,728,729],"linear","oncall","incident-response","codex",{"path":731,"title":732,"description":733,"group":714,"section":734,"order":157,"tags":735,"lastUpdated":16},"\u002Fsymphony\u002Fhousekeeping\u002Fcodex-vacuum","Codex vacuum","Interactive design for the Symphony housekeeping timer that checkpoints and vacuums Codex sqlite stores on the VPS.","Housekeeping",[716,736,737,729,738,739],"timed-jobs","housekeeping","sqlite","vps",{"path":741,"title":742,"description":743,"group":714,"section":734,"order":481,"tags":744,"lastUpdated":16},"\u002Fsymphony\u002Fhousekeeping\u002Fhost-cleanup","Host cleanup","Interactive design for the Symphony housekeeping timer that removes stale \u002Ftmp debris, vacuums the journal, and optionally cleans the apt package cache.",[716,736,737,739,745,746],"disk","cleanup",{"path":748,"title":749,"description":750,"group":714,"section":734,"order":520,"tags":751,"lastUpdated":16},"\u002Fsymphony\u002Fhousekeeping\u002Fworkspace-cleanup","Workspace cleanup","Interactive design for the Symphony housekeeping timer that prunes idle per-issue workspaces after their TTL.",[716,736,737,752,746,739],"workspaces",{"path":754,"title":755,"description":756,"group":714,"section":480,"order":9,"tags":757,"lastUpdated":66},"\u002Fsymphony","Symphony orchestration","How AgencyCore runs OpenAI Symphony as a long-running daemon that turns Linear tickets into isolated, autonomous Codex runs, reviewed by Claude and merged by humans. High-level workflow, system architecture, and the engineer playbook.",[716,729,726,758,111,739,759,760],"claude-review","qa","automation",{"path":762,"title":763,"description":764,"group":714,"section":214,"order":22,"tags":765,"lastUpdated":392},"\u002Fsymphony\u002Fsystem-design\u002Fhigh-level-design","High-level design","The Symphony daemon end to end — the standing agent workforce and its label-routed workflows, then the runtime that polls, dispatches, runs and writes back.",[716,14,111,12,729,726,766],"systemd",{"path":768,"title":769,"description":770,"group":714,"section":771,"order":503,"tags":772,"lastUpdated":16},"\u002Fsymphony\u002Ftimed-jobs\u002Fdaily-security-agent","Daily security agent","Interactive design for a report-only Symphony timed job that reviews the last 24h of commits, scans the system for vulnerabilities, and opens focused follow-up tickets.","Timed jobs",[716,199,736,729,773,774,775],"semgrep","threat-model","ownership",{"path":777,"title":778,"description":779,"group":714,"section":771,"order":481,"tags":780,"lastUpdated":16},"\u002Fsymphony\u002Ftimed-jobs\u002Fdaily-sentry-triage","Daily Sentry triage","Interactive design for the Symphony timed job that performs read-only Sentry triage, deduplicates existing tracked clusters, and creates focused ENG bugs for new actionable errors.",[716,736,284,209,781,726],"triage",{"path":783,"title":784,"description":785,"group":714,"section":771,"order":157,"tags":786,"lastUpdated":16},"\u002Fsymphony\u002Ftimed-jobs\u002Fnightly-local-staging-e2e","Nightly local staging E2E","Interactive design for the Symphony timed job that seeds local Supabase, runs ac-frontend Playwright E2E against the local staging stack, uploads evidence, and cleans artifacts.",[716,736,787,788,789,790],"e2e","playwright","staging","frontend",{"path":792,"title":793,"description":794,"group":714,"section":771,"order":520,"tags":795,"lastUpdated":16},"\u002Fsymphony\u002Ftimed-jobs\u002Fnightly-staging-qa","Nightly staging QA","Interactive design for the Symphony timed job that seeds a staging QA Linear issue, runs an agent-browser crawl, validates feature-map coverage, and files focused follow-up work.",[716,736,789,759,796,726],"agent-browser",{"id":798,"title":278,"body":799,"customComponent":480,"description":279,"extension":4269,"group":139,"lastUpdated":285,"meta":4270,"navigation":1846,"order":280,"path":277,"related":4271,"section":181,"seo":4276,"stem":4277,"tags":4278,"__hash__":4279},"docs\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fobservability-and-operations.md",{"type":800,"value":801,"toc":4237},"minimark",[802,806,810,821,824,860,867,872,882,888,891,912,917,927,959,973,979,998,1008,1024,1028,1034,1041,1051,1057,1068,1072,1075,1099,1102,1126,1143,1160,1164,1176,1189,1195,1198,1201,1204,1217,1224,1241,1256,1276,1287,1303,1314,1332,1336,1342,1345,1350,1357,1373,1387,1391,1394,1465,1478,1515,1536,1554,1560,1566,1572,1586,1605,1628,1631,1635,1638,1693,1696,1700,1710,1733,1755,1761,1771,1775,1791,1797,1804,1810,1877,1894,1919,1940,1950,1957,1963,1969,1985,1989,2136,2152,2161,2165,2183,2189,2214,2231,2237,2244,2248,2251,2257,2260,2263,2280,2298,2308,2312,2401,2428,2439,2448,2462,2472,2484,2493,2496,2508,2523,2546,2550,2553,2559,2562,2603,2606,2612,2625,2631,2634,2637,2663,2666,2669,2672,2678,2696,2700,2703,2707,2717,2720,2723,2730,2767,2770,2774,2777,2877,2893,2896,2930,2946,2949,2952,3049,3057,3060,3064,3109,3115,3121,3124,3128,3134,3518,3530,3562,3566,3569,3575,3578,3598,3604,3618,3624,3633,3643,3649,3652,3655,3661,3664,3667,3827,3831,3841,3845,4014,4018,4127,4131,4233],[803,804,278],"h1",{"id":805},"observability-and-operations",[807,808,809],"p",{},"Observability is a plane. Every layer writes into it, and no layer needs it to function.",[811,812,818],"pre",{"className":813,"code":815,"language":816,"meta":817},[814],"language-text","Policy decides before the work.\nObservability records what the work did.\n","text","",[819,820,815],"code",{"__ignoreMap":817},[807,822,823],{},"Two systems, and one sentence each.",[825,826,827,840],"table",{},[828,829,830],"thead",{},[831,832,833,837],"tr",{},[834,835,836],"th",{},"System",[834,838,839],{},"Answers",[841,842,843,852],"tbody",{},[831,844,845,849],{},[846,847,848],"td",{},"Sentry",[846,850,851],{},"Our software or our infrastructure broke",[831,853,854,857],{},[846,855,856],{},"Run spans",[846,858,859],{},"The agent did this, in this order, and it cost this much",[807,861,862,863,866],{},"A failed vendor call is normal agent work, so it belongs in a span. A ",[819,864,865],{},"KeyError"," in our handler is a defect, so it belongs in Sentry.",[868,869,871],"h2",{"id":870},"one-run-one-span-tree","One run, one span tree",[807,873,874,877,878,881],{},[819,875,876],{},"agent.runs"," is the product run table. A ",[819,879,880],{},"agent.spans"," row is one unit of work inside a run.",[811,883,886],{"className":884,"code":885,"language":816,"meta":817},[814],"agent.spans\n  span_id\n  organization_id      the tenancy boundary; RLS reads it\n  run_id\n  root_run_id          copied from the run, so one query builds a whole tree\n  parent_span_id\n  kind\n  name\n  status               running | ok | error\n  started_at \u002F ended_at\n  duration_ms          generated from the two above; never written\n  updated_at           a BEFORE UPDATE trigger writes it; the reconnect filter reads it\n  input  jsonb         bounded and redacted\n  output jsonb         bounded and redacted\n  usage_id             the usage row this work produced, when there is one\n  attributes jsonb     attempt, replayed, policy, step_path, and similar\n  error jsonb\n",[819,887,885],{"__ignoreMap":817},[807,889,890],{},"Two fields are easy to leave out and expensive to add later.",[892,893,894,904],"ul",{},[895,896,897,903],"li",{},[898,899,900],"strong",{},[819,901,902],{},"organization_id"," is the tenancy boundary. Every product table in this platform carries it, and a span holds tool arguments, so it needs the same protection as the data it describes.",[895,905,906,911],{},[898,907,908],{},[819,909,910],{},"root_run_id"," is what makes the run explorer one query. A workflow with three child runs is one tree to a person. Without this field the explorer walks the run graph first, then queries spans per run.",[913,914,916],"h3",{"id":915},"a-child-run-hangs-from-the-node-that-started-it","A child run hangs from the node that started it",[807,918,919,922,923,926],{},[819,920,921],{},"spans_parent_same_tree_fk"," allows a parent in another Run of the same tree, and the ",[819,924,925],{},"run"," span of a child Run is the one span that uses it. Without that edge the tree has as many roots as it has Runs, and a person reading a workflow cannot see which node produced which child.",[807,928,929,932,933,936,937,940,941,944,945,948,949,952,953,955,956,958],{},[898,930,931],{},"The parent span is the span that was current at the node, and the node opens none of its own."," The eight kinds below are closed, and none of them names an ",[819,934,935],{},"agent"," node or a ",[819,938,939],{},"subworkflow"," node. So ",[819,942,943],{},"parent_span_id"," is ",[819,946,947],{},"current_span_id()"," at the node: the enclosing ",[819,950,951],{},"parallel"," span, or the run's own ",[819,954,925],{}," span. The child's ",[819,957,925],{}," span carries the node ID as its name, so a person still reads which node produced which child, and no ninth kind is needed. Read \"the node span\" below as that span.",[807,960,961,964,965,968,969,972],{},[898,962,963],{},"The child worker learns the parent span from the Run row."," ",[819,966,967],{},"RunSource.parent_span_id"," reaches ",[819,970,971],{},"RunManager"," on the start command, and the child executes in another process, sometimes minutes later and sometimes after a crash. So the value is a column:",[811,974,977],{"className":975,"code":976,"language":816,"meta":817},[814],"agent.runs\n  parent_span_id  uuid   the node span that started this Run; null for every other source\n",[819,978,976],{"__ignoreMap":817},[807,980,981,964,984,986,987,989,990,993,994,997],{},[898,982,983],{},"It carries no foreign key, and retention is the reason.",[819,985,880],{}," is kept 90 days and ",[819,988,876],{}," is kept 13 months, so the parent span is deleted while the child Run is still the product record. ",[819,991,992],{},"ON DELETE CASCADE"," would delete that Run, and ",[819,995,996],{},"ON DELETE SET NULL"," would break the shape rule below.",[807,999,1000,1003,1004,1007],{},[898,1001,1002],{},"The shape rule is an equivalence."," A ",[819,1005,1006],{},"workflow_step"," Run carries a parent span, and every other source carries none. Written as an implication it forbids the value elsewhere and never requires it here, so every child Run would be free to land with a null parent and sever its own tree with nothing raising.",[807,1009,1010,1017,1018,1020,1021,1023],{},[898,1011,1012,1013,1016],{},"A trigger proves the span at the ",[819,1014,1015],{},"INSERT",", and freezes it after."," Without it the value is proved one write later, when the child's own ",[819,1019,925],{}," span meets ",[819,1022,921],{},". By then the Run row is committed and its dispatch event is sent, so the child fails mid flight instead of never starting. The same check refuses a span of another tenant.",[913,1025,1027],{"id":1026},"span-kinds","Span kinds",[811,1029,1032],{"className":1030,"code":1031,"language":816,"meta":817},[814],"run         the root span of one run\nsegment     one agent segment\nllm         one model call\ntool        one tool call\napproval    one human wait\nwait        one event wait or delay\nbranch      one condition choice\nparallel    one parallel container\n",[819,1033,1031],{"__ignoreMap":817},[807,1035,1036,1037,1040],{},"A ",[819,1038,1039],{},"sequence"," container writes no span, because ordering adds no timing a person can use.",[807,1042,1043,1044,1046,1047,1050],{},"There is no ",[819,1045,198],{}," kind. An allowed action is an attribute on the span of the work it gated, and a refusal already writes a ",[819,1048,1049],{},"agent.policy_decisions"," row. An admission decision writes a row whatever it answers, because admission runs once per run. A span per decision would double the writes of the hot path and explain nothing new.",[807,1052,1043,1053,1056],{},[819,1054,1055],{},"eval"," kind either. An evaluation run is a run. It writes the same kinds, with its source in the run row.",[807,1058,1059,1060,1063,1064,1067],{},"Do not add a durable event table beside spans. Events such as ",[819,1061,1062],{},"tool.called"," and ",[819,1065,1066],{},"tool.failed"," are projections of the span lifecycle.",[868,1069,1071],{"id":1070},"who-opens-a-span","Who opens a span",[807,1073,1074],{},"The code doing the work owns the span. Application code uses one scoped context manager and nothing else.",[811,1076,1080],{"className":1077,"code":1078,"language":1079,"meta":817,"style":817},"language-python shiki shiki-themes github-dark","async with recorder.span(kind='tool', name=tool.name, input=args) as span:\n    result = await handler.execute(ctx, args)\n    span.set_output(result)\n","python",[819,1081,1082,1089,1094],{"__ignoreMap":817},[1083,1084,1086],"span",{"class":1085,"line":22},"line",[1083,1087,1088],{},"async with recorder.span(kind='tool', name=tool.name, input=args) as span:\n",[1083,1090,1091],{"class":1085,"line":32},[1083,1092,1093],{},"    result = await handler.execute(ctx, args)\n",[1083,1095,1096],{"class":1085,"line":233},[1083,1097,1098],{},"    span.set_output(result)\n",[807,1100,1101],{},"The recorder opens the span, completes it, or fails it and re-raises the original exception. Tracing never decides a retry and never changes control flow.",[807,1103,1104,1107,1108,1110,1111,1113,1114,1117,1118,1121,1122,1125],{},[898,1105,1106],{},"A span that wraps more than one Inngest step needs an open and a close."," The block above closes where it opens, and a ",[819,1109,951],{}," span or a ",[819,1112,925],{}," span cannot: the work between them is several durable steps, and the SDK re-executes the function body once per step. So ",[819,1115,1116],{},"SpanRecorder"," also offers ",[819,1119,1120],{},"open_span()",", which answers a span ID, and ",[819,1123,1124],{},"close_span()",", which takes it. Each half runs inside its own step, so a replay writes neither again. Everything that fits in one step keeps the block, and that is almost everything.",[807,1127,1128,1139,1140,1142],{},[898,1129,1130,1131,1134,1135,1138],{},"The input is an argument of ",[819,1132,1133],{},"span()",", and there is no ",[819,1136,1137],{},"set_input","."," The input of a unit of work is known before the work starts, and it is written by the ",[819,1141,1015],{}," that opens the span. A setter would offer a second moment to write it, and the only span that reached that moment would be one the crash already left open.",[807,1144,1145,964,1148,1151,1152,1155,1156,1159],{},[898,1146,1147],{},"A payload that is neither a dict nor a list is wrapped before it is bounded.",[819,1149,1150],{},"bound()"," raises ",[819,1153,1154],{},"TypeError"," on any other type, and a span output is often a string or a number: a model turn answers text, and a tool answers a count. The recorder wraps such a value as ",[819,1157,1158],{},"{'value': payload}"," and bounds the wrapper. It does not pass the bare value, and it does not skip the boundary.",[913,1161,1163],{"id":1162},"the-recorder-also-writes-the-heartbeat","The recorder also writes the heartbeat",[807,1165,1166,1168,1169,1172,1173,1138],{},[819,1167,971],{}," stamps ",[819,1170,1171],{},"heartbeat_at"," at the insert and at each lifecycle write, and a segment may then approach its 90 second wall clock. Nothing writes it between those two moments, so the reaper would have no safe ",[819,1174,1175],{},"stale_after",[807,1177,1178,964,1183,1168,1186,1188],{},[898,1179,1180,1182],{},[819,1181,1116],{}," is the only writer between the claim and the finalize. It is not the only writer of the column.",[819,1184,1185],{},"build_transition_values",[819,1187,1171],{}," on every lifecycle write, because the insert and each transition are the two moments no span covers. Read \"one writer\" as \"one writer per moment\", and never as \"one statement in the code base\".",[811,1190,1193],{"className":1191,"code":1192,"language":816,"meta":817},[814],"span opened -> the process wrote this run more than 10s ago\n               -> UPDATE agent.runs SET heartbeat_at = now() WHERE id = :run_id\n            -> the process wrote this root more than 60s ago\n               -> UPDATE agent.runs SET heartbeat_at = now() WHERE id = :root_run_id\n",[819,1194,1192],{"__ignoreMap":817},[807,1196,1197],{},"The window is tested in the process and never in the statement. See Write cost\nbelow for the measurement that removed the SQL guard.",[807,1199,1200],{},"The recorder is the right owner because it already fires on every unit of work and already holds both IDs.",[807,1202,1203],{},"The root row is written because a workflow parent suspended on a child writes no spans of its own. Its child's work is what proves the tree is alive.",[807,1205,1206,1209,1210,1213,1214,1216],{},[898,1207,1208],{},"The two windows are different numbers on purpose."," A run's own row is touched by one worker, so 10 seconds is free. A root row is touched by ",[898,1211,1212],{},"every"," worker in its tree, and an email sequence has five hundred of them. At 10 seconds those five hundred contend on one row for a write that almost always changes nothing. At 60 seconds the root is touched at most once a minute per process however wide the tree grows, and ",[819,1215,1175],{}," is 120 seconds, so the reaper is still nowhere near it.",[807,1218,1219,1220,1138],{},"This is the one place where tracing carries a liveness duty. It is still not control flow: a failed heartbeat write reports to Sentry and changes nothing. See ",[1221,1222,1223],"a",{"href":372},"runtime execution",[807,1225,1226,1227,1230,1231,1063,1234,1236,1237,1240],{},"Current span identity lives in a context variable, read through three accessors: ",[819,1228,1229],{},"current_run_id()",", ",[819,1232,1233],{},"current_root_run_id()",[819,1235,947],{},". No function grows another trace parameter, and no type is named ",[819,1238,1239],{},"Context"," here.",[807,1242,1243,964,1246,1249,1250,1252,1253,1255],{},[898,1244,1245],{},"One writer binds the run, and it is a scope rather than a setter.",[819,1247,1248],{},"run_scope(run_id, root_run_id, organization_id, parent_span_id=None)"," is a context manager. A bare setter has no exit, so a worker that serves the next run reads the identity of the previous one. ",[819,1251,1133],{}," binds ",[819,1254,947],{}," the same way, for the body of its own block.",[807,1257,1258,1261,1262,1265,1266,1270,1271,1063,1273,1275],{},[898,1259,1260],{},"The scope wraps the whole Inngest function body, not the claim step."," A context variable set inside ",[819,1263,1264],{},"step.run('claim', ...)"," is gone when segment 1 starts, and every span of segments 1 to ",[1267,1268,1269],"em",{},"n"," then writes a null run id. The scope needs ",[819,1272,910],{},[819,1274,943],{},", which only the Run row carries, so the order is fixed: claim, then enter the scope, then run the loop inside it.",[807,1277,1278,1283,1284,1286],{},[898,1279,1280,1282],{},[819,1281,943],{}," is how a child Run joins its parent's tree."," The child executes in another process, so it reads the value from its own Run row rather than inheriting a context variable. Seeding the scope with it makes the child's own ",[819,1285,925],{}," span parent into the node, and every span below nests as usual.",[807,1288,1289,964,1292,1151,1295,1298,1299,1302],{},[898,1290,1291],{},"Reset the token in the task that set it.",[819,1293,1294],{},"ContextVar.reset",[819,1296,1297],{},"ValueError"," in another context. ",[819,1300,1301],{},"async with"," satisfies this rule, and a token stored on an object does not.",[807,1304,1305,1306,1309,1310,1313],{},"That last part is deliberate. In this platform ",[898,1307,1308],{},"context means what a model may know",", and ",[819,1311,1312],{},"ContextBrief"," is the only thing that carries it. Ambient execution identity is a different idea, so it gets accessors and no noun.",[807,1315,1043,1316,1319,1320,1323,1324,1327,1328,1331],{},[819,1317,1318],{},"trace_tool"," or ",[819,1321,1322],{},"trace_llm"," helper hierarchy in V1. ",[819,1325,1326],{},"kind"," plus ",[819,1329,1330],{},"name"," is enough.",[868,1333,1335],{"id":1334},"durable-first-live-second","Durable first, live second",[811,1337,1340],{"className":1338,"code":1339,"language":816,"meta":817},[814],"open     -> INSERT agent.spans (status running) -> publish span.started\ncomplete -> UPDATE agent.spans (status ok)      -> publish span.completed\nfail     -> UPDATE agent.spans (status error)   -> publish span.failed\n",[819,1341,1339],{"__ignoreMap":817},[807,1343,1344],{},"Live publishing is best effort. When delivery fails, the durable span is still correct, and the client recovers by refetching.",[807,1346,1347,1349],{},[1221,1348,373],{"href":372}," owns the live event vocabulary, the Redis and SSE path, and the reconnect rule. This page owns only the ordering guarantee above.",[807,1351,1352,1353,1356],{},"If a span write itself fails, report it to Sentry and let the business operation succeed. A tracing fault must never turn a completed send into a failure. ",[898,1354,1355],{},"Usage is the exception",", and the next section says why.",[807,1358,1359,1362,1363,1365,1366,1368,1369,1372],{},[898,1360,1361],{},"A failed open leaves the parent span current."," The span has no row, so a child that took it as a parent writes a ",[819,1364,943],{}," that no row satisfies, and ",[819,1367,921],{}," answers ",[819,1370,1371],{},"23503"," on every child under it. One lost span becomes a lost subtree, and the rule above says a tracing fault costs the caller nothing. The block still runs, and its close is skipped.",[807,1374,1375,1378,1379,1382,1383,1386],{},[898,1376,1377],{},"The live publisher arrived with Phase 2."," The ordering above is the rule the recorder is built to. Phase 1 shipped the seam and a default that published nothing, so no span write waited on a component that did not exist yet. Phase 2 added ",[819,1380,1381],{},"RedisRunEventPublisher",", and a recorder built with ",[819,1384,1385],{},"None"," still publishes nothing.",[868,1388,1390],{"id":1389},"retries-replays-and-orphans","Retries, replays and orphans",[807,1392,1393],{},"A crash leaves marks in the tree. The design chooses honest marks over tidy ones.",[825,1395,1396,1406],{},[828,1397,1398],{},[831,1399,1400,1403],{},[834,1401,1402],{},"Case",[834,1404,1405],{},"What the tree shows",[841,1407,1408,1420,1438,1453],{},[831,1409,1410,1413],{},[846,1411,1412],{},"A step fails and Inngest retries it",[846,1414,1415,1416,1419],{},"The failed span stays. The next attempt writes new spans with ",[819,1417,1418],{},"attributes.attempt"," one higher",[831,1421,1422,1425],{},[846,1423,1424],{},"A restarted segment returns a completed journal result",[846,1426,1036,1427,1430,1431,1434,1435],{},[819,1428,1429],{},"tool"," span with ",[819,1432,1433],{},"attributes.replayed = true",", near zero duration and no ",[819,1436,1437],{},"usage_id",[831,1439,1440,1443],{},[846,1441,1442],{},"The worker dies mid span",[846,1444,1445,1446,1449,1450],{},"The span stays ",[819,1447,1448],{},"running"," with no ",[819,1451,1452],{},"ended_at",[831,1454,1455,1458],{},[846,1456,1457],{},"The worker dies while the run waits",[846,1459,1460,1461,1464],{},"The run stays ",[819,1462,1463],{},"waiting"," with its own deadline as the only clock",[807,1466,1467,1468,1471,1472,1063,1475,1138],{},"The third row needs an owner, or a tree keeps a span that spins for ever. The ",[819,1469,1470],{},"run.reaper"," already fails runs with a stale heartbeat, and it closes their open spans in the same pass, with ",[819,1473,1474],{},"status = error",[819,1476,1477],{},"error.code = worker_lost",[807,1479,1480,1481,1483,1484,1487,1488,1491,1492,1495,1496,1499,1500,1503,1504,1507,1508,1511,1512,1514],{},"The fourth row has its own owner, and it is a third sweep in the same cron. A ",[819,1482,1463],{}," run writes no heartbeat, so the read above never sees it. The ",[898,1485,1486],{},"wait sweep"," reads ",[819,1489,1490],{},"agent.runs.waiting_expires_at"," instead and ends a run past that clock plus ",[819,1493,1494],{},"INNGEST_AGENTIC_WAIT_GRACE_SECONDS",", under ",[819,1497,1498],{},"error.code = wait_abandoned",". It runs ",[898,1501,1502],{},"before"," the second span sweep below, so ",[819,1505,1506],{},"close_orphans()"," stamps that same code on the spans it leaves, rather than the generic ",[819,1509,1510],{},"run_ended",". See ",[1221,1513,1223],{"href":372}," for the deadline, the grace window and the one ending it writes.",[807,1516,1517,964,1520,1523,1524,1527,1528,1531,1532,1535],{},[898,1518,1519],{},"The two span sweeps need two methods, because the second is not scoped to a run.",[819,1521,1522],{},"close_orphans(run_ids)"," closes what the reaper just failed, and it reads ",[819,1525,1526],{},"agent.spans (run_id) where status = 'running'",". It takes a ",[898,1529,1530],{},"list",", because one reaper pass fails many Runs and a list is one statement rather than one per Run. The second sweep below has no Run to be given: it looks for spans under runs that ended by some other path, and the reaper does not know which those are. It is its own query, and it takes a ",[819,1533,1534],{},"limit"," because a bad deploy can leave a lot of them.",[807,1537,1538,1539,1546,1547,1553],{},"⚠️ ",[898,1540,1541,1542,1545],{},"The second sweep is a read and then a write. It is never one ",[819,1543,1544],{},"UPDATE"," with an inner join."," PostgREST ",[898,1548,1549,1550,1552],{},"ignores an embedded resource filter on an ",[819,1551,1544],{},", and it still applies that filter to the returned representation."," The response then shows exactly the rows the author expected, and the statement has already closed every other running span in the schema, in every organization. Measured on the local stack, with one orphan under an ended Run and one healthy span under a live Run:",[811,1555,1558],{"className":1556,"code":1557,"language":816,"meta":817},[814],"PATCH \u002Fspans?select=span_id,runs!inner(ended_at)\n             &status=eq.running&runs.ended_at=not.is.null\n  response  orphan-under-ended                  the join was applied here\n  database  orphan-under-ended  -> error\n            healthy-under-live  -> error        the join was not applied here\n",[819,1559,1557],{"__ignoreMap":817},[807,1561,1562,1563,1565],{},"So the sweep reads the span IDs with the join and the ",[819,1564,1534],{},", then writes them by ID:",[811,1567,1570],{"className":1568,"code":1569,"language":816,"meta":817},[814],"GET   \u002Fspans?select=span_id,runs!inner(ended_at)\n             &status=eq.running&runs.ended_at=not.is.null&limit=:limit\nPATCH \u002Fspans?span_id=in.(...)&status=eq.running\n",[819,1571,1569],{"__ignoreMap":817},[807,1573,1574,1577,1578,1581,1582,1585],{},[819,1575,1576],{},"status = 'running'"," stays on the write, because a span may close honestly between the two calls and a write without it would overwrite ",[819,1579,1580],{},"ok"," with the sweep code. The ID list travels in the URL, so it is written in pages of 200, which is the same bound ",[819,1583,1584],{},"RunRepository"," already measured against Kong.",[807,1587,1588,1589,1595,1596,1598,1599,1319,1602,1604],{},"It sweeps a second set as well: ",[898,1590,1591,1592,1594],{},"any span still ",[819,1593,1448],{}," under a run that has ended",". Inngest cancels a run by stopping its function, so the worker never reaches a finalize, and the run leaves ",[819,1597,1448],{}," before its spans do. The heartbeat clause cannot find those, because the run is no longer ",[819,1600,1601],{},"queued",[819,1603,1448],{},". Without the second clause the orphan span alert below would fire for ever on every cancelled run.",[807,1606,1607,1617,1618,1621,1622,1624,1625,1627],{},[898,1608,1609,1610,1613,1614,1138],{},"The second sweep stamps ",[819,1611,1612],{},"error.code = run_ended",", and never ",[819,1615,1616],{},"worker_lost"," It reads every open span under a run that has ended, whatever ended it, so it cannot name a cause: a person cancels one run, ",[819,1619,1620],{},"run.failed"," ends a second, and the reaper pass above ends a third. ",[819,1623,1616],{}," here sends a person to look for a worker that no run lost. The scoped ",[819,1626,1506],{}," still carries the true code of each run the pass failed, so no cause is lost. A cancel adds none: it runs while the worker is still alive, and a span it stamped would take the outcome the worker is about to write.",[807,1629,1630],{},"A replayed call still writes a span on purpose. Leaving it out makes a resumed run look as if it never touched the CRM. Marking it keeps the tree complete and keeps cost attribution exact, because the replay costs nothing.",[868,1632,1634],{"id":1633},"payload-bounds-and-redaction","Payload bounds and redaction",[807,1636,1637],{},"A span stores tool input and output, because that is what an auditor and an engineer both need. It stores them under three limits.",[1639,1640,1641,1677,1687],"ol",{},[895,1642,1643,964,1646,1649,1650,1653,1654,1658,1659,1662,1665,1666,1669,1670,1673,1674,1138],{},[898,1644,1645],{},"Bounded.",[819,1647,1648],{},"bound(payload, 8 KB)"," each, and ",[819,1651,1652],{},"attributes.truncated"," comes from what it returns. It is the one payload boundary, defined in ",[1221,1655,1657],{"href":1656},"\u002Fengineering\u002Fsystem-design\u002Fagentic-platform\u002Fcontract#the-payload-boundary","the platform contract","; a span uses a smaller limit than a result, and the same algorithm. Storing the head of a serialized body is the thing it exists to prevent.",[1660,1661],"br",{},[898,1663,1664],{},"One flag covers two payloads written at two moments, so the close merges it."," The input is bounded at the open and the output at the close. A close that writes ",[819,1667,1668],{},"attributes"," whole sets ",[819,1671,1672],{},"truncated"," to what the output alone measured, and a trimmed input then reads as intact. The close reads the flag it is about to replace and keeps a ",[819,1675,1676],{},"true",[895,1678,1679,1682,1683,1686],{},[898,1680,1681],{},"Redacted."," Tool input and output pass the same boundary the ",[1221,1684,1685],{"href":190},"tool layer"," applies. Credentials never reach a span. A field the model must not see is absent from the schema, so it is absent here too.",[895,1688,1689,1692],{},[898,1690,1691],{},"No prompts by default."," A full prompt or a full context brief is not stored. Store a hash when a comparison needs one. An optional per-account debug capture must expire on its own, within days.",[807,1694,1695],{},"Business content such as an email body is legitimate span input. It is business data the product already holds.",[868,1697,1699],{"id":1698},"cost-and-usage","Cost and usage",[807,1701,1702,1705,1706,1709],{},[819,1703,1704],{},"ai_usage_log"," is the cost truth. ",[819,1707,1708],{},"ai_usage_daily"," is its rollup.",[892,1711,1712,1715,1721,1724,1730],{},[895,1713,1714],{},"One row per model call, and one row per metered vendor call.",[895,1716,1717,1718,1720],{},"A span points at a usage row with ",[819,1719,1437],{},". A span never carries a second price.",[895,1722,1723],{},"Run totals may be denormalized for the UI. The meter wins any disagreement.",[895,1725,1726,1727,1729],{},"A run tree total sums by ",[819,1728,910],{},". A parent therefore shows what its children spent.",[895,1731,1732],{},"Policy accrual reads this same meter. There is no second counter.",[807,1734,1735,1740,1741,1744,1745,1748,1749,1751,1752,1754],{},[898,1736,1737,1739],{},[819,1738,1704],{}," carries the root run, and the meter stamps it on every row."," The table's existing ",[819,1742,1743],{},"workflow_run_id"," is a foreign key to ",[819,1746,1747],{},"public.workflow_runs",", so it cannot hold an ",[819,1750,876],{}," id. Without a column of its own, a tree total has one path left: read every ",[819,1753,1437],{}," off the span tree, then filter the meter by that set. A discovery run of twenty thousand model turns sends twenty thousand identifiers back as a filter, which is a 740 KB PostgREST URL, and both readers of this number run on a hot path.",[811,1756,1759],{"className":1757,"code":1758,"language":816,"meta":817},[814],"ai_usage_log\n  agent_root_run_id  uuid   the root of the tree that spent this; null for a live-stack row\n  index (agent_root_run_id) where agent_root_run_id is not null\n",[819,1760,1758],{"__ignoreMap":817},[807,1762,1763,1764,1767,1768,1770],{},"The column is a soft reference with no foreign key, exactly as ",[819,1765,1766],{},"agent.spans.usage_id"," is, and for the same reason: a real key would tie the live table back into the isolated schema. ",[819,1769,1437],{}," on the span stays, because it answers the other question: which unit of work produced this row.",[913,1772,1774],{"id":1773},"the-sum-is-a-database-function-because-postgrest-refuses-an-aggregate","The sum is a database function, because PostgREST refuses an aggregate",[807,1776,1538,1777,1782,1783,1786,1787,1790],{},[898,1778,1779,1781],{},[819,1780,1704],{}," cannot be summed over PostgREST."," Aggregate functions are disabled on this instance, and ",[819,1784,1785],{},"authenticator"," carries no ",[819,1788,1789],{},"pgrst.db_aggregates_enabled"," setting to turn them on.",[811,1792,1795],{"className":1793,"code":1794,"language":816,"meta":817},[814],"GET \u002Fai_usage_log?select=cost.sum()\n  -> 400  PGRST123  Use of aggregate functions is not allowed\n",[819,1796,1794],{"__ignoreMap":817},[807,1798,1799,1800,1803],{},"Reading the rows and summing them in Python is not the fallback. ",[819,1801,1802],{},"max_rows"," is 1,000, and PostgREST truncates past it silently, so a discovery run of twenty thousand model turns would report the cost of the first thousand. Both readers of this number are on a hot path, and one of them is a ceiling.",[807,1805,1806,1807,1809],{},"So one read only SQL function answers it, in the ",[819,1808,935],{}," schema beside the tables that need it:",[811,1811,1815],{"className":1812,"code":1813,"language":1814,"meta":817,"style":817},"language-sql shiki shiki-themes github-dark","agent.run_usage(p_root_run_id uuid, p_organization_id uuid)\n  returns (input_tokens bigint, output_tokens bigint,\n           total_tokens bigint, cost_cents bigint)\n  -- STABLE. One indexed aggregate over ai_usage_log.agent_root_run_id.\n  -- REVOKE EXECUTE FROM PUBLIC, per the rule of the agent schema.\n\nagent.organization_day_cost(p_organization_id uuid)\n  returns (spent_cents bigint, cap_cents bigint)\n  -- STABLE. One indexed aggregate over (organization_id, created_at), plus\n  -- the agent.cost_ceilings row of kind daily_cost. cap_cents is null when\n  -- the organization set no ceiling.\n  -- REVOKE EXECUTE FROM PUBLIC, per the rule of the agent schema.\n","sql",[819,1816,1817,1822,1827,1832,1837,1842,1848,1853,1858,1863,1868,1873],{"__ignoreMap":817},[1083,1818,1819],{"class":1085,"line":22},[1083,1820,1821],{},"agent.run_usage(p_root_run_id uuid, p_organization_id uuid)\n",[1083,1823,1824],{"class":1085,"line":32},[1083,1825,1826],{},"  returns (input_tokens bigint, output_tokens bigint,\n",[1083,1828,1829],{"class":1085,"line":233},[1083,1830,1831],{},"           total_tokens bigint, cost_cents bigint)\n",[1083,1833,1834],{"class":1085,"line":244},[1083,1835,1836],{},"  -- STABLE. One indexed aggregate over ai_usage_log.agent_root_run_id.\n",[1083,1838,1839],{"class":1085,"line":264},[1083,1840,1841],{},"  -- REVOKE EXECUTE FROM PUBLIC, per the rule of the agent schema.\n",[1083,1843,1844],{"class":1085,"line":222},[1083,1845,1847],{"emptyLinePlaceholder":1846},true,"\n",[1083,1849,1850],{"class":1085,"line":360},[1083,1851,1852],{},"agent.organization_day_cost(p_organization_id uuid)\n",[1083,1854,1855],{"class":1085,"line":368},[1083,1856,1857],{},"  returns (spent_cents bigint, cap_cents bigint)\n",[1083,1859,1860],{"class":1085,"line":375},[1083,1861,1862],{},"  -- STABLE. One indexed aggregate over (organization_id, created_at), plus\n",[1083,1864,1865],{"class":1085,"line":157},[1083,1866,1867],{},"  -- the agent.cost_ceilings row of kind daily_cost. cap_cents is null when\n",[1083,1869,1870],{"class":1085,"line":182},[1083,1871,1872],{},"  -- the organization set no ceiling.\n",[1083,1874,1875],{"class":1085,"line":290},[1083,1876,1841],{},[807,1878,1879,1882,1885,1886,1889,1890,1893],{},[898,1880,1881],{},"An empty answer from either function is a fault, and never a zero.",[819,1883,1884],{},"run_usage"," aggregates with no ",[819,1887,1888],{},"GROUP BY"," and coalesces every column.\n",[819,1891,1892],{},"organization_day_cost"," selects two scalar subqueries and reads no table in its\nouter query. Each therefore answers exactly one row for every input. A tree\nthat spent nothing answers a row of zeroes, and so does an organization.",[807,1895,1896,1897,1900,1901,1904,1905,1309,1908,1911,1912,1915,1916,1138],{},"Read as zero, an empty answer reports no spend and no cap. Accrual reads that\nas headroom, and the ceiling stays unenforced for the whole fault window.\n",[819,1898,1899],{},"safe_execute_query"," makes the shape reachable: it turns a PostgREST 204 and an\nunparseable body into an empty result. The meter raises ",[819,1902,1903],{},"MissingRunUsage"," and\n",[819,1906,1907],{},"MissingDayCost",[819,1909,1910],{},"AccrualChecker"," turns either into ",[819,1913,1914],{},"deny"," under\n",[819,1917,1918],{},"metering_unavailable",[807,1920,1921,1924,1925,1928,1929,1063,1932,1935,1936,1939],{},[898,1922,1923],{},"A column the row does not carry is the same fault."," PostgREST keeps the key\nof a null column, so a key lookup catches a renamed column and misses a dropped\n",[819,1926,1927],{},"coalesce",". Every column is BIGINT, so a float or a string means the function\nlost a cast. ",[819,1930,1931],{},"UsageSummary",[819,1933,1934],{},"DayCost"," validate nothing, and accrual\ncompares their fields outside the block that catches a meter fault, so a null\nwould raise out of a checker that promises never to raise. Each read therefore\nrefuses an absent, a null and a non integer value. ",[819,1937,1938],{},"cap_cents"," is the one\nexception: null there means the organization set no ceiling.",[807,1941,1942,1945,1946,1949],{},[898,1943,1944],{},"A run detail answers a null usage, and not a 500."," The governance half must\nfail closed. The read half must not. Three endpoints build a detail, and the\nstart and the cancel build it after they committed their effect, so a 5xx there\ntells the caller that a live run did not start. A total the meter did not\nanswer is not a run that spent nothing. ",[819,1947,1948],{},"RunDetail.usage"," is therefore nullable\nand it never falls back to zeroes.",[807,1951,1952,1953,1956],{},"The day function reads the ceiling in the same statement, so the accrual check\ncosts one round trip and not two. ",[898,1954,1955],{},"The UTC day boundary is computed inside the\nfunction."," Two workers must agree which day a call belongs to, and a boundary a\nworker passes in puts every worker clock on the correctness path.",[807,1958,1959,1960,1138],{},"The day sum filters on the organization and the day, and on no source. See\n",[1221,1961,1962],{"href":287},"policy and governance",[807,1964,1965,1966,1968],{},"The function reads and returns. It decides nothing, so it does not break the rule that the database is storage. It carries its own ",[819,1967,902],{}," for the same reason every statement in this schema does: the caller holds the service role, so no policy filters it.",[807,1970,1971,964,1974,944,1977,1980,1981,1984],{},[898,1972,1973],{},"Cost is summed before it is converted.",[819,1975,1976],{},"ai_usage_log.cost",[819,1978,1979],{},"numeric(12,6)"," and it holds a currency amount, not cents. Thirty model calls of 0.004 each convert to 0 cents apiece and to 12 cents together. ",[819,1982,1983],{},"SUM"," first, multiply by 100, then round once.",[913,1986,1988],{"id":1987},"the-two-usage-types","The two usage types",[811,1990,1992],{"className":1077,"code":1991,"language":1079,"meta":817,"style":817},"@dataclass(frozen=True)\nclass UsageRecord:\n    \"\"\"One metered call, on its way to ai_usage_log.\"\"\"\n    organization_id: UUID\n    root_run_id: UUID          # stamped into agent_root_run_id\n    model_id: str\n    model_provider: str\n    cost: Decimal              # the currency amount the vendor charged\n    input_tokens: int = 0\n    output_tokens: int = 0\n    total_tokens: int = 0\n    cache_read_tokens: int = 0\n    cache_write_tokens: int = 0\n    reasoning_tokens: int = 0\n    duration_ms: int | None = None\n    user_id: UUID | None = None\n    source: str = 'agentic'\n    metadata: dict | None = None\n\n\n@dataclass(frozen=True)\nclass UsageSummary:\n    \"\"\"What one run tree spent. RunDetail.usage carries it.\"\"\"\n    input_tokens: int\n    output_tokens: int\n    total_tokens: int\n    cost_cents: int\n",[819,1993,1994,1999,2004,2009,2014,2019,2024,2029,2034,2039,2044,2049,2054,2059,2064,2070,2076,2082,2088,2093,2097,2101,2106,2112,2118,2124,2130],{"__ignoreMap":817},[1083,1995,1996],{"class":1085,"line":22},[1083,1997,1998],{},"@dataclass(frozen=True)\n",[1083,2000,2001],{"class":1085,"line":32},[1083,2002,2003],{},"class UsageRecord:\n",[1083,2005,2006],{"class":1085,"line":233},[1083,2007,2008],{},"    \"\"\"One metered call, on its way to ai_usage_log.\"\"\"\n",[1083,2010,2011],{"class":1085,"line":244},[1083,2012,2013],{},"    organization_id: UUID\n",[1083,2015,2016],{"class":1085,"line":264},[1083,2017,2018],{},"    root_run_id: UUID          # stamped into agent_root_run_id\n",[1083,2020,2021],{"class":1085,"line":222},[1083,2022,2023],{},"    model_id: str\n",[1083,2025,2026],{"class":1085,"line":360},[1083,2027,2028],{},"    model_provider: str\n",[1083,2030,2031],{"class":1085,"line":368},[1083,2032,2033],{},"    cost: Decimal              # the currency amount the vendor charged\n",[1083,2035,2036],{"class":1085,"line":375},[1083,2037,2038],{},"    input_tokens: int = 0\n",[1083,2040,2041],{"class":1085,"line":157},[1083,2042,2043],{},"    output_tokens: int = 0\n",[1083,2045,2046],{"class":1085,"line":182},[1083,2047,2048],{},"    total_tokens: int = 0\n",[1083,2050,2051],{"class":1085,"line":290},[1083,2052,2053],{},"    cache_read_tokens: int = 0\n",[1083,2055,2056],{"class":1085,"line":280},[1083,2057,2058],{},"    cache_write_tokens: int = 0\n",[1083,2060,2061],{"class":1085,"line":272},[1083,2062,2063],{},"    reasoning_tokens: int = 0\n",[1083,2065,2067],{"class":1085,"line":2066},15,[1083,2068,2069],{},"    duration_ms: int | None = None\n",[1083,2071,2073],{"class":1085,"line":2072},16,[1083,2074,2075],{},"    user_id: UUID | None = None\n",[1083,2077,2079],{"class":1085,"line":2078},17,[1083,2080,2081],{},"    source: str = 'agentic'\n",[1083,2083,2085],{"class":1085,"line":2084},18,[1083,2086,2087],{},"    metadata: dict | None = None\n",[1083,2089,2091],{"class":1085,"line":2090},19,[1083,2092,1847],{"emptyLinePlaceholder":1846},[1083,2094,2095],{"class":1085,"line":520},[1083,2096,1847],{"emptyLinePlaceholder":1846},[1083,2098,2099],{"class":1085,"line":318},[1083,2100,1998],{},[1083,2102,2103],{"class":1085,"line":327},[1083,2104,2105],{},"class UsageSummary:\n",[1083,2107,2109],{"class":1085,"line":2108},23,[1083,2110,2111],{},"    \"\"\"What one run tree spent. RunDetail.usage carries it.\"\"\"\n",[1083,2113,2115],{"class":1085,"line":2114},24,[1083,2116,2117],{},"    input_tokens: int\n",[1083,2119,2121],{"class":1085,"line":2120},25,[1083,2122,2123],{},"    output_tokens: int\n",[1083,2125,2127],{"class":1085,"line":2126},26,[1083,2128,2129],{},"    total_tokens: int\n",[1083,2131,2133],{"class":1085,"line":2132},27,[1083,2134,2135],{},"    cost_cents: int\n",[807,2137,2138,2141,2142,2145,2146,2148,2149,2151],{},[819,2139,2140],{},"UsageRecord"," writes no ",[819,2143,2144],{},"agent_run_id"," and no ",[819,2147,1743],{},". Both belong to the live stack, and ",[819,2150,2144],{}," is unique per row.",[807,2153,2154,2156,2157,2160],{},[819,2155,1931],{}," is the whole read. Accrual takes ",[819,2158,2159],{},"cost_cents"," off it rather than calling a second method, because a second method is a second query shape over the same function.",[913,2162,2164],{"id":2163},"who-prices-the-call","Who prices the call",[807,2166,2167,2168,2171,2172,2175,2176,2179,2180,2182],{},"The vendor reports tokens. It reports no price, and neither does the framework\n",[819,2169,2170],{},"RunOutput",". So the caller of ",[819,2173,2174],{},"UsageMeter.record"," prices the call, and the\nplatform has one pricer: ",[819,2177,2178],{},"src\u002Fshared\u002Fmetering\u002Fmodel_pricing.calculate_run_cost",".\nIt is not a new rate table. ",[819,2181,1704],{}," is one table for the whole product,\nso a second rate card would put two prices in one column.",[811,2184,2187],{"className":2185,"code":2186,"language":816,"meta":817},[814],"cost = Decimal(str(calculate_run_cost(model_id, input_tokens, output_tokens,\n                                      cache_read_tokens, cache_write_tokens)))\n",[819,2188,2186],{"__ignoreMap":817},[807,2190,2191,2192,1063,2195,2198,2199,2202,2203,2206,2207,2210,2211,2213],{},"The function answers a ",[819,2193,2194],{},"float",[819,2196,2197],{},"UsageRecord.cost"," is a ",[819,2200,2201],{},"Decimal",", so the\nconversion goes through ",[819,2204,2205],{},"str",". ",[819,2208,2209],{},"Decimal(0.0009)"," carries seventeen digits of\nbinary noise into a ",[819,2212,1979],{}," column.",[807,2215,1538,2216,964,2219,2222,2223,2226,2227,2230],{},[898,2217,2218],{},"The input token count means two different things.",[819,2220,2221],{},"calculate_run_cost","\nwants the count that excludes the cache reads, because it prices a cache read at\nits own discounted rate. Agno reports Anthropic's ",[819,2224,2225],{},"input_tokens",", which already\nexcludes them, and OpenAI's ",[819,2228,2229],{},"prompt_tokens",", which includes them. Pass the raw\nnumber for OpenAI and the cached tokens are priced two times.",[811,2232,2235],{"className":2233,"code":2234,"language":816,"meta":817},[814],"anthropic   input_tokens                              already the non cached count\nopenai      max(0, input_tokens - cache_read_tokens)\n",[819,2236,2234],{"__ignoreMap":817},[807,2238,2239,2240,2243],{},"A model the rate table does not hold prices at zero and logs at ERROR. Zero is a\nwrong number and not an error, and ",[819,2241,2242],{},"max_cost_cents"," reads this column. See\ncross-document invariant 40.",[913,2245,2247],{"id":2246},"metering-is-not-best-effort","Metering is not best effort",[807,2249,2250],{},"Policy accrual depends on the meter, so a run whose spend is not counted cannot be governed.",[811,2252,2255],{"className":2253,"code":2254,"language":816,"meta":817},[814],"the vendor or model call returns\n  -> write the usage row at once, in its own transaction\n  -> a transient failure retries inline, three times\n  -> still failing -> fail the run with metering_unavailable, and do not retry\n",[819,2256,2254],{"__ignoreMap":817},[807,2258,2259],{},"The failure is deliberately terminal. A segment retry would spend more money that also cannot be counted, so retrying makes the hole larger. An uncounted run is worse than a failed run.",[807,2261,2262],{},"The usage row is written from the vendor response, so the amount is already known when the write starts. The only failure mode is the write itself.",[807,2264,2265,2268,2269,2272,2273,2276,2277,1138],{},[898,2266,2267],{},"An asynchronous provider job moves the frame, not the rule."," The amount is\nstill known when the write starts, because the provider reports it in the\ncallback. Only the writer changes. The Run that submitted is often terminal by then. So\n",[819,2270,2271],{},"provider_job.settle"," writes the row on an Inngest frame, and Inngest retries\nit. An exhausted retry reports to Sentry and leaves ",[819,2274,2275],{},"agent.provider_jobs.usage_id","\nnull, which is the query that finds an unsettled charge. It ends no Run, because\nthere is no Run left to end. See\n",[1221,2278,2279],{"href":190},"asynchronous provider jobs",[807,2281,2282,964,2285,2287,2288,2291,2292,2294,2295,2297],{},[898,2283,2284],{},"The meter raises, and the caller ends the run.",[819,2286,2174],{}," retries three times and then raises ",[819,2289,2290],{},"MeteringUnavailable",". It does not call ",[819,2293,971],{},", because a component that both writes the meter and ends runs is two owners of the lifecycle. The caller maps that exception to a terminal failure with the code ",[819,2296,1918],{},", and to a non retryable error for Inngest.",[807,2299,2300,2303,2304,2307],{},[819,2301,2302],{},"record()"," never swallows. The live stack's ",[819,2305,2306],{},"log_usage"," is fire and forget and it catches every exception; this meter is the opposite of it, and it does not call it.",[868,2309,2311],{"id":2310},"retention","Retention",[825,2313,2314,2327],{},[828,2315,2316],{},[831,2317,2318,2321,2324],{},[834,2319,2320],{},"Data",[834,2322,2323],{},"Target",[834,2325,2326],{},"Why",[841,2328,2329,2341,2353,2364,2376,2388],{},[831,2330,2331,2335,2338],{},[846,2332,2333],{},[819,2334,876],{},[846,2336,2337],{},"13 months",[846,2339,2340],{},"The product record of what was done",[831,2342,2343,2347,2350],{},[846,2344,2345],{},[819,2346,880],{},[846,2348,2349],{},"90 days",[846,2351,2352],{},"The debugging record; it is the largest table",[831,2354,2355,2359,2361],{},[846,2356,2357],{},[819,2358,1704],{},[846,2360,2337],{},[846,2362,2363],{},"Billing and year on year comparison",[831,2365,2366,2370,2373],{},[846,2367,2368],{},[819,2369,1708],{},[846,2371,2372],{},"kept",[846,2374,2375],{},"Small, and it survives the log it summarizes",[831,2377,2378,2382,2385],{},[846,2379,2380],{},[819,2381,1049],{},[846,2383,2384],{},"12 months",[846,2386,2387],{},"The audit record of every admission and every refusal",[831,2389,2390,2395,2398],{},[846,2391,2392],{},[819,2393,2394],{},"agent.sessions",[846,2396,2397],{},"with the run",[846,2399,2400],{},"Execution state, and it holds message history",[807,2402,2403,2404,2407,2408,1230,2410,1063,2412,2414,2415,2417,2418,1230,2421,1063,2424,2427],{},"A scheduled job deletes past the window. ",[819,2405,2406],{},"retention.sweeper"," is that job for ",[819,2409,1049],{},[819,2411,880],{},[819,2413,876],{},": an Inngest cron beside ",[819,2416,1470],{},", on the half hour, which reads the oldest rows past each window and deletes them by id. Each table carries its own bound, ",[819,2419,2420],{},"DECISION_SWEEP_LIMIT",[819,2422,2423],{},"SPAN_SWEEP_LIMIT",[819,2425,2426],{},"RUN_SWEEP_LIMIT",". A pass whose read fills its bound logs a warning. A run older than its span window keeps its result and its cost, and loses its step detail.",[807,2429,2430,964,2433,944,2436,2438],{},[898,2431,2432],{},"The order is decisions, then spans, then runs.",[819,2434,2435],{},"agent.policy_decisions.run_id",[819,2437,996],{},", so deleting a run updates every decision that names it. A decision under a 13 month run is already past its own 12 month window, so the log goes first and that update then touches nothing. The spans go before the runs while a backlog drains: both sweeps read the oldest trees first, so during a drain they read the same trees and the span sweep leaves the run cascade less to carry. At steady state the two cohorts sit ten months apart and never meet, and what helps the run sweep then is that the span sweep runs at all — a tree reaches 13 months with its spans already gone.",[807,2440,2441,2444,2445,2447],{},[898,2442,2443],{},"A sweep that cannot reach its table does not stop the sweeps after it."," Each sweep records its own failure, the pass finishes every sweep, and the error is raised at the end. A span tree that refuses every hour would otherwise stop ",[819,2446,876],{}," from ever being swept again.",[807,2449,2450,964,2453,1063,2455,2458,2459,2461],{},[898,2451,2452],{},"A run is deleted by its root, one root to one statement.",[819,2454,910],{},[819,2456,2457],{},"parent_run_id"," both carry ",[819,2460,992],{},", so one root row takes its whole tree, and the tree takes its spans, its sessions, its approvals and its cancellation row. A statement naming a page of roots would cascade into millions of rows and meet the statement timeout, which rolls back every root it named. A child run is created after its root, so a child past the window always has a root past the same window and no tree is left half deleted. A tree the database refuses is counted and skipped, because the read is ascending and a raise there would stall the sweep on its oldest rows for ever.",[807,2463,2464,2467,2468,2471],{},[898,2465,2466],{},"A span tree is chosen by its ROOT span, and the run stays."," The read tests the span that carries no parent. A child span opens after its parent, so a tree whose root span is past the window holds no span inside it, except the live tree case below. ",[819,2469,2470],{},"agent.runs.parent_span_id"," carries no foreign key, and this window is the reason it does not: the run outlives the node span it hung from.",[807,2473,2474,2477,2478,2480,2481,2483],{},[898,2475,2476],{},"The delete is newest first, one bounded page to one statement."," It does not name the root span. ",[819,2479,921],{}," carries ",[819,2482,992],{},", so a statement naming the root cascades the whole subtree inside itself. A tree of 150,000 spans then meets the 8 second statement timeout on every pass. The widest trees hold the most content, so those are the trees that would never delete.",[807,2485,2486,2489,2490,2492],{},[819,2487,2488],{},"started_at"," comes from the column default at the insert, and ",[819,2491,1116],{}," awaits a parent open before it opens a child. So the parent carries the earlier value. A page is a prefix of the descending order, so every descendant of a row in the page is in the page as well. The order rests on that await and not on the foreign key, which needs the parent committed first and not started first.",[807,2494,2495],{},"Three consequences are worth stating.",[807,2497,2498,2500,2501,2504,2505,2507],{},[819,2499,2423],{}," counts ",[898,2502,2503],{},"trees, not spans",". A span arrives far more often than a run, and this bound reads none of that rate: a tree is one read and then one paged delete, and the page is what bounds the rows. So it is the same arrival unit as ",[819,2506,2426],{},". The run bound's headroom argument does not transfer, though — that one is set against a table 13 months behind the writes it sweeps, and this sweep lags 90 days.",[807,2509,2510,2511,2514,2515,2518,2519,2522],{},"A tree goes whole, so a span ",[898,2512,2513],{},"inside"," the window goes with a tree whose root span is outside it. The wait sweep bounds a parked run at its own deadline, and ",[819,2516,2517],{},"approval_ttl_seconds"," carries no upper bound, so that deadline can sit past 90 days and a run parked on an approval reaches it. ",[819,2520,2521],{},"MAX_RUN_DURATION_S"," is 30 days and it is measured at a segment boundary, so it ends such a run only after the approval answers.",[807,2524,2525,2528,2529,2531,2532,2534,2535,2538,2539,2541,2542,2545],{},[898,2526,2527],{},"A live tree that loses its spans stops recording, and it can also stop running."," A later span insert names a parent row that has gone, so ",[819,2530,921],{}," answers 23503; ",[819,2533,1116],{}," reports that to Sentry and the work continues. ",[819,2536,2537],{},"RunExecutor"," raises when the run span does not open, and Inngest memoizes that step, so a resumed run of that tree fails. And a ",[819,2540,1006],{}," child run INSERT meets ",[819,2543,2544],{},"agent.assert_parent_span_shape",", which answers 23503, so the child never starts. The window outranks a run, which is what ENG-2166 states, but the cost is a failed run and not only a thinner trace.",[868,2547,2549],{"id":2548},"deletion-on-request","Deletion on request",[807,2551,2552],{},"The windows above answer \"when does this age out\". They do not answer \"delete this person now\", and that request arrives from a customer, not from a schedule. We hold EU personal data, so the answer must exist before V1 ships.",[807,2554,2555,2558],{},[898,2556,2557],{},"A span is the hard case."," It stores tool input and output, so a span already holds the email address it sent to and the message body it sent. Every other table holds a reference; a span holds the content.",[807,2560,2561],{},"One job answers both shapes of request.",[825,2563,2564,2577],{},[828,2565,2566],{},[831,2567,2568,2571,2574],{},[834,2569,2570],{},"Request",[834,2572,2573],{},"Scope",[834,2575,2576],{},"Action",[841,2578,2579,2592],{},[831,2580,2581,2584,2589],{},[846,2582,2583],{},"Delete an organization",[846,2585,2586,2587],{},"every table carrying ",[819,2588,902],{},[846,2590,2591],{},"delete the rows, keep nothing",[831,2593,2594,2597,2600],{},[846,2595,2596],{},"Delete a person",[846,2598,2599],{},"that person's rows, and any payload naming them",[846,2601,2602],{},"redact, and keep the shape",[807,2604,2605],{},"Redaction rather than deletion, for the person case:",[811,2607,2610],{"className":2608,"code":2609,"language":816,"meta":817},[814],"agent.spans            input and output -> {\"redacted\": \"subject_erasure\"}; kind, timing, status and usage_id stay\nagent.runs             input and result summary redacted; status, timing and refs stay\nagent.approvals        proposed arguments and proposed_content redacted; decision, actor and timing stay\nagent.sessions         deleted; it is execution state and it is reconstructable from nothing\nagent.memories         deleted, including superseded rows\nagent.conversations    messages deleted\nagent.knowledge_chunks deleted with their source\nagent.idempotency_keys response redacted; the claim, the status and the hashes stay\nai_usage_log           kept; it carries a run reference and a price, and no personal data\nagent.policy_decisions arguments_hash stays, principal user_id redacted\n",[819,2611,2609],{"__ignoreMap":817},[807,2613,2614,2617,2618,2621,2622,1138],{},[898,2615,2616],{},"A pending approval is cancelled, not redacted."," This is the one row where redaction would be worse than doing nothing. The resume path executes the approved call ",[898,2619,2620],{},"from the approval row",", so redacting a pending proposal and then letting a person approve it sends a message whose recipient and body are the string ",[819,2623,2624],{},"redacted",[811,2626,2629],{"className":2627,"code":2628,"language":816,"meta":817},[814],"approval pending   -> cancel it, then redact.  The Run ends; nobody approves a hollow proposal\napproval resolved  -> redact.  The decision, the actor and the timing stay\n",[819,2630,2628],{"__ignoreMap":817},[807,2632,2633],{},"Erasure and an unanswered approval are the same request from two directions, and cancelling is the only answer that respects both.",[807,2635,2636],{},"Two of the redacted rows are easy to miss, and both hold content rather than a reference.",[892,2638,2639,2651],{},[895,2640,2641,2646,2647,2650],{},[898,2642,2643],{},[819,2644,2645],{},"agent.approvals"," stores the exact proposed arguments so a person can decide without opening the run. For ",[819,2648,2649],{},"email.send"," that is the recipient and the body.",[895,2652,2653,2658,2659,2662],{},[898,2654,2655],{},[819,2656,2657],{},"agent.idempotency_keys"," stores the tool response, because the claim is the replay journal. A ",[819,2660,2661],{},"crm.search"," response is a list of people.",[807,2664,2665],{},"Redacting the response does not break the journal for any live run: a run still inside its retention window is younger than the erasure request that reaches it, and a replay of a redacted claim returns a redacted result to a run whose subject asked to be erased. That is the correct answer.",[807,2667,2668],{},"Two reasons the timing survives. A deleted span would break the tree that explains what an agent did, and the cost record must still add up for the invoice already issued.",[807,2670,2671],{},"The job runs against one organization or one subject reference, reports what it would touch, and writes only under an explicit apply. It is a scheduled query like the reaper, not a new subsystem.",[807,2673,2674,2677],{},[898,2675,2676],{},"Every new table that stores a payload joins this list in the same change that creates it."," A table that is easy to forget here is the one that leaks.",[807,2679,2680,2681,2684,2685,2688,2689,2691,2692,2695],{},"The test for whether a table belongs here is one question: does it store ",[898,2682,2683],{},"content",", or a ",[898,2686,2687],{},"reference","? ",[819,2690,876],{}," stores an input summary, so it is here. ",[819,2693,2694],{},"agent.run_control"," stores a reason and three IDs, so it is not.",[868,2697,2699],{"id":2698},"capability-attribution","Capability attribution",[807,2701,2702],{},"ENG-2298 owns Run attribution, frozen child execution and bounded operational metrics.\nENG-2276 owns the stable start endpoint. ENG-2283 owns stable Front Door routing.\nENG-2297 owns routing evaluation fixtures and expected outcomes.",[913,2704,2706],{"id":2705},"identity-and-frozen-execution","Identity and frozen execution",[807,2708,2709,2710,1063,2713,2716],{},"Each new Run copies ",[819,2711,2712],{},"capability_id",[819,2714,2715],{},"contract_version"," from its resolved published binding.\nBoth fields are null for an unbound definition. Old Runs keep null fields; no migration infers identity from current definitions.\nThe two fields are immutable and appear in narrow Run reads, API responses and CLI output.\nA definition UUID start uses the same attribution path as a capability start. Inputs cannot override identity.\nInactive bindings still identify retained executors. Registry activation controls new product selection, not identity.",[807,2718,2719],{},"A new root snapshot freezes all reachable workflow and agent definitions, including their rendered skills, contracts and scope ceilings.\nEach child entry has its executor UUID, kind, capability binding, execution snapshot and published scope sets.\nThe snapshot format has an explicit version. Missing children in that format fail closed.\nA legacy snapshot keeps its existing child-start behavior and does not claim frozen capability metadata.",[807,2721,2722],{},"Child starts read the parent's stored snapshot in the same organization. They never use caller-supplied snapshots.\nEach child copies its frozen subtree. Later binding switches, configuration changes or disable operations cannot retarget it.\nCurrent actor rights, parent cancellation, policy and root budgets still apply. Frozen scope sets do not grant new rights.\nCycle, missing reference, unsupported format, excessive depth and an oversized tree refuse the start before insertion.\nInput and result limits remain 32 KiB. A complete execution snapshot has a separate 256 KiB limit.\nThe Signals Search fixture already exceeds 32 KiB when its four agent snapshots are included.\nBefore rollout, a read-only preflight must freeze every active published tenant root and report failures and sizes.\nAn oversized graph requires a smaller published graph before rollout. Do not trim it or silently use live children.\nA descendant publish can grow an indirect ancestor. Start always checks the complete size, even when publish checked only direct referrers.\nPublish measures the same complete snapshot. Repeated child references do not trigger repeated definition reads during one freeze.",[807,2724,2725,2726,2729],{},"From ",[819,2727,2728],{},"ac-python-api",", run this command once for each tenant against the target environment:",[811,2731,2735],{"className":2732,"code":2733,"language":2734,"meta":817,"style":817},"language-bash shiki shiki-themes github-dark","python -m scripts.audit_capability_snapshots --organization-id \u003Corganization-uuid>\n","bash",[819,2736,2737],{"__ignoreMap":817},[1083,2738,2739,2742,2746,2750,2753,2757,2760,2764],{"class":1085,"line":22},[1083,2740,1079],{"class":2741},"svObZ",[1083,2743,2745],{"class":2744},"sDLfK"," -m",[1083,2747,2749],{"class":2748},"sU2Wk"," scripts.audit_capability_snapshots",[1083,2751,2752],{"class":2744}," --organization-id",[1083,2754,2756],{"class":2755},"snl16"," \u003C",[1083,2758,2759],{"class":2748},"organization-uui",[1083,2761,2763],{"class":2762},"s95oV","d",[1083,2765,2766],{"class":2755},">\n",[807,2768,2769],{},"The command reads definitions only. It reports each snapshot size and exits with a failure if any tree cannot freeze.",[913,2771,2773],{"id":2772},"metric-contract","Metric contract",[807,2775,2776],{},"Run rows, results and the canonical usage meter remain the durable facts. Metrics are operational samples, not billing records.\nUse the existing Sentry metrics transport. Add no metrics database, scheduler, dashboard or public metrics endpoint.\nEmit a start only after a new row is inserted. A duplicate start emits no second start.\nEmit completion only for a successful terminal state write. A retry, wait, resume or stale terminal write emits no completion.\nMetrics are best effort. A crash between a database commit and export can lose a sample; exporter failure cannot fail a Run.\nDo not claim exact delivery. Durable Run records provide the audit source when samples are incomplete.",[825,2778,2779,2789],{},[828,2780,2781],{},[831,2782,2783,2786],{},[834,2784,2785],{},"Measurement",[834,2787,2788],{},"Rule",[841,2790,2791,2799,2819,2827,2835,2843,2869],{},[831,2792,2793,2796],{},[846,2794,2795],{},"Starts",[846,2797,2798],{},"One admitted Run row, including a row that admission later denies",[831,2800,2801,2804],{},[846,2802,2803],{},"Completion",[846,2805,2806,2807,1230,2810,1230,2813,1230,2816],{},"One of ",[819,2808,2809],{},"success",[819,2811,2812],{},"partial",[819,2814,2815],{},"failure",[819,2817,2818],{},"cancelled",[831,2820,2821,2824],{},[846,2822,2823],{},"Zero results",[846,2825,2826],{},"Successful complete or empty output with a known zero item count; unknown output is not zero",[831,2828,2829,2832],{},[846,2830,2831],{},"Latency",[846,2833,2834],{},"Nonnegative time from creation to terminal state, including queue and wait time",[831,2836,2837,2840],{},[846,2838,2839],{},"Cost",[846,2841,2842],{},"Decimal USD deltas after successful canonical usage writes; never export tree totals",[831,2844,2845,2848],{},[846,2846,2847],{},"Enrichment hit",[846,2849,2850,2851,1230,2854,1319,2857,2860,2861,1230,2864,1063,2867],{},"One requested field with ",[819,2852,2853],{},"current",[819,2855,2856],{},"retained",[819,2858,2859],{},"refreshed"," state; denominator also includes ",[819,2862,2863],{},"not_found",[819,2865,2866],{},"failed",[819,2868,2818],{},[831,2870,2871,2874],{},[846,2872,2873],{},"Routing",[846,2875,2876],{},"Bounded live decision outcomes; accuracy uses only labelled evaluation expectations",[807,2878,2879,2880,2883,2884,2886,2887,2889,2890,2892],{},"Run failure or cancellation takes precedence over output. Runtime partial reasons and product ",[819,2881,2882],{},"outcome: partial"," both produce ",[819,2885,2812],{},".\nAn empty successful search is ",[819,2888,2809],{}," with a zero-result sample. Total provider failure is ",[819,2891,2815],{},".\nSignals keeps its existing output schema. Missing result counts, field states or usage remain unknown, not zero.\nCost export matches the ambient Run organization and root against the usage record. It occurs after the write retry loop.\nKeep fractional amounts; rounding each call to integer cents loses small charges.\nWrites without a matching capability scope have unknown attribution and emit no capability cost sample.\nThis includes asynchronous provider settlement without a Run scope. The canonical meter still records those charges.\nInternal Runs have no capability metrics. Child capabilities receive their own identity, not the parent's identity.",[807,2894,2895],{},"Metric attributes use only the five capability IDs, bounded source kinds, root\u002Fchild level and fixed outcome enums.\nContract versions remain exact on Runs, logs and spans. They are not metric attributes because upgrades can create unbounded versions.\nDo not include UUIDs, tenant IDs, entity IDs, executor names, error text, schema values or prompts in metric attributes.\nStructured diagnostic logs may carry Run correlation IDs and the exact version, but never copy input or output payloads.\nRun scope carries identity into both span APIs and resets it after concurrent or nested execution.",[807,2897,2898,2899,2902,2903,1230,2906,2909,2910,1230,2913,2916,2917,2920,2921,2909,2924,1063,2927,2929],{},"Metric names use the ",[819,2900,2901],{},"agentic.capability."," prefix: ",[819,2904,2905],{},"starts",[819,2907,2908],{},"completions",",\n",[819,2911,2912],{},"zero_results",[819,2914,2915],{},"latency"," (seconds), ",[819,2918,2919],{},"cost"," (USD), ",[819,2922,2923],{},"enrichment_fields",[819,2925,2926],{},"enrichment_hits",[819,2928,247],{},". The final Sentry export filter removes inherited\nrequest and user attributes from these metrics.",[807,2931,2932,2933,1230,2936,1063,2939,2942,2943,2945],{},"Routing accuracy has ",[819,2934,2935],{},"correct",[819,2937,2938],{},"incorrect",[819,2940,2941],{},"unscored"," outcomes. Missing expected routes are ",[819,2944,2941],{},".\nUnknown product IDs and custom definition UUIDs use fixed buckets. Never export the unknown text as a label.\nA correct capability selection does not imply a successful Run. Routing and execution metrics stay separate.",[868,2947,2948],{"id":62},"Alerts",[807,2950,2951],{},"An alert is a scheduled query. V1 ships six, and adds none until scale proves the need.",[825,2953,2954,2967],{},[828,2955,2956],{},[831,2957,2958,2961,2964],{},[834,2959,2960],{},"Alert",[834,2962,2963],{},"Query",[834,2965,2966],{},"Goes to",[841,2968,2969,2986,2998,3012,3030,3040],{},[831,2970,2971,2974,2983],{},[846,2972,2973],{},"Stuck runs",[846,2975,2976,1319,2978,1230,2980,2982],{},[819,2977,1601],{},[819,2979,1448],{},[819,2981,1171],{}," older than 120 seconds, and no live Inngest run",[846,2984,2985],{},"Sentry, after the reaper acts",[831,2987,2988,2991,2996],{},[846,2989,2990],{},"Orphan spans",[846,2992,2993,2995],{},[819,2994,1448],{}," under a run that already ended, after the reaper swept",[846,2997,848],{},[831,2999,3000,3003,3010],{},[846,3001,3002],{},"Abandoned waits",[846,3004,3005,3006,3009],{},"runs the wait sweep ended under ",[819,3007,3008],{},"wait_abandoned"," in the last hour",[846,3011,848],{},[831,3013,3014,3017,3027],{},[846,3015,3016],{},"Expiring approvals",[846,3018,3019,3022,3023,3026],{},[819,3020,3021],{},"pending"," with ",[819,3024,3025],{},"expires_at"," inside one hour",[846,3028,3029],{},"Slack",[831,3031,3032,3035,3038],{},[846,3033,3034],{},"Failure rate",[846,3036,3037],{},"failed runs per organization per hour, above a threshold",[846,3039,3029],{},[831,3041,3042,3044,3047],{},[846,3043,2839],{},[846,3045,3046],{},"organization spend today above 80 percent of its cap",[846,3048,3029],{},[807,3050,3051,3054,3055,1138],{},[898,3052,3053],{},"The stuck run alert carries the same conditions as the reaper, deliberately."," It reports what the reaper acted on, so a narrower query would alert on runs the reaper spared on purpose — a run merely queued behind its concurrency lane writes no heartbeat and is not stuck. See ",[1221,3056,1223],{"href":372},[807,3058,3059],{},"There is no metrics database and no second scheduler. Inngest cron runs the queries, exactly as it runs the reaper.",[868,3061,3063],{"id":3062},"run-explorer","Run explorer",[825,3065,3066,3075],{},[828,3067,3068],{},[831,3069,3070,3073],{},[834,3071,3072],{},"Section",[834,3074,2320],{},[841,3076,3077,3085,3093,3101],{},[831,3078,3079,3082],{},[846,3080,3081],{},"Runs",[846,3083,3084],{},"Current and recent status, and what a waiting Run waits on",[831,3086,3087,3090],{},[846,3088,3089],{},"Run detail",[846,3091,3092],{},"The span tree, child runs, and the result or error",[831,3094,3095,3098],{},[846,3096,3097],{},"Usage",[846,3099,3100],{},"Tokens and cost from the canonical meter",[831,3102,3103,3106],{},[846,3104,3105],{},"Health",[846,3107,3108],{},"Links to Sentry and Inngest, not copies of their dashboards",[811,3110,3113],{"className":3111,"code":3112,"language":816,"meta":817},[814],"Prospecting workflow\n  |- ok       Find companies      12s\n  |- ok       Find founders       35s\n  |- error    Find emails         18s   upstream timeout\n  \\- waiting  Update CRM                approval\n",[819,3114,3112],{"__ignoreMap":817},[807,3116,3117,3118,3120],{},"The tree comes from one query on ",[819,3119,910],{},". The explorer shows the newest attempt by default, and can show earlier attempts and replayed calls on request.",[807,3122,3123],{},"Mission Control stays a separate internal fleet product. The customer facing run explorer stays in the product frontend.",[868,3125,3127],{"id":3126},"core-code","Core code",[811,3129,3132],{"className":3130,"code":3131,"language":816,"meta":817},[814],"governance\u002Fobservability\u002F\n  models.py        RunSpan, SpanKind, SpanStatus, UsageRecord, UsageSummary\n  tracing.py       run_scope(), current_run_id(), current_root_run_id(),\n                   current_span_id()\n  recorder.py      SpanRecorder, the scoped context manager.\n                   LiveEventPublisher, the live seam it publishes through\n  repository.py    RunSpanRepository\n  usage.py         UsageMeter\n  alerts.py        the five scheduled queries (not built yet)\n",[819,3133,3131],{"__ignoreMap":817},[811,3135,3137],{"className":1077,"code":3136,"language":1079,"meta":817,"style":817},"class RunSpanRepository(Protocol):\n    async def open(self, span: RunSpan) -> None: ...\n    async def close(\n        self,\n        span_id: UUID,\n        organization_id: UUID,\n        *,\n        status: Literal['ok', 'error'],\n        output: dict | None = None,\n        usage_id: UUID | None = None,\n        error: dict | None = None,\n        attributes: dict | None = None,\n    ) -> None: ...\n    async def tree(self, root_run_id: UUID, organization_id: UUID) -> list[RunSpan]:\n        \"\"\"Every span of one tree, in (started_at, span_id) order.\n\n        It pages by keyset. max_rows is 1,000 and PostgREST truncates past it\n        without saying so, and a discovery tree holds tens of thousands.\n        \"\"\"\n    async def close_orphans(\n        self, run_ids: Sequence[UUID], organization_id: UUID, error: dict\n    ) -> int:\n        \"\"\"The open spans of the Runs the reaper just failed. One statement.\"\"\"\n    async def close_orphans_of_ended_runs(self, error: dict, limit: int) -> int:\n        \"\"\"Every span still running under a Run that has ended. Not scoped to one Run.\n\n        A read with the join, then a write by span id. Never one UPDATE with an\n        embedded filter; see the sweep section above for what that closes.\n        \"\"\"\n    async def count_of_kind(\n        self, run_id: UUID, organization_id: UUID, kind: SpanKind,\n        *, exclude_replayed: bool = False,\n    ) -> int:\n        \"\"\"The reader of (run_id, kind). It keys on the Run and never the tree.\n\n        An exact count travels in Content-Range, so no row travels at all.\n\n        exclude_replayed drops a span whose attributes carry replayed = true.\n        The tool call ceiling needs it and the turn ceiling does not: a replayed\n        call made no call, and a replayed model turn really spent the tokens.\n        The key is absent on most spans, so the filter is a two member `or`\n        rather than a `not.eq`, which would drop every span with no key.\n        \"\"\"\n\nclass UsageMeter(Protocol):\n    async def record(self, usage: UsageRecord) -> UUID:\n        \"\"\"Writes one ai_usage_log row and returns its id. Not best effort.\"\"\"\n    async def run_usage(\n        self, root_run_id: UUID, organization_id: UUID\n    ) -> UsageSummary:\n        \"\"\"One call to agent.run_usage(). It reads no span.\n\n        Raises MissingRunUsage on an empty answer or an odd column.\n        \"\"\"\n    async def organization_day_cents(\n        self, organization_id: UUID\n    ) -> DayCost:\n        \"\"\"One call to agent.organization_day_cost(). Accrual reads it.\n\n        Raises MissingDayCost on the same two shapes.\n        \"\"\"\n\nclass SpanRecorder:\n    def span(\n        self,\n        *,\n        kind: SpanKind,\n        name: str,\n        input: Payload | None = None,\n        attributes: dict | None = None,\n    ): ...\n",[819,3138,3139,3144,3149,3154,3159,3164,3169,3174,3179,3184,3189,3194,3199,3204,3209,3214,3218,3223,3228,3233,3238,3243,3248,3253,3258,3263,3267,3272,3278,3283,3288,3294,3300,3305,3311,3316,3322,3327,3333,3339,3344,3350,3356,3361,3366,3372,3378,3384,3390,3396,3401,3407,3412,3418,3423,3429,3435,3441,3447,3452,3457,3462,3467,3473,3479,3484,3489,3495,3501,3507,3512],{"__ignoreMap":817},[1083,3140,3141],{"class":1085,"line":22},[1083,3142,3143],{},"class RunSpanRepository(Protocol):\n",[1083,3145,3146],{"class":1085,"line":32},[1083,3147,3148],{},"    async def open(self, span: RunSpan) -> None: ...\n",[1083,3150,3151],{"class":1085,"line":233},[1083,3152,3153],{},"    async def close(\n",[1083,3155,3156],{"class":1085,"line":244},[1083,3157,3158],{},"        self,\n",[1083,3160,3161],{"class":1085,"line":264},[1083,3162,3163],{},"        span_id: UUID,\n",[1083,3165,3166],{"class":1085,"line":222},[1083,3167,3168],{},"        organization_id: UUID,\n",[1083,3170,3171],{"class":1085,"line":360},[1083,3172,3173],{},"        *,\n",[1083,3175,3176],{"class":1085,"line":368},[1083,3177,3178],{},"        status: Literal['ok', 'error'],\n",[1083,3180,3181],{"class":1085,"line":375},[1083,3182,3183],{},"        output: dict | None = None,\n",[1083,3185,3186],{"class":1085,"line":157},[1083,3187,3188],{},"        usage_id: UUID | None = None,\n",[1083,3190,3191],{"class":1085,"line":182},[1083,3192,3193],{},"        error: dict | None = None,\n",[1083,3195,3196],{"class":1085,"line":290},[1083,3197,3198],{},"        attributes: dict | None = None,\n",[1083,3200,3201],{"class":1085,"line":280},[1083,3202,3203],{},"    ) -> None: ...\n",[1083,3205,3206],{"class":1085,"line":272},[1083,3207,3208],{},"    async def tree(self, root_run_id: UUID, organization_id: UUID) -> list[RunSpan]:\n",[1083,3210,3211],{"class":1085,"line":2066},[1083,3212,3213],{},"        \"\"\"Every span of one tree, in (started_at, span_id) order.\n",[1083,3215,3216],{"class":1085,"line":2072},[1083,3217,1847],{"emptyLinePlaceholder":1846},[1083,3219,3220],{"class":1085,"line":2078},[1083,3221,3222],{},"        It pages by keyset. max_rows is 1,000 and PostgREST truncates past it\n",[1083,3224,3225],{"class":1085,"line":2084},[1083,3226,3227],{},"        without saying so, and a discovery tree holds tens of thousands.\n",[1083,3229,3230],{"class":1085,"line":2090},[1083,3231,3232],{},"        \"\"\"\n",[1083,3234,3235],{"class":1085,"line":520},[1083,3236,3237],{},"    async def close_orphans(\n",[1083,3239,3240],{"class":1085,"line":318},[1083,3241,3242],{},"        self, run_ids: Sequence[UUID], organization_id: UUID, error: dict\n",[1083,3244,3245],{"class":1085,"line":327},[1083,3246,3247],{},"    ) -> int:\n",[1083,3249,3250],{"class":1085,"line":2108},[1083,3251,3252],{},"        \"\"\"The open spans of the Runs the reaper just failed. One statement.\"\"\"\n",[1083,3254,3255],{"class":1085,"line":2114},[1083,3256,3257],{},"    async def close_orphans_of_ended_runs(self, error: dict, limit: int) -> int:\n",[1083,3259,3260],{"class":1085,"line":2120},[1083,3261,3262],{},"        \"\"\"Every span still running under a Run that has ended. Not scoped to one Run.\n",[1083,3264,3265],{"class":1085,"line":2126},[1083,3266,1847],{"emptyLinePlaceholder":1846},[1083,3268,3269],{"class":1085,"line":2132},[1083,3270,3271],{},"        A read with the join, then a write by span id. Never one UPDATE with an\n",[1083,3273,3275],{"class":1085,"line":3274},28,[1083,3276,3277],{},"        embedded filter; see the sweep section above for what that closes.\n",[1083,3279,3281],{"class":1085,"line":3280},29,[1083,3282,3232],{},[1083,3284,3285],{"class":1085,"line":481},[1083,3286,3287],{},"    async def count_of_kind(\n",[1083,3289,3291],{"class":1085,"line":3290},31,[1083,3292,3293],{},"        self, run_id: UUID, organization_id: UUID, kind: SpanKind,\n",[1083,3295,3297],{"class":1085,"line":3296},32,[1083,3298,3299],{},"        *, exclude_replayed: bool = False,\n",[1083,3301,3303],{"class":1085,"line":3302},33,[1083,3304,3247],{},[1083,3306,3308],{"class":1085,"line":3307},34,[1083,3309,3310],{},"        \"\"\"The reader of (run_id, kind). It keys on the Run and never the tree.\n",[1083,3312,3314],{"class":1085,"line":3313},35,[1083,3315,1847],{"emptyLinePlaceholder":1846},[1083,3317,3319],{"class":1085,"line":3318},36,[1083,3320,3321],{},"        An exact count travels in Content-Range, so no row travels at all.\n",[1083,3323,3325],{"class":1085,"line":3324},37,[1083,3326,1847],{"emptyLinePlaceholder":1846},[1083,3328,3330],{"class":1085,"line":3329},38,[1083,3331,3332],{},"        exclude_replayed drops a span whose attributes carry replayed = true.\n",[1083,3334,3336],{"class":1085,"line":3335},39,[1083,3337,3338],{},"        The tool call ceiling needs it and the turn ceiling does not: a replayed\n",[1083,3340,3341],{"class":1085,"line":503},[1083,3342,3343],{},"        call made no call, and a replayed model turn really spent the tokens.\n",[1083,3345,3347],{"class":1085,"line":3346},41,[1083,3348,3349],{},"        The key is absent on most spans, so the filter is a two member `or`\n",[1083,3351,3353],{"class":1085,"line":3352},42,[1083,3354,3355],{},"        rather than a `not.eq`, which would drop every span with no key.\n",[1083,3357,3359],{"class":1085,"line":3358},43,[1083,3360,3232],{},[1083,3362,3364],{"class":1085,"line":3363},44,[1083,3365,1847],{"emptyLinePlaceholder":1846},[1083,3367,3369],{"class":1085,"line":3368},45,[1083,3370,3371],{},"class UsageMeter(Protocol):\n",[1083,3373,3375],{"class":1085,"line":3374},46,[1083,3376,3377],{},"    async def record(self, usage: UsageRecord) -> UUID:\n",[1083,3379,3381],{"class":1085,"line":3380},47,[1083,3382,3383],{},"        \"\"\"Writes one ai_usage_log row and returns its id. Not best effort.\"\"\"\n",[1083,3385,3387],{"class":1085,"line":3386},48,[1083,3388,3389],{},"    async def run_usage(\n",[1083,3391,3393],{"class":1085,"line":3392},49,[1083,3394,3395],{},"        self, root_run_id: UUID, organization_id: UUID\n",[1083,3397,3398],{"class":1085,"line":491},[1083,3399,3400],{},"    ) -> UsageSummary:\n",[1083,3402,3404],{"class":1085,"line":3403},51,[1083,3405,3406],{},"        \"\"\"One call to agent.run_usage(). It reads no span.\n",[1083,3408,3410],{"class":1085,"line":3409},52,[1083,3411,1847],{"emptyLinePlaceholder":1846},[1083,3413,3415],{"class":1085,"line":3414},53,[1083,3416,3417],{},"        Raises MissingRunUsage on an empty answer or an odd column.\n",[1083,3419,3421],{"class":1085,"line":3420},54,[1083,3422,3232],{},[1083,3424,3426],{"class":1085,"line":3425},55,[1083,3427,3428],{},"    async def organization_day_cents(\n",[1083,3430,3432],{"class":1085,"line":3431},56,[1083,3433,3434],{},"        self, organization_id: UUID\n",[1083,3436,3438],{"class":1085,"line":3437},57,[1083,3439,3440],{},"    ) -> DayCost:\n",[1083,3442,3444],{"class":1085,"line":3443},58,[1083,3445,3446],{},"        \"\"\"One call to agent.organization_day_cost(). Accrual reads it.\n",[1083,3448,3450],{"class":1085,"line":3449},59,[1083,3451,1847],{"emptyLinePlaceholder":1846},[1083,3453,3454],{"class":1085,"line":529},[1083,3455,3456],{},"        Raises MissingDayCost on the same two shapes.\n",[1083,3458,3460],{"class":1085,"line":3459},61,[1083,3461,3232],{},[1083,3463,3465],{"class":1085,"line":3464},62,[1083,3466,1847],{"emptyLinePlaceholder":1846},[1083,3468,3470],{"class":1085,"line":3469},63,[1083,3471,3472],{},"class SpanRecorder:\n",[1083,3474,3476],{"class":1085,"line":3475},64,[1083,3477,3478],{},"    def span(\n",[1083,3480,3482],{"class":1085,"line":3481},65,[1083,3483,3158],{},[1083,3485,3487],{"class":1085,"line":3486},66,[1083,3488,3173],{},[1083,3490,3492],{"class":1085,"line":3491},67,[1083,3493,3494],{},"        kind: SpanKind,\n",[1083,3496,3498],{"class":1085,"line":3497},68,[1083,3499,3500],{},"        name: str,\n",[1083,3502,3504],{"class":1085,"line":3503},69,[1083,3505,3506],{},"        input: Payload | None = None,\n",[1083,3508,3510],{"class":1085,"line":3509},70,[1083,3511,3198],{},[1083,3513,3515],{"class":1085,"line":3514},71,[1083,3516,3517],{},"    ): ...\n",[807,3519,3520,3523,3524,1309,3526,3529],{},[819,3521,3522],{},"organization_day_cents"," is not here. Its only caller is ",[819,3525,1910],{},[819,3527,3528],{},"agent.cost_ceilings"," does not exist yet, so it ships with accrual rather than ahead of it.",[807,3531,3532,3535,3536,3538,3539,2206,3542,3544,3545,2206,3548,3551,3552,1063,3555,3558,3559,3561],{},[819,3533,3534],{},"LiveEventPublisher"," is declared beside ",[819,3537,1116],{},", in ",[819,3540,3541],{},"recorder.py",[819,3543,1381],{}," is the one implementation, in ",[819,3546,3547],{},"runtime\u002Fevents\u002Fpublisher.py",[819,3549,3550],{},"PublishingAgentEventSink"," is the second caller, for ",[819,3553,3554],{},"text.delta",[819,3556,3557],{},"tool.updated",", which no span produces. ",[819,3560,971],{}," is the third, for the run lifecycle events.",[868,3563,3565],{"id":3564},"write-cost","Write cost",[807,3567,3568],{},"The numbers are small enough to keep the design boring, and worth stating so nobody guesses.",[811,3570,3573],{"className":3571,"code":3572,"language":816,"meta":817},[814],"one agent run, 30 model turns, 20 tool calls\n  spans      1 run + 4 segment + 30 llm + 20 tool  = 55 rows, 110 statements\n  heartbeat  2 more statements per span open       = 110 statements\n  usage      30 model rows + any metered vendor rows\n                                                   = 250 round trips in total\n",[819,3574,3572],{"__ignoreMap":817},[807,3576,3577],{},"Insert and update are two statements per span, on the same connection. Do not batch a span open across a step boundary. A crash must leave the open span visible, which is the whole point of writing it first.",[807,3579,3580,3583,3584,3586,3587,3589,3590,3593,3594,3597],{},[898,3581,3582],{},"Count the heartbeat, because it doubles the number."," Every span open fires the two conditional ",[819,3585,1544],{},"s above, and ",[819,3588,2728],{}," reaches Postgres through PostgREST, so each statement is an HTTP request. A conditional statement suppresses the ",[898,3591,3592],{},"write",", not the ",[898,3595,3596],{},"request",": a statement that matches zero rows still costs a round trip. A run that looked like 110 statements is 250.",[807,3599,3600,3603],{},[898,3601,3602],{},"So the recorder guards in the process."," It already knows when it last wrote each row, so it keeps the last write time and skips the call entirely inside the window. One worker owns its Run's row, so that check is exact for it. Many workers touch the root row, so the local check is per process and still removes most of the traffic.",[807,3605,1538,3606,3609,3610,3613,3614,3617],{},[898,3607,3608],{},"There is no second guard in the statement, because PostgREST cannot write one."," A filter value is a literal that Postgres casts; it is not SQL that Postgres evaluates. ",[819,3611,3612],{},"heartbeat_at=lt.now"," parses, because ",[819,3615,3616],{},"now"," is a timestamp input string, and every interval form is refused:",[811,3619,3622],{"className":3620,"code":3621,"language":816,"meta":817},[814],"GET \u002Fruns?heartbeat_at=lt.now                      200\nGET \u002Fruns?heartbeat_at=lt.now()-interval'10 seconds'\n  -> 400  22007  invalid input syntax for type timestamp with time zone\n",[819,3623,3621],{"__ignoreMap":817},[807,3625,3626,3627,3629,3630,3632],{},"An application clock cannot stand in for it. The write itself sends the string ",[819,3628,3616],{},", so ",[819,3631,1171],{}," carries the database clock, and a dyno running one minute behind would compare a server timestamp against a cutoff a minute in the past. The filter would then match nothing, the heartbeat would never be written, and the reaper would fail a healthy Run at 120 seconds. A guard whose failure mode is a dead Run is worse than no guard.",[807,3634,3635,3638,3639,3642],{},[898,3636,3637],{},"Losing it costs writes and never costs correctness, which is the opposite of the fear."," The process guard only ever skips a write this process already made; a fresh process has no memory and writes at once. So no heartbeat is missed. And the guard is per ",[898,3640,3641],{},"process",", not per Run: one worker holds hundreds of Runs of one tree in one event loop, so five hundred children on ten dynos touch the root ten times a minute, not five hundred.",[807,3644,1538,3645,3648],{},[898,3646,3647],{},"The guard is a map keyed by run id, and never two timestamps."," One worker process serves many runs at once, and an email sequence puts hundreds of them in one event loop. Two scalars let the newest run suppress the heartbeat of every other run in the process, and the reaper then fails healthy work. The map holds one entry per live run and one per root, and it evicts the oldest entry past a fixed size, because a worker lives for days. An evicted entry costs one extra round trip, which the SQL guard still answers correctly.",[807,3650,3651],{},"That takes the run above from 250 round trips to about 120.",[807,3653,3654],{},"Indexes:",[811,3656,3659],{"className":3657,"code":3658,"language":816,"meta":817},[814],"agent.spans  (run_id, started_at)\nagent.spans  (root_run_id, started_at)\nagent.spans  (run_id, kind)                          -- the ceiling arithmetic\nagent.spans  (run_id) where status = 'running'       -- close_orphans\nagent.spans  (parent_span_id) where parent_span_id is not null\nagent.runs   (id) where status in ('queued','running')\nai_usage_log (agent_root_run_id) where agent_root_run_id is not null\nai_usage_log (organization_id, created_at)\n",[819,3660,3658],{"__ignoreMap":817},[807,3662,3663],{},"Three of those were wrong on this page until 2026-08-20, and each was wrong in the same way: a key that reads well and answers no query the platform makes.",[807,3665,3666],{},"Two of those are easy to leave out, because neither serves a screen. Both serve a\nloop that runs constantly, and both were measured on 200,000 runs and a 27,500 span tree.",[892,3668,3669,3735,3753,3802],{},[895,3670,3671,3676,3677,3680,3681,3684,3685,3687,3688,3690,3700,3701,3703,3704,3706,3707,3710,3711,3714,3715,3717,3724,3022,3727,3730,3731,3734],{},[898,3672,3673],{},[819,3674,3675],{},"agent.spans (run_id, kind)"," serves the ceiling arithmetic. A segment computes\n",[819,3678,3679],{},"ceiling - count of llm spans of this Run"," before it builds its request, and\n",[819,3682,3683],{},"(run_id, started_at)"," cannot answer it, because the filter is on ",[819,3686,1326],{},".\nWithout the index that count is a sequential scan, at the top of every\nsegment and every node: 509 buffers against 11, and 8.3 ms against 1.8 ms as an\nindex only scan.",[1660,3689],{},[898,3691,3692,3693,3696,3697,3699],{},"The leading column is ",[819,3694,3695],{},"run_id",", and it was ",[819,3698,910],{}," until the ceiling scope\nwas corrected."," Four of the five ceilings bound one Run, and only ",[819,3702,2242],{},"\nsums the tree — which reads ",[819,3705,1704],{},", not a span. So no query counts spans by\nkind across a tree, and ",[819,3708,3709],{},"(root_run_id, kind)"," has no reader left. ",[819,3712,3713],{},"20260819120200","\nshipped it, and a later migration drops it and creates the new key. A merged\nmigration is not edited, so the drop is a statement a reader can find.",[1660,3716],{},[898,3718,3719,3720,3723],{},"The count itself is a ",[819,3721,3722],{},"HEAD"," with an exact count, not a read of the rows.",[819,3725,3726],{},"Prefer: count=exact",[819,3728,3729],{},"limit=0"," answers in ",[819,3732,3733],{},"Content-Range",", so the ceiling\ncosts one index only scan and carries no row back.",[895,3736,3737,3741,3742,3745,3746,3748,3749,3752],{},[898,3738,3739],{},[819,3740,1526],{}," serves ",[819,3743,3744],{},"close_orphans",". The key\nis ",[819,3747,3695],{},", not ",[819,3750,3751],{},"status",": the predicate already fixes the status, so a fixed key\ncarries no information and the index degenerates to one value.",[895,3754,3755,1309,3760,3766,3767,3769,3770,1309,3772,3774,3775,1138,3778,1538,3780,3785,3786,3789,3790,3793,3794,3797,3798,3801],{},[898,3756,3757],{},[819,3758,3759],{},"ai_usage_log (agent_root_run_id)",[898,3761,3762,3763,3765],{},"not ",[819,3764,3695],{},", which does not exist on\nthat table."," The live columns are ",[819,3768,1743],{},", a key into\n",[819,3771,1747],{},[819,3773,2144],{},", a key into the legacy\n",[819,3776,3777],{},"public.agent_runs",[1660,3779],{},[898,3781,3782,3784],{},[819,3783,2144],{}," looks like the obvious column and it must not be used."," It\ncarries ",[819,3787,3788],{},"ai_usage_log_agent_run_uniq",", a ",[898,3791,3792],{},"UNIQUE"," index, so it holds at most one\nusage row per Run. A model call writes one row per call, and one Run makes thirty.\nThe second write answers ",[819,3795,3796],{},"23505"," and metering is terminal, so the Run fails. The new\ncolumn is ",[819,3799,3800],{},"agent_root_run_id",", it is nullable, it takes no unique index and no\nforeign key.",[895,3803,3804,3809,3810,3812,3813,3815,964,3824,3826],{},[898,3805,3806],{},[819,3807,3808],{},"agent.runs (id) where status in ('queued','running')"," serves the reaper,\nwhich runs every minute over a table kept for 13 months. Without it the cron is a\nparallel sequential scan: 3,461 buffers and 29.7 ms, against 8 buffers and 0.4 ms.\nThe partial predicate keeps the index at tens of kilobytes however large the table\ngrows, because only live Runs are in it, so the reaper reads a few hundred rows and\nfilters ",[819,3811,1171],{}," in the heap.",[1660,3814],{},[898,3816,3817,3818,3821,3822,1138],{},"The key is ",[819,3819,3820],{},"id",", and it must never be ",[819,3823,1171],{},[819,3825,1116],{}," writes that\ncolumn every ten seconds per Run, and an update of an indexed column cannot be a HOT\nupdate. A 2,000 row probe with page headroom measured 902 HOT updates with the column\nunindexed and 0 with it. Indexing the column the hot loop writes would trade the\nreaper's 29 ms a minute for a new index entry and a dead tuple on every heartbeat.",[868,3828,3830],{"id":3829},"evaluation","Evaluation",[807,3832,3833,3834,3837,3838,3840],{},"Production observability and offline evaluation share the span shape, not the run time path. An evaluation run writes ordinary spans and carries its source in the run row. Evaluation may sample production spans as fixtures. The ",[819,3835,3836],{},"eval_spans"," concept converges into ",[819,3839,880],{}," rather than living beside it.",[868,3842,3844],{"id":3843},"scenarios-that-shaped-this-design","Scenarios that shaped this design",[825,3846,3847,3857],{},[828,3848,3849],{},[831,3850,3851,3854],{},[834,3852,3853],{},"Scenario",[834,3855,3856],{},"What answers it",[841,3858,3859,3871,3883,3891,3899,3907,3918,3930,3938,3949,3959,3967,3977,3985,3995,4006],{},[831,3860,3861,3864],{},[846,3862,3863],{},"The worker dies inside a tool call",[846,3865,1445,3866,3868,3869],{},[819,3867,1448],{},", and the reaper closes it ",[819,3870,1616],{},[831,3872,3873,3876],{},[846,3874,3875],{},"A segment retries after a crash",[846,3877,3878,3879,3882],{},"Replayed calls write spans marked ",[819,3880,3881],{},"replayed",", with no usage",[831,3884,3885,3888],{},[846,3886,3887],{},"Someone asks why one run cost twice as much",[846,3889,3890],{},"Two attempts appear in the tree, each with its own usage rows",[831,3892,3893,3896],{},[846,3894,3895],{},"Redis is down while a person watches",[846,3897,3898],{},"The durable span is written, and the client refetches on reconnect",[831,3900,3901,3904],{},[846,3902,3903],{},"The span write fails but the email was sent",[846,3905,3906],{},"Sentry receives the tracing fault; the send stays successful",[831,3908,3909,3912],{},[846,3910,3911],{},"The usage write fails after a model call",[846,3913,3914,3915,3917],{},"The run fails at once with ",[819,3916,1918],{},", and does not retry",[831,3919,3920,3923],{},[846,3921,3922],{},"A tool returns 4 MB of pages",[846,3924,3925,3927,3928],{},[819,3926,1648],{}," drops whole items, and the span sets ",[819,3929,1672],{},[831,3931,3932,3935],{},[846,3933,3934],{},"A vendor key appears in a handler response",[846,3936,3937],{},"The tool result boundary removes it before the span sees it",[831,3939,3940,3943],{},[846,3941,3942],{},"A parent workflow has three child runs",[846,3944,3945,3946,3948],{},"Spans carry ",[819,3947,910],{},", so one query builds the tree",[831,3950,3951,3954],{},[846,3952,3953],{},"Finance asks what one customer spent last quarter",[846,3955,3956,3958],{},[819,3957,1704],{},", kept 13 months, and its daily rollup",[831,3960,3961,3964],{},[846,3962,3963],{},"An auditor asks what an agent did 6 months ago",[846,3965,3966],{},"The run row and its cost survive; the step detail does not",[831,3968,3969,3972],{},[846,3970,3971],{},"An auditor asks what was refused",[846,3973,3974,3976],{},[819,3975,1049],{}," holds every refusal, gate and stop",[831,3978,3979,3982],{},[846,3980,3981],{},"A customer asks us to erase one person",[846,3983,3984],{},"The deletion job redacts span payloads and keeps the tree and the cost",[831,3986,3987,3990],{},[846,3988,3989],{},"A customer leaves and asks for full deletion",[846,3991,3992,3993],{},"One job over every table carrying ",[819,3994,902],{},[831,3996,3997,4000],{},[846,3998,3999],{},"An operator wants the Inngest trace for a run",[846,4001,4002,4005],{},[819,4003,4004],{},"execution_ref"," on the run row links out to it",[831,4007,4008,4011],{},[846,4009,4010],{},"An engineer asks whether an alert needs a metrics store",[846,4012,4013],{},"Five scheduled queries on Postgres, run by Inngest cron",[868,4015,4017],{"id":4016},"rules","Rules",[892,4019,4020,4023,4030,4033,4036,4039,4042,4045,4048,4051,4062,4065,4068,4071,4074,4080,4086,4095,4100,4103,4106,4109,4112,4115,4118,4121,4124],{},[895,4021,4022],{},"Every meaningful unit of work opens a span, owned by the code doing the work.",[895,4024,4025,4026,1063,4028,1138],{},"A span carries ",[819,4027,902],{},[819,4029,910],{},[895,4031,4032],{},"No durable event table beside spans.",[895,4034,4035],{},"One run table, one span table, one usage meter.",[895,4037,4038],{},"The durable write precedes the best effort live event.",[895,4040,4041],{},"A tracing failure never fails a successful business action.",[895,4043,4044],{},"Metering is not best effort, and a metering failure is terminal.",[895,4046,4047],{},"An open span has an owner that closes it, even after a crash.",[895,4049,4050],{},"A replayed call is recorded and marked, not hidden.",[895,4052,4053,4055,4056,4058,4059,4061],{},[819,4054,1116],{}," writes ",[819,4057,1171],{}," between the claim and the finalize: at most once every 10 seconds for the run, and once every 60 for its root. The window is tested in the process, because a PostgREST filter holds a literal and not an expression. ",[819,4060,971],{}," owns the insert and each lifecycle write.",[895,4063,4064],{},"The process guard is a map keyed by run id, never two timestamps. One worker serves many runs at once.",[895,4066,4067],{},"A filter value is a literal Postgres casts. It is never SQL Postgres evaluates.",[895,4069,4070],{},"An index key is the column a query filters on. A key the predicate already fixes carries no information.",[895,4072,4073],{},"A write never filters on an embedded resource. PostgREST drops that filter and keeps it in the response.",[895,4075,4076,4077,4079],{},"A read that grows with a run tree pages by keyset. ",[819,4078,1802],{}," truncates in silence.",[895,4081,4082,4083,4085],{},"A sum over ",[819,4084,1704],{}," is a read only database function. PostgREST refuses an aggregate.",[895,4087,4088,4091,4092,4094],{},[819,4089,4090],{},"ai_usage_log.agent_root_run_id"," is the agentic column. ",[819,4093,2144],{}," belongs to the legacy stack and is unique.",[895,4096,4097,4098,1138],{},"No type in this package is named ",[819,4099,1239],{},[895,4101,4102],{},"Span payloads are bounded and redacted; prompts are not stored by default.",[895,4104,4105],{},"Sentry is for our defects. Spans are for agent work.",[895,4107,4108],{},"A defect on the path of one call reaches Sentry one time, and the log every time. A durable step bounds it per step execution. A component with no durable step holds the report itself. A static defect keeps its entry for the life of the component. A fault that recovers takes a time window, because a fault that alternates makes every failure follow a success.",[895,4110,4111],{},"An alert is a query on a schedule, until scale proves otherwise.",[895,4113,4114],{},"Health dashboards are linked, not rebuilt.",[895,4116,4117],{},"A retention window is not a deletion path. Both exist.",[895,4119,4120],{},"Erasure redacts a payload and keeps the timing, the tree and the cost.",[895,4122,4123],{},"A table that stores content joins the erasure list in the change that creates it.",[895,4125,4126],{},"A new table that stores a payload joins the deletion job in the same change.",[868,4128,4130],{"id":4129},"minimum-contract-tests","Minimum contract tests",[892,4132,4133,4142,4145,4148,4154,4157,4160,4163,4170,4173,4176,4179,4189,4192,4198,4201,4204,4207,4210,4213,4216,4219,4222,4225,4230],{},[895,4134,4135,4136,4138,4139,1138],{},"The span context manager completes ",[819,4137,1580],{},", and re-raises after writing ",[819,4140,4141],{},"error",[895,4143,4144],{},"A parent and child context produce the correct tree.",[895,4146,4147],{},"A span carries the organization of its run, and RLS refuses a foreign read.",[895,4149,4150,4151,4153],{},"One query on ",[819,4152,910],{}," returns the spans of a whole run tree.",[895,4155,4156],{},"A pub\u002Fsub failure leaves the durable span intact.",[895,4158,4159],{},"A span write failure does not fail the tool call it describes.",[895,4161,4162],{},"A usage write failure fails the run, and does not trigger a segment retry.",[895,4164,4165,4166,1449,4168,1138],{},"A replayed tool call writes a span marked ",[819,4167,3881],{},[819,4169,1437],{},[895,4171,4172],{},"The reaper closes open spans when it fails a run with a stale heartbeat.",[895,4174,4175],{},"The second orphan sweep finds a running span under an ended run without being told which run.",[895,4177,4178],{},"The second orphan sweep leaves a running span under a live run untouched.",[895,4180,4181,4182,4184,4185,4188],{},"A tree of more than ",[819,4183,1802],{}," spans returns every span, in ",[819,4186,4187],{},"(started_at, span_id)"," order.",[895,4190,4191],{},"A span open that fails leaves the parent span current, and the next child still writes.",[895,4193,4194,4195,4197],{},"A truncated input keeps ",[819,4196,1652],{}," true after the close writes the output.",[895,4199,4200],{},"A span output that is a string is stored and bounded, and it does not raise.",[895,4202,4203],{},"Two runs in one process each write their own heartbeat inside one guard window.",[895,4205,4206],{},"Thirty model calls in one run write thirty usage rows.",[895,4208,4209],{},"Thirty calls of a currency amount that rounds to zero cents each sum to a non zero total.",[895,4211,4212],{},"The ceiling count for one Run reads only that Run's spans, not its tree's.",[895,4214,4215],{},"An oversized tool result is stored truncated, the flag is set, and the stored value still parses.",[895,4217,4218],{},"A run tree total equals the sum of its usage rows.",[895,4220,4221],{},"The run explorer rebuilds current state after a disconnect, from the run and its spans alone.",[895,4223,4224],{},"Erasure of a person leaves no payload in any span, and leaves the span tree walkable.",[895,4226,4227,4228,1138],{},"Erasure of an organization leaves no row in any table carrying its ",[819,4229,902],{},[895,4231,4232],{},"A run tree total still sums correctly after an erasure.",[4234,4235,4236],"style",{},"html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html pre.shiki code .svObZ, html code.shiki .svObZ{--shiki-default:#B392F0}html pre.shiki code .sDLfK, html code.shiki .sDLfK{--shiki-default:#79B8FF}html pre.shiki code .sU2Wk, html code.shiki .sU2Wk{--shiki-default:#9ECBFF}html pre.shiki code .snl16, html code.shiki .snl16{--shiki-default:#F97583}html pre.shiki code .s95oV, html code.shiki .s95oV{--shiki-default:#E1E4E8}",{"title":817,"searchDepth":32,"depth":233,"links":4238},[4239,4243,4246,4247,4248,4249,4255,4256,4257,4261,4262,4263,4264,4265,4266,4267,4268],{"id":870,"depth":32,"text":871,"children":4240},[4241,4242],{"id":915,"depth":233,"text":916},{"id":1026,"depth":233,"text":1027},{"id":1070,"depth":32,"text":1071,"children":4244},[4245],{"id":1162,"depth":233,"text":1163},{"id":1334,"depth":32,"text":1335},{"id":1389,"depth":32,"text":1390},{"id":1633,"depth":32,"text":1634},{"id":1698,"depth":32,"text":1699,"children":4250},[4251,4252,4253,4254],{"id":1773,"depth":233,"text":1774},{"id":1987,"depth":233,"text":1988},{"id":2163,"depth":233,"text":2164},{"id":2246,"depth":233,"text":2247},{"id":2310,"depth":32,"text":2311},{"id":2548,"depth":32,"text":2549},{"id":2698,"depth":32,"text":2699,"children":4258},[4259,4260],{"id":2705,"depth":233,"text":2706},{"id":2772,"depth":233,"text":2773},{"id":62,"depth":32,"text":2948},{"id":3062,"depth":32,"text":3063},{"id":3126,"depth":32,"text":3127},{"id":3564,"depth":32,"text":3565},{"id":3829,"depth":32,"text":3830},{"id":3843,"depth":32,"text":3844},{"id":4016,"depth":32,"text":4017},{"id":4129,"depth":32,"text":4130},"md",{},[4272,4273,4274,4275],"engineering\u002Fsystem-design\u002Fagentic-platform\u002Fcontract","engineering\u002Fsystem-design\u002Fagentic-platform\u002Fruntime\u002Fexecution","engineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fpolicy-and-governance","engineering\u002Fsystem-design\u002Fagentic-platform\u002Fcapabilities\u002Ftools-and-integrations",{"title":278,"description":279},"engineering\u002Fsystem-design\u002Fagentic-platform\u002Fplanes\u002Fobservability-and-operations",[209,217,282,283,64,284],"upp10_VRv5BlJkdNjJ4NWLQC97WjeISdK-UfbXlWoBY",1788650194272]