[{"data":1,"prerenderedAt":2418},["ShallowReactive",2],{"site-nav-content":3,"blog:\u002Fblog\u002Fagent-harness-architecture":179,"blog-index-copy":1018,"blog:\u002Fblog\u002Fagent-harness-architecture:surround":1039,"hiring-banner-content":2387,"site-cta-content":2399},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":132,"id":163,"title":164,"archived":165,"authors":166,"badge":166,"body":167,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":175,"relatedHeading":166,"seo":176,"series":166,"sitemap":165,"status":166,"stem":177,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":178},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","\u002Foverview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","\u002Fproduct\u002Fconflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","\u002Fproduct\u002Fagent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","\u002Fproduct\u002Flifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","\u002Fproduct\u002Fwiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","\u002Fproduct\u002Fworkstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","\u002Fproduct\u002Fperception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","\u002Fproduct\u002Fgovernance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","\u002Fmodels",{"label":47,"to":48},"Pricing","\u002Fpricing",{"label":50,"to":51},"Integrations","\u002Fintegrations",{"label":53,"to":54},"Security","\u002Fsecurity",{"label":56,"to":57},"Partners","\u002Fpartners",{"label":59,"to":60},"Insights","\u002Fblog",{"productHeading":5,"companyHeading":62,"resourcesHeading":63,"legalHeading":64,"docsLabel":65,"docsUrl":66,"statementLines":67,"copyright":70,"companyLinks":71,"resourcesLinks":86,"legalLinks":102,"socialLinks":109,"bottomLinks":119},"Company","Resources","Legal","Docs","https:\u002F\u002Fdocs.gonimbus.ai",[68,69],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[72,73,74,75,76,78,81,84],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":77,"to":57},"Partner Program",{"label":79,"to":80},"Careers","\u002Fcareers",{"label":82,"to":83},"System status","\u002Fstatus",{"label":7,"to":85},"\u002Fcontact",[87,90,93,96,99],{"label":88,"to":89},"Glossary","\u002Fglossary",{"label":91,"to":92},"Compare","\u002Fcompare",{"label":94,"to":95},"Evaluate","\u002Fevaluate",{"label":97,"to":98},"Problems","\u002Fproblems",{"label":100,"to":101},"Use cases","\u002Fuse-cases",[103,106],{"label":104,"to":105},"Terms of Service","\u002Fterms",{"label":107,"to":108},"Privacy Policy","\u002Fprivacy",[110,113,116],{"label":111,"href":112},"LinkedIn","https:\u002F\u002Fwww.linkedin.com\u002Fcompany\u002Fgonimbusai\u002F",{"label":114,"href":115},"X","https:\u002F\u002Fx.com\u002Fgonimbusai",{"label":117,"href":118},"Instagram","https:\u002F\u002Fwww.instagram.com\u002Fgonimbus_ai\u002F",[120,122,124,127,128],{"label":121,"to":105},"Terms",{"label":123,"to":108},"Privacy",{"label":125,"to":126},"Compliance","\u002Fcompliance",{"label":82,"to":83},{"label":129,"to":130,"external":131},"LLMs.txt","\u002Fllms.txt",true,{"text":133,"prompt":134},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":135,"platforms":137},{" Summarize the highlights from Nimbus's website":136},"https:\u002F\u002Fgonimbus.ai",[138,143,148,153,158],{"name":139,"label":140,"icon":141,"hrefPrefix":142},"chatgpt","ChatGPT","simple-icons:openai","https:\u002F\u002Fchatgpt.com\u002F?prompt=",{"name":144,"label":145,"icon":146,"hrefPrefix":147},"perplexity","Perplexity","mdi:magnify","https:\u002F\u002Fwww.perplexity.ai\u002Fsearch\u002Fnew?q=",{"name":149,"label":150,"icon":151,"hrefPrefix":152},"grok","Grok","simple-icons:x","https:\u002F\u002Fx.com\u002Fi\u002Fgrok?text=",{"name":154,"label":155,"icon":156,"hrefPrefix":157},"claude","Claude","simple-icons:anthropic","https:\u002F\u002Fclaude.ai\u002Fnew?q=",{"name":159,"label":160,"icon":161,"hrefPrefix":162},"google-ai","Google AI","simple-icons:google","https:\u002F\u002Fwww.google.com\u002Fsearch?udm=50&aep=11&q=","content\u002Fshared\u002Fnav.md","Site navigation",false,null,{"type":168,"value":169,"toc":170},"minimark",[],{"title":171,"searchDepth":172,"depth":172,"links":173},"",2,[],"md","\u002Fshared\u002Fnav",{"title":164,"description":171},"shared\u002Fnav","1dD7ahDRl0SQ4hz53-kKo0tEFrGaLuaztZ3PPfp6a9k",{"id":180,"title":181,"archived":165,"authors":182,"badge":185,"body":187,"date":1008,"definedTerm":166,"department":166,"description":1009,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1010,"relatedHeading":166,"seo":1011,"series":1012,"sitemap":131,"status":166,"stem":1013,"subhead":166,"tags":1014,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1017},"content\u002Fblog\u002Fagent-harness-architecture.md","Agent Harness Architecture",[183],{"name":184,"to":136},"Nimbus Research",{"label":186},"Architecture",{"type":168,"value":188,"toc":990},[189,197,231,254,259,321,349,353,379,387,390,394,397,415,426,446,452,476,485,495,506,516,527,533,539,542,546,595,603,607,614,723,726,738,744,752,759,767,784,790,794,804,808,813,816,820,827,831,839,843,850,854,866,870,880,884],[190,191,192,196],"p",{},[193,194,195],"strong",{},"Agent harness architecture"," is the design of the runtime around a model: who owns the loop, how tools run, what context is injected, which hooks can refuse, which identity the tools use, and how “done” is checked without taking the model’s word.",[190,198,199,206,207,212,213,217,218,217,222,217,226,230],{},[200,201,205],"a",{"href":202,"rel":203},"https:\u002F\u002Fwww.langchain.com\u002Fblog\u002Fthe-anatomy-of-an-agent-harness",[204],"nofollow","LangChain’s anatomy"," is the public parts list: prompts, tools and MCP, bundled infrastructure (filesystem, sandbox, browser), orchestration (subagents, routing), hooks and middleware (compaction, lint, continuation). ",[200,208,211],{"href":209,"rel":210},"https:\u002F\u002Fwww.databricks.com\u002Fblog\u002Fai-harness",[204],"Databricks"," groups the same into tools, memory, workspace, guardrails. This article is that list as an architecture you can inspect — then the mapping onto company jobs: ",[200,214,216],{"href":215},"what-is-an-ai-workstream","workstreams",", ",[200,219,221],{"href":220},"agent-team-architecture","agent teams",[200,223,225],{"href":224},"connector-and-permissions-architecture","connectors",[200,227,229],{"href":228},"what-is-write-back-governance","write-back",".",[190,232,233,234,238,239,243,244,248,249,253],{},"It is not a novel about kernels. It is not ",[200,235,237],{"href":236},"multi-agent-ai-architecture","multi-agent protocol"," (hand-offs between specialists) and not ",[200,240,242],{"href":241},"human-in-the-loop-approval-architecture","HITL state machines"," (quote → sign → execute), though a complete outer harness contains both. Start from ",[200,245,247],{"href":246},"what-is-an-agent-harness","what is an agent harness",". Use ",[200,250,252],{"href":251},"how-to-evaluate-an-agent-harness","how to evaluate"," as the test of this diagram.",[255,256,258],"h2",{"id":257},"words-youll-hear","Words you’ll hear",[260,261,262,278,288,299,311],"ul",{},[263,264,265,268,269,273,274,277],"li",{},[193,266,267],{},"Control plane vs data plane."," Control: grants, budgets, gates, routing policy — known independently of the model. Data: tokens, tool results, artefacts. If the orchestrator is only a system prompt, a jailbreak ",[270,271,272],"em",{},"is"," a privilege escalation. ",[200,275,276],{"href":236},"Multi-agent architecture"," already said this; it is a harness invariant.",[263,279,280,283,284,230],{},[193,281,282],{},"Workspace."," Inner: checkout \u002F sandbox. Outer: workstream. ",[200,285,287],{"href":286},"inner-vs-outer-agent-harness","Inner vs outer",[263,289,290,293,294,230],{},[193,291,292],{},"Tool plane vs write plane."," Reads default on. Mutations fail-closed. MCP may implement both; architecture must split them. ",[200,295,298],{"href":296,"rel":297},"https:\u002F\u002Fmodelcontextprotocol.io\u002Fspecification\u002F2025-11-25\u002Findex",[204],"MCP spec",[263,300,301,304,305,310],{},[193,302,303],{},"Compaction."," Harness-owned context management so the window does not become the only memory. Anthropic’s ",[200,306,309],{"href":307,"rel":308},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-harnesses-for-long-running-agents",[204],"long-running harness"," offloads state to files and git.",[263,312,313,316,317,230],{},[193,314,315],{},"Routing."," Model class per step, not a user-picked mascot. ",[200,318,320],{"href":319},"model-routing-architecture","Model routing architecture",[190,322,323,324,327,328,331,332,334,335,337,338,341,342,344,345,348],{},"Nimbus maps this architecture onto product objects rather than asking operators to draw LangGraph: ",[200,325,326],{"href":28},"wiki"," (guides), ",[200,329,330],{"href":51},"integrations"," (tool plane), ",[200,333,221],{"href":20}," (orchestration contract), ",[200,336,216],{"href":32}," (workspace), ",[200,339,340],{"href":40},"governance"," (write plane), ",[200,343,23],{"href":24}," (eval and memory), ",[200,346,347],{"href":45},"models"," (routing). Other vendors map the same boxes differently. Score the boxes.",[255,350,352],{"id":351},"why-architecture-not-a-bigger-prompt","Why architecture (not a bigger prompt)",[190,354,355,356,360,361,366,367,372,373,378],{},"A prompt cannot own tool execution, identity, or a stop that survives a tired model. ",[200,357,359],{"href":358},"what-is-harness-engineering","Harness engineering"," is the practice; this page is the structure the practice edits. ",[200,362,365],{"href":363,"rel":364},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework",[204],"NIST AI RMF"," Govern\u002FMap need a system you can point to. ",[200,368,371],{"href":369,"rel":370},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F42001",[204],"ISO 42001"," needs operational controls. ",[200,374,377],{"href":375,"rel":376},"https:\u002F\u002Fgenai.owasp.org\u002Fllm-top-10\u002F",[204],"OWASP LLM Top 10"," excessive agency is what happens when the tool plane has no architecture.",[190,380,381,386],{},[200,382,385],{"href":383,"rel":384},"https:\u002F\u002Fwww.mckinsey.com\u002Fcapabilities\u002Fquantumblack\u002Four-insights\u002Fthe-state-of-ai",[204],"McKinsey 2025"," treats agentic value as organisational. Architecture is how you stop “every team’s unofficial loop” from becoming the estate.",[190,388,389],{},"It affects you if you are combining MCP servers, a coding agent, a copilot, and a CRM writer without a single grant and quote rule. Two writers to one object is an architecture bug, not a training issue.",[255,391,393],{"id":392},"the-pieces","The pieces",[190,395,396],{},"Keep these as inspectable contracts.",[190,398,399,402,403,408,409,414],{},[193,400,401],{},"1. Loop runtime."," Plan → act → observe, with max steps and a cost budget the model cannot waive. Frameworks (",[200,404,407],{"href":405,"rel":406},"https:\u002F\u002Fdocs.langchain.com\u002Foss\u002Fpython\u002Flangchain\u002Fagents",[204],"create_agent",", LangGraph, CrewAI) implement this in process. Product harnesses implement it as a hosted run. ",[200,410,413],{"href":411,"rel":412},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents",[204],"Anthropic’s effective agents"," is still the best short note on bounding the loop. “The model says it is done” is an input to the runtime, not the runtime.",[190,416,417,420,421,425],{},[193,418,419],{},"2. Workspace and filesystem."," Inner harnesses treat the directory as externalised memory — Manus-style and Anthropic-style artefacts. Outer harnesses treat the workstream as the directory analogue: artefacts on a canvas, not a hidden ",[422,423,424],"code",{},"\u002Ftmp"," on a laptop. Do not store approved discounts only in a coding agent’s memory file.",[190,427,428,431,432,435,436,440,441,445],{},[193,429,430],{},"3. Context assembly."," System prompt, skills, ",[422,433,434],{},"AGENTS.md"," \u002F wiki slices, retrieved records, prior graph nodes. Guides in Böckeler’s sense. Compaction and retrieval belong here. ",[200,437,439],{"href":438},"what-is-enterprise-rag","Enterprise RAG"," is a pattern inside this box, not the architecture. ",[200,442,444],{"href":443},"what-is-a-company-wiki-for-ai-agents","Company wiki"," is asserted policy; do not collapse it into a private vector bucket per agent.",[190,447,448,451],{},[193,449,450],{},"4. Tool dispatch."," Host executes; model proposes. Sandbox for shell. Adapters for SaaS. Timeouts, retries, structured errors back into the loop. Generic HTTP with a production token is not this box. It is a confused deputy.",[190,453,454,457,458,463,464,467,468,471,472,475],{},[193,455,456],{},"5. Hooks \u002F middleware."," Deterministic intercepts: ",[200,459,462],{"href":460,"rel":461},"https:\u002F\u002Fcode.claude.com\u002Fdocs\u002Fen\u002Fhooks",[204],"Claude Code"," ",[422,465,466],{},"PreToolUse"," \u002F ",[422,469,470],{},"PostToolUse","; LangChain middleware; outer interceptor that never exposes the write API unsigned. ",[200,473,474],{"href":228},"Write-back governance",". Advice in markdown does not live in this box.",[190,477,478,481,482,230],{},[193,479,480],{},"6. Permissions and identity."," Who the harness authenticates as, per tool, per object, per job. Roster and workstream membership on the outer side. Repo and sandbox roles on the inner side. Teams declare required connectors; the workspace still grants. ",[200,483,484],{"href":220},"Agent team architecture",[190,486,487,490,491,230],{},[193,488,489],{},"7. Orchestration."," Subagents, specialist hand-offs, stop on gate. Optional until duties already split. Orchestrator in the product, not a manager persona with every login. ",[200,492,494],{"href":493},"what-is-multi-agent-ai","What is multi-agent AI",[190,496,497,500,501,505],{},[193,498,499],{},"8. Sensors and eval."," Compiler, tests, schema, quote-hash, SoR read-back, human review. Independent of the generator. ",[200,502,504],{"href":503},"eval-loops-for-enterprise-agent-harnesses","Eval loops",". SWE-bench \u002F Terminal-Bench measure inner coding harnesses; they do not close this box for GL posts.",[190,507,508,511,512,230],{},[193,509,510],{},"9. Durable memory of operations."," Files and git (inner). Wiki + Lifecycle Graph (outer). Session transcripts are a debug aid. They are not the ledger. ",[200,513,515],{"href":514},"causal-memory-architecture-for-enterprise-ai","Causal memory",[190,517,518,521,522,526],{},[193,519,520],{},"10. Routing and spend."," Step classes → model classes. Caps on the run. ",[200,523,525],{"href":524},"ai-cost-control-architecture","AI cost control",". Seat-unlimited flagship is an architectural choice (always-frontier), not a missing feature.",[190,528,529,532],{},[193,530,531],{},"Flow (outer)."," Brief on a workstream → satisfy connector contract → plan → retrieve (logged, scoped) → draft on canvas → quote if write in scope → gate → execute signed payload only → commit graph. If steps 5–7 live only in a prompt, jailbreaks and tired operators fall through the same hole.",[190,534,535,538],{},[193,536,537],{},"Flow (inner)."," Session start loads guides → loop with shell\u002Feditor tools → hooks on tool events → testsensor → commit \u002F PR → CI as outer-loop sensor in Osmani’s sense. Anthropic’s initializer vs coding agent is a two-role inner architecture for work that outlasts one window.",[190,540,541],{},"Nimbus’s hosted flow is the outer sequence. Perception and Conflux sit on retrieve\u002Fdraft; they must not skip the quote. That is architecture, not brand.",[255,543,545],{"id":544},"failure-modes-the-diagram-exists-to-prevent","Failure modes the diagram exists to prevent",[547,548,549,555,561,567,573,579,585],"ol",{},[263,550,551,554],{},[193,552,553],{},"Orchestrator-in-the-model."," Jailbreak equals admin.",[263,556,557,560],{},[193,558,559],{},"Shared toolbox."," Every specialist has every write.",[263,562,563,566],{},[193,564,565],{},"Context as only memory."," Compaction deletes the approval.",[263,568,569,572],{},[193,570,571],{},"MCP as control plane."," Plug without grants.",[263,574,575,578],{},[193,576,577],{},"Eval = transcript."," The model graded itself.",[263,580,581,584],{},[193,582,583],{},"Two harnesses, one SoR writer."," IDE MCP and OS both PATCH.",[263,586,587,590,591,230],{},[193,588,589],{},"Framework mistaken for architecture."," Nodes without identity. ",[200,592,594],{"href":593},"agent-harness-vs-agent-framework","Harness vs framework",[190,596,597,602],{},[200,598,601],{"href":599,"rel":600},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2024\u002F1689\u002Foj",[204],"EU AI Act"," oversight needs interrupt and record. Those are boxes 5, 6, and 9.",[255,604,606],{"id":605},"mapping-langchains-anatomy-onto-company-objects","Mapping LangChain’s anatomy onto company objects",[190,608,609,613],{},[200,610,612],{"href":202,"rel":611},[204],"LangChain’s parts list"," is built from coding and general agents. Translate, do not copy:",[615,616,617,633],"table",{},[618,619,620],"thead",{},[621,622,623,627,630],"tr",{},[624,625,626],"th",{},"Anatomy piece",[624,628,629],{},"Inner binding",[624,631,632],{},"Outer binding",[634,635,636,651,662,673,687,698,712],"tbody",{},[621,637,638,642,648],{},[639,640,641],"td",{},"System prompts \u002F skills",[639,643,644,647],{},[422,645,646],{},"CLAUDE.md",", skills",[639,649,650],{},"Wiki playbooks, versioned with the run",[621,652,653,656,659],{},[639,654,655],{},"Tools + MCP",[639,657,658],{},"Shell, apply_patch, browser",[639,660,661],{},"Connectors; MCP behind the same grant",[621,663,664,667,670],{},[639,665,666],{},"Filesystem \u002F sandbox",[639,668,669],{},"Checkout, container",[639,671,672],{},"Workstream canvas + isolated grants",[621,674,675,678,681],{},[639,676,677],{},"Orchestration",[639,679,680],{},"Subagents in the IDE",[639,682,683,686],{},[200,684,685],{"href":220},"Agent teams"," on a roster",[621,688,689,692,695],{},[639,690,691],{},"Hooks \u002F middleware",[639,693,694],{},"PreToolUse, lint",[639,696,697],{},"Write interceptor, spend cap",[621,699,700,703,706],{},[639,701,702],{},"Memory",[639,704,705],{},"Files, git, memory md",[639,707,708,709],{},"Wiki + ",[200,710,23],{"href":711},"what-is-a-lifecycle-graph",[621,713,714,717,720],{},[639,715,716],{},"Eval",[639,718,719],{},"Tests, Terminal-Bench",[639,721,722],{},"Quote hash, SoR read-back",[190,724,725],{},"If a vendor cannot fill the outer column, they are an inner (or framework) product. That is allowed. Do not invent the column in a slide.",[190,727,728,731,732,735,736,230],{},[193,729,730],{},"Control plane independence."," Whatever sits in the Orchestration row must know grants, budget, and gates ",[270,733,734],{},"without"," asking the model. LangGraph can do that if the nodes are code. A “manager agent” with every tool cannot. Nimbus’s orchestrator is product-hosted for that reason; you should still ask it to refuse when NetSuite is missing. ",[200,737,94],{"href":251},[190,739,740,743],{},[193,741,742],{},"Thoughtworks’ four combinations"," (deterministic\u002Fprobabilistic × feed-forward\u002Ffeedback) overlay this table. Whitelists and spend ceilings are box 5\u002F6 deterministic feed-forward. Schema validation is box 8 deterministic feedback. Wiki retrieval is probabilistic feed-forward. LLM critic is probabilistic feedback — never the only item in box 8 for a GL post.",[190,745,746,749,750,230],{},[193,747,748],{},"Two harnesses, one SoR rule."," Draw both columns on one whiteboard. Draw one write plane. If two arrows reach Salesforce, you have an architecture incident waiting. ",[200,751,287],{"href":286},[190,753,754,755,230],{},"Version the diagram when you add a tool. A new MCP server is a change to boxes 4 and 6, not a chat plugin. ",[200,756,758],{"href":757},"what-is-model-context-protocol","MCP",[190,760,761,762,766],{},"Implementation order for a company that has none of this: (1) split write plane from read plane — even if the “harness” is still a single agent; (2) pin policy version on the run; (3) add one deterministic sensor on the artefact you cannot get wrong; (4) host the orchestrator’s grants outside the prompt; (5) only then add specialists. Reversing that order is how shared-toolbox packs ship. ",[200,763,765],{"href":411,"rel":764},[204],"Anthropic"," starts with bounding tools and defining done for a reason.",[190,768,769,770,217,772,217,774,217,776,779,780,783],{},"Framework teams should draw the ten boxes on the README of the graph repo and tick which are code, which are still prompts, which are missing. Product teams should map each box to a screen an operator can see. If box 8 is “the model reflects,” you do not have eval architecture. If box 6 is “the service account,” you do not have identity architecture. Nimbus’s screens are ",[200,771,216],{"href":32},[200,773,340],{"href":40},[200,775,326],{"href":28},[200,777,778],{"href":24},"graph"," — use them as a checklist, not as proof that the boxes exist in ",[270,781,782],{},"your"," configuration.",[190,785,786,789],{},[200,787,211],{"href":209,"rel":788},[204]," calls the model the brain and the harness the body. Architecture is the anatomy of that body so Security can review it. If the diagram is only “LLM in the middle, tools around it,” you have a marketing poster. Add identity, the write split, the sensor that does not trust the brain, and the ledger. Then the poster is a design.",[255,791,793],{"id":792},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[190,795,796,797,799,800,803],{},"The product is a particular binding of the ten boxes for operators: hosted loop, workstream workspace, wiki context, connector dispatch, governance hooks, team orchestration, graph memory, NTU routing. ",[200,798,11],{"href":12},". Inspect each box in a PoV the way you would inspect Claude Code’s hooks and sandbox for an inner buy. ",[200,801,802],{"href":251},"How to evaluate",". AIP and Agentforce bind the same boxes to Ontology or CRM; the architecture still applies.",[255,805,807],{"id":806},"questions-people-actually-ask","Questions people actually ask",[809,810,812],"h3",{"id":811},"do-we-need-all-ten-boxes-on-day-one","Do we need all ten boxes on day one?",[190,814,815],{},"You need loop, tools, a stop, and a sensor for the job you are running. Add orchestration when duties split. Add graph when people leave. Do not add every MCP server first.",[809,817,819],{"id":818},"is-this-the-same-as-an-enterprise-ai-os-architecture","Is this the same as an enterprise AI OS architecture?",[190,821,822,826],{},[200,823,825],{"href":824},"enterprise-ai-operating-system-architecture","OS architecture"," is the product category (collaboration, gates, ledger, routing). Harness architecture is the runtime idea that also covers Claude Code. Overlap on the outer side is expected.",[809,828,830],{"id":829},"where-do-skills-fit","Where do skills fit?",[190,832,833,834,230],{},"Reusable procedures in the context box. Not a substitute for hooks. ",[200,835,838],{"href":836,"rel":837},"https:\u002F\u002Fclaude.com\u002Fblog\u002Fsteering-claude-code-skills-hooks-rules-subagents-and-more",[204],"Anthropic on steering",[809,840,842],{"id":841},"can-langgraph-implement-this","Can LangGraph implement this?",[190,844,845,846,230],{},"Yes. You will implement boxes 5, 6, and 9 yourself for enterprise writes. That is ",[200,847,849],{"href":848},"build-vs-buy-an-enterprise-ai-os","build vs buy",[809,851,853],{"id":852},"what-should-i-read-next","What should I read next?",[190,855,856,858,859,858,862,230],{},[200,857,504],{"href":503},". ",[200,860,861],{"href":358},"What is harness engineering",[200,863,865],{"href":864},"what-is-an-enterprise-agent-harness","What is an enterprise agent harness",[255,867,869],{"id":868},"related-reading","Related reading",[190,871,872,876,877,230],{},[200,873,875],{"href":874},"workstream-architecture","Workstream architecture"," and ",[200,878,879],{"href":224},"Connector and permissions architecture",[255,881,883],{"id":882},"sources","Sources",[260,885,886,892,898,905,911,918,924,930,936,942,949,956,961,967,972,978,984],{},[263,887,888],{},[200,889,891],{"href":202,"rel":890},[204],"LangChain, The anatomy of an agent harness",[263,893,894],{},[200,895,897],{"href":405,"rel":896},[204],"LangChain, Agents",[263,899,900],{},[200,901,904],{"href":902,"rel":903},"https:\u002F\u002Fwww.langchain.com\u002Fblog\u002Fhow-to-build-a-custom-agent-harness",[204],"LangChain, How to build a custom agent harness",[263,906,907],{},[200,908,910],{"href":209,"rel":909},[204],"Databricks, What is an AI agent harness?",[263,912,913],{},[200,914,917],{"href":915,"rel":916},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FAgent_harness",[204],"Wikipedia, Agent harness",[263,919,920],{},[200,921,923],{"href":411,"rel":922},[204],"Anthropic, Building effective agents",[263,925,926],{},[200,927,929],{"href":307,"rel":928},[204],"Anthropic, Effective harnesses for long-running agents",[263,931,932],{},[200,933,935],{"href":836,"rel":934},[204],"Anthropic, Steering Claude Code",[263,937,938],{},[200,939,941],{"href":460,"rel":940},[204],"Claude Code, Hooks",[263,943,944],{},[200,945,948],{"href":946,"rel":947},"https:\u002F\u002Fmartinfowler.com\u002Farticles\u002Fharness-engineering.html",[204],"Böckeler, Harness engineering for coding agent users",[263,950,951],{},[200,952,955],{"href":953,"rel":954},"https:\u002F\u002Faddyosmani.com\u002Fblog\u002Fagent-harness-engineering\u002F",[204],"Addy Osmani, Agent harness engineering",[263,957,958],{},[200,959,365],{"href":363,"rel":960},[204],[263,962,963],{},[200,964,966],{"href":369,"rel":965},[204],"ISO\u002FIEC 42001",[263,968,969],{},[200,970,601],{"href":599,"rel":971},[204],[263,973,974],{},[200,975,977],{"href":375,"rel":976},[204],"OWASP Top 10 for LLM applications",[263,979,980],{},[200,981,983],{"href":383,"rel":982},[204],"McKinsey, The state of AI in 2025",[263,985,986],{},[200,987,989],{"href":296,"rel":988},[204],"Model Context Protocol specification",{"title":171,"searchDepth":172,"depth":172,"links":991},[992,993,994,995,996,997,998,1006,1007],{"id":257,"depth":172,"text":258},{"id":351,"depth":172,"text":352},{"id":392,"depth":172,"text":393},{"id":544,"depth":172,"text":545},{"id":605,"depth":172,"text":606},{"id":792,"depth":172,"text":793},{"id":806,"depth":172,"text":807,"children":999},[1000,1002,1003,1004,1005],{"id":811,"depth":1001,"text":812},3,{"id":818,"depth":1001,"text":819},{"id":829,"depth":1001,"text":830},{"id":841,"depth":1001,"text":842},{"id":852,"depth":1001,"text":853},{"id":868,"depth":172,"text":869},{"id":882,"depth":172,"text":883},"2026-08-24","Agent harness architecture is the runtime around a model: loop, tools, context, hooks, permissions, and eval — mapped, for company jobs, onto workstreams, agent teams, connectors, and write gates.","\u002Fblog\u002Fagent-harness-architecture",{"title":181,"description":1009},"architecture","blog\u002Fagent-harness-architecture",[1012,1015,1016,340],"agent-harness","orchestration","XdfJCVoG0qzYyl-w8rY36RiV_eYemVYaVeWUfQRTJCA",{"hero":1019,"id":1021,"title":1022,"archived":165,"authors":166,"badge":166,"body":1023,"date":166,"definedTerm":166,"department":166,"description":1027,"extension":174,"eyebrow":1028,"faqHeader":166,"faqs":166,"footerBand":1029,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":60,"relatedHeading":1035,"seo":1036,"series":166,"sitemap":131,"status":166,"stem":1037,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1038},{"filename":1020},"u2221455217_Flat_design_of_a_futuristic_minimalist_landscape__5d589295-cdea-4ea9-a262-be766881accf_1.png","content\u002Fblog\u002Findex.md","Exploring the future of intelligence.",{"type":168,"value":1024,"toc":1025},[],{"title":171,"searchDepth":172,"depth":172,"links":1026},[],"Deep dives into pre-cognitive intelligence, sentient enterprises, and the evolving landscape of AI-driven business transformation.","Latest Research",{"headline":1030,"description":1031,"primaryLabel":1032,"primaryTo":1033,"secondaryLabel":1034,"secondaryTo":12},"Stay at the frontier.","Subscribe for product updates and new insights.","Subscribe","\u002Fnewsletter","Explore the platform","More research",{"title":1022,"description":1027},"blog\u002Findex","BFSWGYO9bcTlaulivKYWyg08_DJHsdGg3OC6g_CG1Hw",[1040,1707],{"id":1041,"title":1042,"archived":165,"authors":1043,"badge":1045,"body":1046,"date":1008,"definedTerm":166,"department":166,"description":1700,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1701,"relatedHeading":166,"seo":1702,"series":1012,"sitemap":131,"status":166,"stem":1703,"subhead":166,"tags":1704,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1706},"content\u002Fblog\u002Feval-loops-for-enterprise-agent-harnesses.md","Eval Loops for Enterprise Agent Harnesses",[1044],{"name":184,"to":136},{"label":186},{"type":168,"value":1047,"toc":1682},[1048,1055,1083,1094,1096,1154,1162,1166,1176,1179,1215,1232,1242,1246,1257,1266,1275,1280,1307,1317,1323,1333,1342,1346,1349,1385,1393,1405,1411,1415,1418,1424,1433,1449,1455,1461,1472,1485,1490,1493,1508,1518,1520,1526,1528,1532,1535,1539,1542,1546,1552,1556,1559,1563,1570,1572,1583,1585,1593,1595],[190,1049,1050,1051,1054],{},"An ",[193,1052,1053],{},"eval loop"," for an agent harness is an independent check that the job is actually done — tests, schemas, read-backs, humans — that does not take the model’s word.",[190,1056,1057,1058,1063,1064,1069,1070,1073,1074,1077,1078,1082],{},"Coding harnesses already have a public language for this. ",[200,1059,1062],{"href":1060,"rel":1061},"https:\u002F\u002Fwww.swebench.com\u002F",[204],"SWE-bench"," gives an agent a GitHub issue and grades a patch with the repo’s tests. ",[200,1065,1068],{"href":1066,"rel":1067},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.11868",[204],"Terminal-Bench"," (Stanford \u002F Laude Institute) gives an agent a machine and grades the ",[270,1071,1072],{},"end state"," of a container, not the transcript. Leaderboards even report ",[193,1075,1076],{},"agent + model"," as a pair, which is the right unit: ",[200,1079,1081],{"href":405,"rel":1080},[204],"Agent = Model + Harness",". Steal that honesty. Do not steal the benchmark as your control for Salesforce.",[190,1084,1085,1086,1089,1090,1093],{},"Enterprise eval is: did the quoted CRM write match the signed payload, and can you replay who signed. A 40% Terminal-Bench score does not tell you whether Opportunity.Amount was authorised. ",[200,1087,1088],{"href":251},"How to evaluate an agent harness"," is the buying sheet. This page is the architecture of the sensor loop ",[200,1091,1092],{"href":358},"harness engineering"," keeps tightening.",[255,1095,258],{"id":257},[260,1097,1098,1108,1118,1133,1142,1148],{},[263,1099,1100,1103,1104,1107],{},[193,1101,1102],{},"Oracle \u002F verifier."," The independent test. SWE-bench: ",[422,1105,1106],{},"FAIL_TO_PASS"," tests. Terminal-Bench: pytest-style assertions on container state. Enterprise: SoR read-back and payload hash.",[263,1109,1110,1113,1114,1117],{},[193,1111,1112],{},"Transcript eval."," Grading the chain-of-thought. Useful for debugging. Insufficient as a release gate. Models claim victory; Anthropic’s ",[200,1115,309],{"href":307,"rel":1116},[204]," names premature victory as a failure mode.",[263,1119,1120,463,1123,467,1127,1132],{},[193,1121,1122],{},"Computational vs inferential sensors.",[200,1124,1126],{"href":946,"rel":1125},[204],"Böckeler",[200,1128,1131],{"href":1129,"rel":1130},"https:\u002F\u002Fwww.thoughtworks.com\u002Fen-us\u002Finsights\u002Fblog\u002Fgenerative-ai\u002Fharness-engineering-agent-feedback-exploring-ai-coding-sensors",[204],"Thoughtworks",". Compiler vs LLM-as-judge. Prefer computational for invariants (schema, identity, hash). Use inferential for taste (narrative quality), never as the only SoR gate.",[263,1134,1135,1138,1139,230],{},[193,1136,1137],{},"LLM-as-judge."," Another stochastic component. Fine as a critic specialist. Not the signer. ",[200,1140,1141],{"href":241},"HITL architecture",[263,1143,1144,1147],{},[193,1145,1146],{},"Offline vs online eval."," Offline: golden jobs, replay. Online: shadow reads, canary writes, production sensors. You need both; most teams only have a demo recording.",[263,1149,1150,1153],{},[193,1151,1152],{},"Harness eval vs model eval."," Changing Claude vs GPT on the same tools is model eval. Changing hooks, grants, or wiki and keeping the model is harness eval. Report them separately or you will buy a new model for a missing schema check.",[190,1155,1156,1157,876,1159,1161],{},"Nimbus’s production sensor for writes is the quote-and-gate plus graph: ",[200,1158,340],{"href":40},[200,1160,23],{"href":24},". That is computational. Wiki playbooks are guides. Do not confuse a fluent Conflux draft with a passed eval.",[255,1163,1165],{"id":1164},"why-coding-benchmarks-are-the-wrong-outer-score","Why coding benchmarks are the wrong outer score",[190,1167,1168,1169,1172,1173,1175],{},"They are the ",[270,1170,1171],{},"right"," inner score. ",[200,1174,287],{"href":286},". Terminal-Bench’s design is even a lesson: grade the environment, not the story. The environment for RevOps is Salesforce, not a Docker VM with a hidden oracle.",[190,1177,1178],{},"Problems when you import SWE-bench into an enterprise RFP:",[260,1180,1181,1187,1193,1199,1209],{},[263,1182,1183,1186],{},[193,1184,1185],{},"Wrong workspace."," Patch quality ≠ payload authorisation.",[263,1188,1189,1192],{},[193,1190,1191],{},"Saturation and leakage."," Public coding benches get gamed; your CRM schema is not a public task.",[263,1194,1195,1198],{},[193,1196,1197],{},"No identity."," Benchmarks do not have a Finance signer.",[263,1200,1201,1204,1205,1208],{},[193,1202,1203],{},"No replay duty."," A leaderboard row is not ",[200,1206,371],{"href":369,"rel":1207},[204]," evidence.",[263,1210,1211,1214],{},[193,1212,1213],{},"Wrong “done.”"," Tests pass on a fixture; production field still wrong.",[190,1216,1217,1220,1221,1224,1225,1227,1228,1231],{},[200,1218,385],{"href":383,"rel":1219},[204]," is about scaling work, not about bash tasks. ",[200,1222,365],{"href":363,"rel":1223},[204]," Measure is: did the control work in ",[270,1226,782],{}," context of use. ",[200,1229,601],{"href":599,"rel":1230},[204]," wants interrupt and record, not a percentile on Terminal-Bench 2.1.",[190,1233,1234,1235,858,1238,230],{},"Use coding benches to pick an inner harness for engineering. Use quote\u002Freplay to pick an ",[200,1236,1237],{"href":864},"enterprise harness",[200,1239,1241],{"href":1240},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose",[255,1243,1245],{"id":1244},"what-an-enterprise-eval-loop-actually-runs","What an enterprise eval loop actually runs",[190,1247,1248,1249,1252,1253,1256],{},"Design it like Terminal-Bench in spirit: ",[193,1250,1251],{},"end state of the systems that matter",", plus ",[193,1254,1255],{},"process constraints"," the company cannot waive.",[190,1258,1259,1262,1263,1265],{},[193,1260,1261],{},"Precondition sensors (feed-forward that is checkable)."," Required connectors attached. Roster includes the signer role. Wiki revision pinned. Spend quote accepted. If any fail, the run does not start. That is a harness eval of configuration, not of eloquence. ",[200,1264,685],{"href":220}," declaring required systems belong here.",[190,1267,1268,1271,1272,230],{},[193,1269,1270],{},"Step sensors."," Retrieval logged and in-scope (no confused-deputy dump). Tool errors do not silently retry a write. Routing used compact on extract if that is policy. ",[200,1273,1274],{"href":224},"Connector architecture",[190,1276,1277],{},[193,1278,1279],{},"Release sensors (the outer oracle).",[547,1281,1282,1285,1295,1298,1301],{},[263,1283,1284],{},"Quote is structured: object, fields, values, cardinality, hash.",[263,1286,1287,1288,1291,1292,1294],{},"Named human with the right role signed ",[270,1289,1290],{},"that"," hash (",[200,1293,229],{"href":228},").",[263,1296,1297],{},"Adapter executed only that payload.",[263,1299,1300],{},"SoR read-back equals quote (or a documented, signed delta).",[263,1302,1303,1304,230],{},"Graph (or equivalent ledger) contains brief, team, policy version, signer, payload, result. Export works without the vendor. ",[200,1305,1306],{"href":711},"Lifecycle graph",[190,1308,1309,1312,1313,230],{},[193,1310,1311],{},"Negative tests."," Reject path: SoR unchanged. Detached grant: write impossible. Wrong role: Hard\u002FCritical cannot complete. These are the analogue of tests that must stay red. If your PoV never fails, you did not eval the harness. You evaluated a happy path. ",[200,1314,1316],{"href":1315},"how-to-run-an-enterprise-ai-proof-of-value","Proof of value",[190,1318,1319,1322],{},[193,1320,1321],{},"Inferential sensors (optional, never sole)."," A legal specialist flags language. A critic agent scores a narrative. Useful. If they can waive a Hard gate, you added a second stochastic writer.",[190,1324,1325,463,1328,1332],{},[193,1326,1327],{},"Human as sensor, not as folklore.",[200,1329,1331],{"href":1330},"what-is-human-in-the-loop-ai","HITL"," is a step with identity. A Slack emoji is transport. A six-month zero-reject rate is a finding: either perfect or unread.",[190,1334,1335,1336,1341],{},"Air Canada and the ",[200,1337,1340],{"href":1338,"rel":1339},"https:\u002F\u002Fwww.reuters.com\u002Flegal\u002Fnew-york-lawyers-sanctioned-using-fake-chatgpt-cases-legal-brief-2023-06-22\u002F",[204],"ChatGPT brief sanctions"," are eval-loop absences: no independent check before a system of record (policy page, court docket) changed.",[255,1343,1345],{"id":1344},"offline-suites-you-can-actually-keep","Offline suites you can actually keep",[190,1347,1348],{},"You will not publish a public “CRM-bench.” You can keep a private suite:",[260,1350,1351,1357,1363,1369,1379],{},[263,1352,1353,1356],{},[193,1354,1355],{},"Golden jobs."," Anonymised or sandbox SoR. Expected quote. Expected refuse.",[263,1358,1359,1362],{},[193,1360,1361],{},"Replay."," Last month’s signed write: same hash, same graph nodes.",[263,1364,1365,1368],{},[193,1366,1367],{},"Policy diffs."," Change wiki cap; next run must quote the new cap or refuse.",[263,1370,1371,1374,1375,230],{},[193,1372,1373],{},"Model swap."," Same harness, new weights: tools still dispatch; sensors still fire. That isolates model eval. ",[200,1376,1378],{"href":1377},"what-is-model-routing","Model routing",[263,1380,1381,1384],{},[193,1382,1383],{},"Chaos."," Kill the interceptor; writes must not fail open.",[190,1386,1387,1388,1392],{},"Version the suite with the harness. ",[200,1389,1391],{"href":1390},"what-is-an-agentic-workflow","What is an agentic workflow",": the definition that ran is an input. A golden job that still “passes” after you removed the Hard gate is a broken eval, not a better model.",[190,1394,1395,1396,1399,1400,1404],{},"LangSmith, Phoenix, and similar tracing tools help ",[270,1397,1398],{},"observe"," inner and framework loops. They are not the SoR oracle. ",[200,1401,1403],{"href":1402},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",". Tracing without a hash match is a nicer transcript.",[190,1406,1407,1408,1410],{},"Nimbus should be scored on whether you can automate those golden jobs on a sandbox org: attach, refuse, sign, read-back, export. ",[200,1409,31],{"href":32}," are the fixture runner. If we cannot show a red refuse, we fail this architecture too.",[255,1412,1414],{"id":1413},"building-a-private-suite-without-a-public-crm-bench","Building a private suite without a public CRM-bench",[190,1416,1417],{},"You do not need 2,294 GitHub issues. You need a dozen jobs that hurt when they are wrong.",[190,1419,1420,1423],{},[193,1421,1422],{},"Pick three families."," (1) A write that must refuse (wrong role, missing field, detached grant). (2) A write that must match a fixture after sign-off. (3) A read-only job that must not call a write tool at all. Encode each as a workstream template or a scripted PoV. Run weekly. When a wiki cap changes, family (2) must fail until the quote updates — that is harness eval, not flaky CI.",[190,1425,1426,1429,1430,230],{},[193,1427,1428],{},"Grade environment state."," Terminal-Bench does not score the agent’s diary. Copy that. After the run, query the sandbox SoR. Compare to the signed hash. If you only grade the canvas prose, you are back to transcript eval. ",[200,1431,1432],{"href":228},"Write-back",[190,1434,1435,1438,1439,1442,1443,1446,1447,230],{},[193,1436,1437],{},"Keep model and harness scores apart."," Swap GPT vs Claude on the same golden job: if sensors still fire and hashes still match, the harness held. If a new model skips a field and the schema sensor catches it, that is a ",[270,1440,1441],{},"pass"," for the harness and a ",[270,1444,1445],{},"note"," for the model. If the sensor does not catch it, you do not need a larger model. You need a sensor. ",[200,1448,359],{"href":358},[190,1450,1451,1454],{},[193,1452,1453],{},"Report agent + model."," SWE-bench leaderboards already do this. Your internal dashboard should too: “Nimbus + routed compact\u002Ffrontier” or “LangGraph + GPT + our interceptor.” Hiding the harness is how you buy a new model for a missing hook.",[190,1456,1457,1460],{},[193,1458,1459],{},"Budget the eval itself."," Inferential judges on every step will cost more than the job. Thoughtworks’ advice: deterministic checks on every transaction; probabilistic judges on critical paths. Schema and identity are every-transaction. Narrative quality is not.",[190,1462,1463,1466,1467,1471],{},[193,1464,1465],{},"What you can cite externally."," You can say you run refuse tests and read-backs. You cannot honestly say “we scored 83% on Terminal-Bench therefore Finance is safe.” ",[200,1468,1470],{"href":1066,"rel":1469},[204],"Stanford \u002F Laude’s paper"," is a CLI benchmark. Use it for CLI harnesses.",[190,1473,1474,1475,1480,1481,230],{},"Air Canada needed a sensor on “did we emit a policy commitment.” The court docket needed a sensor on “do these citations exist.” Your suite is that instinct with fixtures. ",[200,1476,1479],{"href":1477,"rel":1478},"https:\u002F\u002Fwww.cbc.ca\u002Fnews\u002Fcanada\u002Fbritish-columbia\u002Fair-canada-chatbot-lawsuit-1.7116416",[204],"CBC","; ",[200,1482,1484],{"href":1338,"rel":1483},[204],"Reuters",[190,1486,1487,1488,230],{},"Online eval is the part teams skip. Offline goldens rot when the wiki moves. Shadow mode — agent quotes, human still writes, compare payloads — is an eval loop that does not need production write permission. Canary — one workstream, one object type, Hard gate, weekly refuse report — is how you learn whether operators rubber-stamp. A six-month zero-reject chart is not a quality medal. It is a sensor that may be dead. ",[200,1489,1331],{"href":1330},[190,1491,1492],{},"Compare this to CI for software. You would not ship because the developer said the tests passed on their laptop. You would not replace CI with an LLM that reads the diff and scores “looks good.” You might add that LLM as a critic. Enterprise write eval is CI for mutations. Nimbus’s gate is the required check; your SoR read-back is the assertion file. If we only store the transcript, we are the laptop. Demand the assertion.",[190,1494,1495,1500,1501,1504,1505,230],{},[200,1496,1499],{"href":1497,"rel":1498},"https:\u002F\u002Fdocs.langchain.com\u002Flangsmith\u002Fobservability",[204],"LangSmith"," and similar are the right place to ",[270,1502,1503],{},"debug"," traces for framework and inner loops. Export those traces into your golden runner; do not let the tracing UI become the only evidence for audit. Auditors will ask for the hash and the signer. ",[200,1506,1507],{"href":1402},"How to evaluate AI audit",[190,1509,1510,1511,1513,1514,1517],{},"Do not wait for a consortium bench. Your suite is a competitive advantage if it encodes ",[270,1512,782],{}," caps and objects. Share the ",[270,1515,1516],{},"method"," (refuse, read-back, replay) in the RFP. Keep the fixtures. Vendors who cannot run against your sandbox are not ready for your SoR, however they score on Terminal-Bench.",[255,1519,793],{"id":792},[190,1521,1522,1523,1525],{},"The product’s eval spine is: NTU quote before the run, scoped retrieval, canvas artefacts, write quotes, tiered gates, graph commit. Sensors you should still add: your own SoR read-back in the sandbox, your own golden files (the analogue of pytest). The platform cannot know your “correct Amount” without your oracle. Terminal-Bench ships oracles per task. You must ship oracles per job. That is ",[200,1524,1092],{"href":358},", not a missing model.",[255,1527,807],{"id":806},[809,1529,1531],{"id":1530},"can-we-use-an-llm-as-judge-on-the-quote","Can we use an LLM-as-judge on the quote?",[190,1533,1534],{},"As a critic, yes. As the only signer, no. Computational match of fields is cheap and stable.",[809,1536,1538],{"id":1537},"do-we-wait-for-an-industry-enterprise-swe-bench","Do we wait for an industry “enterprise SWE-bench”?",[190,1540,1541],{},"You would still need private oracles. Start this quarter with sandbox read-backs.",[809,1543,1545],{"id":1544},"our-vendor-only-shares-swe-bench","Our vendor only shares SWE-bench.",[190,1547,1548,1549,230],{},"File as inner evidence. Demand refuse\u002Freplay for outer. ",[200,1550,1551],{"href":251},"Evaluate the harness",[809,1553,1555],{"id":1554},"is-tracing-enough-for-iso-42001","Is tracing enough for ISO 42001?",[190,1557,1558],{},"Traces help Measure. You still need Manage: a control that fired. A pretty trace of an unsigned write is a better incident report.",[809,1560,1562],{"id":1561},"how-does-this-relate-to-agent-teams-vs-single-agents","How does this relate to agent teams vs single agents?",[190,1564,1565,1566,230],{},"Teams add hand-off evals (typed artefacts). They do not replace the write oracle. ",[200,1567,1569],{"href":1568},"how-to-evaluate-agent-teams-vs-single-agents","How to evaluate agent teams",[809,1571,853],{"id":852},[190,1573,1574,858,1577,858,1581,230],{},[200,1575,195],{"href":1576},"agent-harness-architecture",[200,1578,1580],{"href":1579},"how-to-evaluate-write-back-governance","How to evaluate write-back governance",[200,1582,861],{"href":358},[255,1584,869],{"id":868},[190,1586,1587,876,1589,230],{},[200,1588,1403],{"href":1402},[200,1590,1592],{"href":1591},"write-back-governance-for-systems-of-record","Write-back governance for systems of record",[255,1594,883],{"id":882},[260,1596,1597,1602,1608,1613,1618,1623,1628,1633,1639,1644,1649,1654,1659,1665,1671,1677],{},[263,1598,1599],{},[200,1600,1062],{"href":1060,"rel":1601},[204],[263,1603,1604],{},[200,1605,1607],{"href":1066,"rel":1606},[204],"Terminal-Bench (arXiv:2601.11868)",[263,1609,1610],{},[200,1611,897],{"href":405,"rel":1612},[204],[263,1614,1615],{},[200,1616,891],{"href":202,"rel":1617},[204],[263,1619,1620],{},[200,1621,929],{"href":307,"rel":1622},[204],[263,1624,1625],{},[200,1626,923],{"href":411,"rel":1627},[204],[263,1629,1630],{},[200,1631,948],{"href":946,"rel":1632},[204],[263,1634,1635],{},[200,1636,1638],{"href":1129,"rel":1637},[204],"Thoughtworks, Harness engineering and agent feedback",[263,1640,1641],{},[200,1642,983],{"href":383,"rel":1643},[204],[263,1645,1646],{},[200,1647,365],{"href":363,"rel":1648},[204],[263,1650,1651],{},[200,1652,966],{"href":369,"rel":1653},[204],[263,1655,1656],{},[200,1657,601],{"href":599,"rel":1658},[204],[263,1660,1661],{},[200,1662,1664],{"href":1477,"rel":1663},[204],"CBC, Air Canada chatbot lawsuit",[263,1666,1667],{},[200,1668,1670],{"href":1338,"rel":1669},[204],"Reuters, ChatGPT legal brief sanctions",[263,1672,1673],{},[200,1674,1676],{"href":1497,"rel":1675},[204],"LangSmith observability",[263,1678,1679],{},[200,1680,989],{"href":296,"rel":1681},[204],{"title":171,"searchDepth":172,"depth":172,"links":1683},[1684,1685,1686,1687,1688,1689,1690,1698,1699],{"id":257,"depth":172,"text":258},{"id":1164,"depth":172,"text":1165},{"id":1244,"depth":172,"text":1245},{"id":1344,"depth":172,"text":1345},{"id":1413,"depth":172,"text":1414},{"id":792,"depth":172,"text":793},{"id":806,"depth":172,"text":807,"children":1691},[1692,1693,1694,1695,1696,1697],{"id":1530,"depth":1001,"text":1531},{"id":1537,"depth":1001,"text":1538},{"id":1544,"depth":1001,"text":1545},{"id":1554,"depth":1001,"text":1555},{"id":1561,"depth":1001,"text":1562},{"id":852,"depth":1001,"text":853},{"id":868,"depth":172,"text":869},{"id":882,"depth":172,"text":883},"Coding agents can be scored on SWE-bench and Terminal-Bench. An enterprise harness is scored on whether the executed write matched the signed payload — independent sensors, not the model’s own claim that it was done.","\u002Fblog\u002Feval-loops-for-enterprise-agent-harnesses",{"title":1042,"description":1700},"blog\u002Feval-loops-for-enterprise-agent-harnesses",[1012,1015,1705,340],"evaluation","CM1xXUF5XL1F03Rrcyw9I74LA1IpOJ_4H95nVjEY3r4",{"id":1708,"title":1709,"archived":165,"authors":1710,"badge":1712,"body":1713,"date":2363,"definedTerm":166,"department":166,"description":2364,"extension":174,"eyebrow":166,"faqHeader":2365,"faqs":2368,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2381,"relatedHeading":166,"seo":2382,"series":1012,"sitemap":131,"status":166,"stem":2383,"subhead":166,"tags":2384,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2386},"content\u002Fblog\u002Fgovernance-as-a-multiplayer-primitive.md","Governance as a Multiplayer Primitive",[1711],{"name":184,"to":136},{"label":186},{"type":168,"value":1714,"toc":2352},[1715,1732,1750,1762,1774,1778,1781,1789,1792,1822,1831,1839,1843,1846,1863,1866,1872,1878,1899,1913,1936,1949,1953,1960,1967,1976,1985,1993,2005,2012,2016,2023,2038,2041,2044,2062,2071,2078,2082,2085,2088,2105,2113,2125,2187,2192,2196,2199,2202,2213,2224,2235,2255,2259,2266,2271,2278,2284,2288,2291,2311,2317,2323,2326,2328,2338,2341,2348],[190,1716,1717,1718,1721,1722,1727,1728,1731],{},"Governance as a multiplayer primitive means the controls live ",[193,1719,1720],{},"in the room where the work happens"," — on the roster, on the payload, on the spend cap for this job — not in a PDF that arrives after someone already changed the record. Gartner’s ",[200,1723,1726],{"href":1724,"rel":1725},"https:\u002F\u002Fwww.gartner.com\u002Fen\u002Farticles\u002Fai-governance-trism",[204],"TRiSM"," framing — trust, risk, and security management — stresses ",[193,1729,1730],{},"runtime"," controls, not slide decks. If your proof of governance is a quarterly attestation while CRM writes still succeed without a named signer, you have principles, not a primitive.",[190,1733,1734,1738,1739,1743,1744,1746,1747,1749],{},[200,1735,1737],{"href":1736},"what-is-ai-governance","What is AI governance"," is the programme: who may use which AI, on which data, with a record afterwards. ",[200,1740,1742],{"href":1741},"rbac-for-enterprise-ai","RBAC for enterprise AI"," is who may see which jobs and tools. ",[200,1745,474],{"href":228}," is the fail-closed rule on live-system changes. This page is the ",[193,1748,1012],{}," that makes those ideas multiplayer: several people and an agent on one job, with authority that cannot exceed the humans in the room.",[190,1751,1752,1753,1757,1758,1761],{},"The ",[200,1754,1756],{"href":363,"rel":1755},[204],"NIST AI Risk Management Framework"," (2023) says Govern, Map, Measure, Manage. Mapping still fails if the people who click can PATCH Salesforce without showing finance the exact fields. ",[200,1759,966],{"href":369,"rel":1760},[204]," wants named actors and operational controls. Neither standard is satisfied by a training video and a hopeful prompt.",[190,1763,1764,1768,1769,1773],{},[200,1765,1767],{"href":1766},"multiplayer-ai-and-multi-agent-ai","Multiplayer AI vs multi-agent AI"," names the room. Governance is what keeps the room from being a demo with extra seats. ",[200,1770,1772],{"href":1771},"what-is-collaborative-ai","What is collaborative AI"," is the shared-job definition. This page is why the stop is visible to everyone on that job — including the partner who joins mid-week.",[255,1775,1777],{"id":1776},"why-governance-belongs-in-the-room","Why governance belongs in the room",[190,1779,1780],{},"Enterprise software already knew maker-checker: one person proposes, another authorises. Generative AI added a proposer that never sleeps and never feels embarrassment. The instinct to “ask legal later” produces the same failure mode as “ask legal in email after the post.”",[190,1782,1783,1784,1788],{},"McKinsey’s ",[200,1785,1787],{"href":383,"rel":1786},[204],"State of AI"," (2025) found that 88% of organisations use AI in at least one function while most remain in pilot. Pilots hide the multiplayer problem. One enthusiast and one assistant do not need a roster. The first time finance, ops, and a partner share an exception, the missing primitive shows up as a write nobody can refuse in time.",[190,1790,1791],{},"Multiplayer governance means four things, together:",[260,1793,1794,1800,1806,1812],{},[263,1795,1796,1799],{},[193,1797,1798],{},"Roster."," Who is on this job, with what role — including which AI role may propose but not sign.",[263,1801,1802,1805],{},[193,1803,1804],{},"Inherited authority."," The agent cannot exceed the person whose credentials it uses. A junior analyst’s session does not inherit the CFO’s write token because the connector was set up as a superuser.",[263,1807,1808,1811],{},[193,1809,1810],{},"Spend as scheduling input."," Token or step budgets are not only finance dashboards; they are “this job pauses until a named delegate raises the cap.”",[263,1813,1814,1817,1818,1821],{},[193,1815,1816],{},"Notify the roster, not the org."," Alerts go to people who can act on ",[193,1819,1820],{},"this"," payload, not a company-wide channel where the signal drowns.",[190,1823,1824,1825,1827,1828,230],{},"That is different from broadcasting “AI policy updated” to ten thousand inboxes. The order hold needs finance on ",[193,1826,1820],{}," job, not ",[422,1829,1830],{},"#general",[190,1832,1752,1833,1838],{},[200,1834,1837],{"href":1835,"rel":1836},"https:\u002F\u002Foecd.ai\u002Fen\u002Fai-principles",[204],"OECD AI Principles"," put accountability and transparency next to robustness. Accountability is empty if the accountable person is not in the room when the field is about to move. Transparency is empty if the payload lives in a private thread. Multiplayer governance is how those principles become clickable.",[255,1840,1842],{"id":1841},"roster-the-multiplayer-unit-of-accountability","Roster: the multiplayer unit of accountability",[190,1844,1845],{},"A roster is not “everyone with a login.” It is the list of people — and bounded AI roles — who may see this brief, these files, and this proposal before it executes.",[190,1847,1848,1849,463,1852,1854,1855,1858,1859,1862],{},"Classic RBAC assigns permissions to roles. Enterprise AI must answer a sharper question: ",[193,1850,1851],{},"on this job, right now, whose name is on the execute step?",[200,1853,1742],{"href":1741}," is the full guide; do not re-derive it here. The multiplayer twist is that RBAC entries attach to the ",[193,1856,1857],{},"work object",", not only to the org chart. ",[200,1860,1861],{"href":215},"What an AI workstream is"," is that object.",[190,1864,1865],{},"Three roster mistakes show up in every pilot:",[190,1867,1868,1871],{},[193,1869,1870],{},"1. Shared inbox as signer."," “Finance@” is not a person. Auditors ask for a name, not a distribution list. A mailbox cannot refuse a payload at 22:00. A named delegate can.",[190,1873,1874,1877],{},[193,1875,1876],{},"2. Guest with write by accident."," A partner who should see their slice inherits the production token because setup was easy. Least privilege died at the invite dialog.",[190,1879,1880,1883,1884,1888,1889,1892,1893,1895,1896,1898],{},[193,1881,1882],{},"3. Agent as implicit admin."," The model runs with a service account that can edit every object “because integration.” ",[200,1885,1887],{"href":375,"rel":1886},[204],"OWASP’s LLM Top 10"," lists excessive agency as a first-class risk. Multiplayer governance is the organisational version: ",[193,1890,1891],{},"who"," could have stopped ",[193,1894,1820],{}," change on ",[193,1897,1820],{}," job.",[190,1900,1901,1904,1905,1908,1909,1912],{},[200,1902,1903],{"href":1330},"What is human-in-the-loop AI"," becomes real when the loop shows ",[193,1906,1907],{},"whose"," loop — on this roster — for ",[193,1910,1911],{},"which"," class of write. A node labelled “review” in a vendor diagram is not governance until it is a person who can refuse while others watch.",[190,1914,1915,1916,1920,1921,1925,1926,1930,1931,1935],{},"Function walkthroughs change the names, not the primitive. ",[200,1917,1919],{"href":1918},"collaborative-ai-for-customer-support","Collaborative AI for customer support"," needs a roster on the ticket. ",[200,1922,1924],{"href":1923},"collaborative-ai-for-human-resources","Collaborative AI for human resources"," needs a roster on the people decision, with policy version attached. ",[200,1927,1929],{"href":1928},"collaborative-ai-for-revenue-operations","Collaborative AI for revenue operations"," needs finance on the stage conflict. ",[200,1932,1934],{"href":1933},"collaborative-ai-for-operations","Collaborative AI for operations"," needs the closer on the hold. Same primitive. Different artefact.",[190,1937,1938,1939,1944,1945,230],{},"Microsoft and LinkedIn’s ",[200,1940,1943],{"href":1941,"rel":1942},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fworklab\u002Fwork-trend-index\u002Fai-at-work-is-here-now-comes-the-hard-part",[204],"2024 Work Trend Index"," reported that 78% of AI users at work bring their own tools. BYO tools have no roster. That is the point. Personal convenience and multiplayer accountability do not share a session. If the exception already requires two departments, a personal assistant is the wrong container — see ",[200,1946,1948],{"href":1947},"collaborative-ai-and-personal-assistants","collaborative AI and personal assistants",[255,1950,1952],{"id":1951},"inherited-authority-the-agent-is-not-a-superuser","Inherited authority: the agent is not a superuser",[190,1954,1955,1956,1959],{},"“Inherited authority” means the agent’s grants are the ",[193,1957,1958],{},"intersection"," of what the job allows and what the acting person may do — never the union of every connector in the tenant.",[190,1961,1962,1963,1966],{},"If the WMS integration can release any hold but the operator is a warehouse supervisor without credit authority, the agent must not release a credit hold because the prompt was confident. Least privilege for AI is not a smaller model. It is a smaller ",[193,1964,1965],{},"effective"," identity for this session.",[190,1968,1969,1970,1975],{},"IBM’s ",[200,1971,1974],{"href":1972,"rel":1973},"https:\u002F\u002Fnewsroom.ibm.com\u002F2024-07-30-ibm-report-escalating-data-breach-disruption-pushes-costs-to-new-highs",[204],"Cost of a Data Breach"," report (2024) is often cited for breach dollars. The operational cousin is a customer record changed without a name — no breach required. Inherited authority is how you keep the fluent paragraph from becoming a fact the company inherits.",[190,1977,1978,1980,1981,1984],{},[200,1979,474],{"href":228}," is the execute gate: show the payload, require the signer, fail-closed if missing. Inherited authority is the ",[193,1982,1983],{},"read and propose"," gate: which sources this session may touch at all. A model that can see every forecast “for context” has already lost the multiplayer plot, even if it never writes.",[190,1986,1987,1988,1992],{},"When agents are ",[200,1989,1991],{"href":1990},"agents-should-be-disposable","disposable",", inherited authority survives swaps. The next model does not arrive with a fresh superuser token. It arrives with the same bounded grants on the same job. Continuity without bounds is how yesterday’s exception becomes tomorrow’s god session.",[190,1994,1995,1999,2000,2004],{},[200,1996,1998],{"href":1997},"search-is-not-memory","Search is not memory"," is the retrieval mistake; ",[200,2001,2003],{"href":2002},"what-is-institutional-memory-in-enterprise-ai","institutional memory in enterprise AI"," is the company-scale layering. Multiplayer governance is the live-job layer: asserted policy and decision rights sit on the roster while the work is open. Do not collapse those layers into one index and call it control.",[190,2006,1752,2007,2011],{},[200,2008,601],{"href":2009,"rel":2010},"https:\u002F\u002Fartificialintelligenceact.eu\u002F",[204]," documentation instinct — who did what, under which procedure — is easier to meet when authority is inherited per job than when a tenant-wide service account did everything. You do not need to be in scope of the Act to want a name on the execute step.",[255,2013,2015],{"id":2014},"spend-as-scheduling-input","Spend as scheduling input",[190,2017,2018,2019,2022],{},"Spend caps are usually drawn on finance slides: monthly tokens, seat tiers, flagship defaults. In a multiplayer job, spend is also ",[193,2020,2021],{},"scheduling",": the run stops, the roster is notified, a delegate decides whether this exception is worth another hour of frontier model on extraction.",[190,2024,2025,2026,2030,2031,2034,2035,2037],{},"That is not stinginess. It is visibility. ",[200,2027,2029],{"href":2028},"what-is-ai-token-economics","What is AI token economics"," covers the categories. Multiplayer governance covers ",[193,2032,2033],{},"who on the roster"," may raise the cap for ",[193,2036,1820],{}," hold, with a record.",[190,2039,2040],{},"Uncapped “always flagship” on shared jobs is how two departments burn budget on the same conflict without seeing each other’s runs. McKinsey’s 2025 survey keeps showing use without redesign. Spend-as-scheduling is a small redesign: the job names a default route, a cap stops the loop, and a human decides.",[190,2042,2043],{},"Treat spend like overtime approval:",[260,2045,2046,2049,2059],{},[263,2047,2048],{},"The job names a default route — compact for classify, frontier for judgement.",[263,2050,2051,2052,2055,2056,230],{},"A cap stops the loop and pings the ",[193,2053,2054],{},"roster",", not ",[422,2057,2058],{},"#ai-governance",[263,2060,2061],{},"A delegate’s approval is stored next to the job, not only in the billing console.",[190,2063,2064,2066,2067,2070],{},[200,2065,1088],{"href":251}," asks vendors for per-step model breakdown on a live run. Multiplayer governance asks whether that breakdown is visible to the finance delegate ",[193,2068,2069],{},"on the job"," when a hold must close tonight. A platform invoice that finance sees next quarter is not a control. A pause the closer can lift tonight is.",[190,2072,2073,2077],{},[200,2074,2076],{"href":2075},"what-to-look-for-in-model-routing","What to look for in model routing"," is the routing sheet. Routing without a roster is still a private optimisation. Routing with a roster is a scheduling decision other people can see.",[255,2079,2081],{"id":2080},"notify-the-roster-not-the-org","Notify the roster, not the org",[190,2083,2084],{},"Alert fatigue kills governance. Company-wide “AI used sensitive data” emails train people to ignore the channel that matters.",[190,2086,2087],{},"Multiplayer primitives route signal to people who can act:",[260,2089,2090,2093,2099],{},[263,2091,2092],{},"Finance when a release payload is waiting for sign.",[263,2094,2095,2096,2098],{},"Legal when a clause draft touches export control on ",[193,2097,1820],{}," contract job.",[263,2100,2101,2102,2104],{},"Ops when a WMS exception ages past SLA on ",[193,2103,1820],{}," SKU lane.",[190,2106,1752,2107,2112],{},[200,2108,2111],{"href":2109,"rel":2110},"https:\u002F\u002Fwww.whitehouse.gov\u002Fwp-content\u002Fuploads\u002F2024\u002F03\u002FM-24-10-Advancing-Governance-Innovation-and-Risk-Management-for-Agency-Use-of-Artificial-Intelligence.pdf",[204],"US OMB M-24-10"," memorandum on federal AI use puts inventory and named owners first. Multiplayer notify is the operational version: the owner is on the roster for the job that changed, not an abstract “AI council.”",[190,2114,2115,2120,2121,2124],{},[200,2116,2119],{"href":2117,"rel":2118},"https:\u002F\u002Fico.org.uk\u002Ffor-organisations\u002Fuk-gdpr-guidance-and-resources\u002Fartificial-intelligence\u002F",[204],"UK ICO guidance on AI"," still wants purpose and retention ",[193,2122,2123],{},"before"," you turn a tool loose. Purpose is easier to defend when notify is scoped to the people who needed the processing. A blast to the company is not a purpose. It is a habit.",[2126,2127,2131],"pre",{"className":2128,"code":2129,"language":2130,"meta":171,"style":171},"language-mermaid shiki shiki-themes github-light github-dark","flowchart TB\n  job[\"Shared job\"] --> roster[\"People on the roster\"]\n  agent[\"Someone proposes a change\"] --> payload[\"The exact change, quoted\"]\n  payload --> roster\n  roster --> sign{\"Named signer\"}\n  sign -->|approve| write[\"The write goes through\"]\n  sign -->|reject| stored[\"The no stays on the job\"]\n  stored --> job\n  spend[\"Spend cap hit\"] --> roster\n","mermaid",[422,2132,2133,2141,2146,2151,2157,2163,2169,2175,2181],{"__ignoreMap":171},[2134,2135,2138],"span",{"class":2136,"line":2137},"line",1,[2134,2139,2140],{},"flowchart TB\n",[2134,2142,2143],{"class":2136,"line":172},[2134,2144,2145],{},"  job[\"Shared job\"] --> roster[\"People on the roster\"]\n",[2134,2147,2148],{"class":2136,"line":1001},[2134,2149,2150],{},"  agent[\"Someone proposes a change\"] --> payload[\"The exact change, quoted\"]\n",[2134,2152,2154],{"class":2136,"line":2153},4,[2134,2155,2156],{},"  payload --> roster\n",[2134,2158,2160],{"class":2136,"line":2159},5,[2134,2161,2162],{},"  roster --> sign{\"Named signer\"}\n",[2134,2164,2166],{"class":2136,"line":2165},6,[2134,2167,2168],{},"  sign -->|approve| write[\"The write goes through\"]\n",[2134,2170,2172],{"class":2136,"line":2171},7,[2134,2173,2174],{},"  sign -->|reject| stored[\"The no stays on the job\"]\n",[2134,2176,2178],{"class":2136,"line":2177},8,[2134,2179,2180],{},"  stored --> job\n",[2134,2182,2184],{"class":2136,"line":2183},9,[2134,2185,2186],{},"  spend[\"Spend cap hit\"] --> roster\n",[190,2188,2189,2191],{},[200,2190,1861],{"href":215}," is the object that holds roster, payload, and stored rejection together. Demand that shape in any shared product. Do not accept “we have audit logs somewhere” as a substitute for a visible stop on the job.",[255,2193,2195],{"id":2194},"why-a-pdf-after-the-write-is-not-governance","Why a PDF after the write is not governance",[190,2197,2198],{},"PDFs and training videos have a role: policy, inventory, DPIA thinking. None of that stops an unsigned PATCH at 22:00.",[190,2200,2201],{},"Multiplayer governance is enforceable at the moment of change:",[260,2203,2204,2207,2210],{},[263,2205,2206],{},"The payload is visible to the roster before execution.",[263,2208,2209],{},"A rejection is stored on the job — not “we discussed in Teams.”",[263,2211,2212],{},"Fail-closed is default; approval is explicit.",[190,2214,2215,2216,2220,2221,2223],{},"Air Canada’s chatbot case — ",[200,2217,2219],{"href":1477,"rel":2218},[204],"CBC’s report"," on bereavement fares — is customer-facing write-back without a gate. Multiplayer governance would have required a named signer on the message, visible to whoever owns customer commitments, ",[193,2222,2123],{}," the customer relied on it. The lawsuit is about a commitment. The missing primitive was a stop in the room that produced the commitment.",[190,2225,2226,2230,2231,2234],{},[200,2227,2229],{"href":2228},"four-pillars-of-an-enterprise-ai-platform","Four pillars of an enterprise AI platform"," is the wider stack — wiki, workstreams, routing, governance as layers. This page is the ",[193,2232,2233],{},"multiplayer"," layer: controls co-located with the job, not bolted on after adoption. A wiki without a roster is asserted policy nobody on the exception can see. A workstream without a stop is a shared folder with a model.",[190,2236,2237,2242,2243,2246,2247,2250,2251,2254],{},[200,2238,2241],{"href":2239,"rel":2240},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\u002Fnist-ai-rmf-playbook",[204],"NIST’s AI RMF Playbook"," gives measurement language. Translate it: ",[193,2244,2245],{},"Map"," the job and the write class, ",[193,2248,2249],{},"Measure"," signer completeness and time-to-rejection, ",[193,2252,2253],{},"Manage"," by refusing write tokens until a stored “no” exists on this exception type. Measurement that lives only in a GRC tool the operators never open will not change the 22:00 write.",[255,2256,2258],{"id":2257},"how-multiplayer-governance-connects-to-evaluation","How multiplayer governance connects to evaluation",[190,2260,2261,2265],{},[200,2262,2264],{"href":2263},"how-to-evaluate-collaborative-ai","How to evaluate collaborative AI"," asks whether a second department can join, whether rejections store, and what happens after the session. Those are governance questions dressed as collaboration questions. If you only score fluency, you will buy a shared login.",[190,2267,2268,2270],{},[200,2269,1088],{"href":251}," asks for a refused write and a replay export. Multiplayer governance asks whether finance on the roster saw the same payload the harness refused. A harness that refuses in a log finance cannot open is a control for engineers, not for the job.",[190,2272,2273,2274,2277],{},"When you run a proof of value, script a ",[193,2275,2276],{},"stored rejection"," before any write token. If the vendor cannot show the “no” on the job Monday, you do not have multiplayer governance — you have a chat with audit logs somewhere else.",[190,2279,2280,2283],{},[200,2281,2282],{"href":1990},"Agents should be disposable"," is the continuity claim: roster, payload, and rejection survive agent swaps. Governance as a multiplayer primitive is why those artefacts are in the room — enforced, visible, fail-closed — not in a PDF that arrives after the field moved.",[255,2285,2287],{"id":2286},"how-to-start-without-a-committee-project","How to start without a committee project",[190,2289,2290],{},"You do not need a new steering group to put a stop in one room.",[547,2292,2293,2296,2299,2305,2308],{},[263,2294,2295],{},"Pick one recurring cross-team job already fought in chat.",[263,2297,2298],{},"Put it on a named work object with a roster — finance, ops, owner — not a shared login.",[263,2300,2301,2302,2304],{},"Require a ",[193,2303,2276],{}," before enabling write. Week one’s “no” is the control.",[263,2306,2307],{},"Route spend-cap alerts to the roster, not the company feed.",[263,2309,2310],{},"After two cycles, ask an independent reader to reconstruct signer and payload without Slack.",[190,2312,2313,2316],{},[200,2314,2315],{"href":1947},"Collaborative AI and personal assistants"," is when the work should stay personal. Multiplayer governance begins when two teams must stand on the same change. If only one person drafts and nothing writes, do not invent a roster for theatre.",[190,2318,2319,2320,2322],{},"Start read-only. Lists and payloads first. Customer-facing mail and bulk writes last. ",[200,2321,359],{"href":358}," is why the prompt alone is not the system; the environment around the model — tools, stops, checks — is. Multiplayer governance is the human half of that environment: who is in the room, what they can refuse, and who gets paged when the cap hits.",[190,2324,2325],{},"If you cannot name the signer for this class of write, you are not ready for execute. If you can name them but they are not on the job, you have an org chart, not a primitive.",[255,2327,793],{"id":792},[190,2329,2330,2331,2333,2334,2337],{},"In Nimbus, ",[200,2332,340],{"href":40}," is attached to the ",[200,2335,2336],{"href":32},"workstream",": roster, inherited connector grants, spend cap as a pause that notifies the people on that job, and fail-closed writes with a stored rejection.",[190,2339,2340],{},"The agent on the workstream proposes. It does not inherit a tenant-wide superuser token. A partner guest sees a scoped slice. A finance delegate sees the same payload ops sees. Alerts go to the roster, not the company feed.",[190,2342,2343,2344,2347],{},"Score that shape with ",[200,2345,2346],{"href":2263},"how to evaluate collaborative AI"," on any vendor, including this one. If the proof of value cannot show a named “no” on the job before a write token exists, you are still looking at a chat with a policy PDF.",[2349,2350,2351],"style",{},"html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":171,"searchDepth":172,"depth":172,"links":2353},[2354,2355,2356,2357,2358,2359,2360,2361,2362],{"id":1776,"depth":172,"text":1777},{"id":1841,"depth":172,"text":1842},{"id":1951,"depth":172,"text":1952},{"id":2014,"depth":172,"text":2015},{"id":2080,"depth":172,"text":2081},{"id":2194,"depth":172,"text":2195},{"id":2257,"depth":172,"text":2258},{"id":2286,"depth":172,"text":2287},{"id":792,"depth":172,"text":793},"2026-09-12","Governance belongs in the room — roster, inherited authority, spend as scheduling, notify the job not the org. Not a PDF after the write. A guide to multiplayer controls.",{"eyebrow":2366,"title":2367},"Common questions","Controls in the room",[2369,2372,2375,2378],{"question":2370,"answer":2371},"Is this the same as RBAC?","RBAC answers who may do what. This page answers why those rules must live on the shared job — roster, inherited authority, spend caps — so a second department sees the same stop. Org-wide roles are necessary and not sufficient. A finance director who can theoretically approve writes still needs to be on this job, looking at this payload, when the hold is about to clear. If the only access model is the login page, two people can share a room and still not share a control. Put the grant on the work object.",{"question":2373,"answer":2374},"Do we still need a governance committee?","Yes for policy, inventory, and the questions a committee is good at: which uses are allowed, which data classes are in scope, who owns the programme. No committee replaces a named signer on the job that is about to change CRM. Quarterly attestation while unsigned writes succeed is principles, not a primitive. Keep the committee. Put the stop in the room. Measure both: policy coverage and stored rejections on live jobs.",{"question":2376,"answer":2377},"What is the first multiplayer gate to add?","A roster with a stored rejection before any write token. If nobody can say no on the job, you have a room without a door. Do not start with a company-wide AI policy email. Do not start with a spend dashboard nobody on the exception can see. Name the people on one recurring job, attach the files, and require a visible “no” before anyone enables execute. Week one’s rejection is the control. Week four’s write, if you get there, inherits it.",{"question":2379,"answer":2380},"Is notify-the-roster just another Slack channel?","No. A company-wide “AI used sensitive data” channel trains people to ignore the signal. Multiplayer notify goes to people who can act on this payload — finance on this release, legal on this clause, ops on this SKU lane. The owner is on the job that changed, not an abstract council. If the alert cannot name the job, the payload, and the next action, it is noise. Route spend-cap pauses the same way: the roster decides whether this exception is worth another hour, and the decision stays on the job.","\u002Fblog\u002Fgovernance-as-a-multiplayer-primitive",{"title":1709,"description":2364},"blog\u002Fgovernance-as-a-multiplayer-primitive",[1012,340,2233,2385],"RBAC","KJKqlk2jj1ICBw4Jp4lWmnvHVFXM8PcPEvWRNaGbxAk",{"enabled":165,"message":2388,"linkLabel":79,"linkHref":80,"id":2389,"title":2390,"archived":165,"authors":166,"badge":166,"body":2391,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2395,"relatedHeading":166,"seo":2396,"series":166,"sitemap":165,"status":166,"stem":2397,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2398},"We're hiring! Join the team building the Sentient Enterprise.","content\u002Fshared\u002Fhiring.md","Hiring banner",{"type":168,"value":2392,"toc":2393},[],{"title":171,"searchDepth":172,"depth":172,"links":2394},[],"\u002Fshared\u002Fhiring",{"title":2390,"description":171},"shared\u002Fhiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",{"fold":2400,"id":2404,"title":2405,"archived":165,"authors":166,"badge":166,"body":2406,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":2410,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2414,"relatedHeading":166,"seo":2415,"series":166,"sitemap":165,"status":166,"stem":2416,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2417},{"headline":2401,"description":2402,"primaryLabel":8,"primaryTo":2403,"secondaryLabel":1034,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Start on Free.","\u002Fsignup?plan=free","content\u002Fshared\u002Fcta.md","Site CTAs",{"type":168,"value":2407,"toc":2408},[],{"title":171,"searchDepth":172,"depth":172,"links":2409},[],{"headline":2411,"description":2412,"primaryLabel":8,"primaryTo":2403,"secondaryLabel":2413,"secondaryTo":85},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger. Start on Free.","Talk to our team","\u002Fshared\u002Fcta",{"title":2405,"description":171},"shared\u002Fcta","PS2VPJsszmUpMBZT6nEp8cWXCdeiN6zDRl-p8d0uY2k",1790215701211]