[{"data":1,"prerenderedAt":2220},["ShallowReactive",2],{"site-nav-content":3,"blog:\u002Fblog\u002Fwhat-is-an-enterprise-agent-harness":179,"blog-index-copy":899,"blog:\u002Fblog\u002Fwhat-is-an-enterprise-agent-harness:surround":920,"hiring-banner-content":2189,"site-cta-content":2201},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":132,"id":163,"title":164,"archived":165,"authors":166,"badge":166,"body":167,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":175,"relatedHeading":166,"seo":176,"series":166,"sitemap":165,"status":166,"stem":177,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":178},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","\u002Foverview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","\u002Fproduct\u002Fconflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","\u002Fproduct\u002Fagent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","\u002Fproduct\u002Flifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","\u002Fproduct\u002Fwiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","\u002Fproduct\u002Fworkstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","\u002Fproduct\u002Fperception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","\u002Fproduct\u002Fgovernance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","\u002Fmodels",{"label":47,"to":48},"Pricing","\u002Fpricing",{"label":50,"to":51},"Integrations","\u002Fintegrations",{"label":53,"to":54},"Security","\u002Fsecurity",{"label":56,"to":57},"Partners","\u002Fpartners",{"label":59,"to":60},"Insights","\u002Fblog",{"productHeading":5,"companyHeading":62,"resourcesHeading":63,"legalHeading":64,"docsLabel":65,"docsUrl":66,"statementLines":67,"copyright":70,"companyLinks":71,"resourcesLinks":86,"legalLinks":102,"socialLinks":109,"bottomLinks":119},"Company","Resources","Legal","Docs","https:\u002F\u002Fdocs.gonimbus.ai",[68,69],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[72,73,74,75,76,78,81,84],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":77,"to":57},"Partner Program",{"label":79,"to":80},"Careers","\u002Fcareers",{"label":82,"to":83},"System status","\u002Fstatus",{"label":7,"to":85},"\u002Fcontact",[87,90,93,96,99],{"label":88,"to":89},"Glossary","\u002Fglossary",{"label":91,"to":92},"Compare","\u002Fcompare",{"label":94,"to":95},"Evaluate","\u002Fevaluate",{"label":97,"to":98},"Problems","\u002Fproblems",{"label":100,"to":101},"Use cases","\u002Fuse-cases",[103,106],{"label":104,"to":105},"Terms of Service","\u002Fterms",{"label":107,"to":108},"Privacy Policy","\u002Fprivacy",[110,113,116],{"label":111,"href":112},"LinkedIn","https:\u002F\u002Fwww.linkedin.com\u002Fcompany\u002Fgonimbusai\u002F",{"label":114,"href":115},"X","https:\u002F\u002Fx.com\u002Fgonimbusai",{"label":117,"href":118},"Instagram","https:\u002F\u002Fwww.instagram.com\u002Fgonimbus_ai\u002F",[120,122,124,127,128],{"label":121,"to":105},"Terms",{"label":123,"to":108},"Privacy",{"label":125,"to":126},"Compliance","\u002Fcompliance",{"label":82,"to":83},{"label":129,"to":130,"external":131},"LLMs.txt","\u002Fllms.txt",true,{"text":133,"prompt":134},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":135,"platforms":137},{" Summarize the highlights from Nimbus's website":136},"https:\u002F\u002Fgonimbus.ai",[138,143,148,153,158],{"name":139,"label":140,"icon":141,"hrefPrefix":142},"chatgpt","ChatGPT","simple-icons:openai","https:\u002F\u002Fchatgpt.com\u002F?prompt=",{"name":144,"label":145,"icon":146,"hrefPrefix":147},"perplexity","Perplexity","mdi:magnify","https:\u002F\u002Fwww.perplexity.ai\u002Fsearch\u002Fnew?q=",{"name":149,"label":150,"icon":151,"hrefPrefix":152},"grok","Grok","simple-icons:x","https:\u002F\u002Fx.com\u002Fi\u002Fgrok?text=",{"name":154,"label":155,"icon":156,"hrefPrefix":157},"claude","Claude","simple-icons:anthropic","https:\u002F\u002Fclaude.ai\u002Fnew?q=",{"name":159,"label":160,"icon":161,"hrefPrefix":162},"google-ai","Google AI","simple-icons:google","https:\u002F\u002Fwww.google.com\u002Fsearch?udm=50&aep=11&q=","content\u002Fshared\u002Fnav.md","Site navigation",false,null,{"type":168,"value":169,"toc":170},"minimark",[],{"title":171,"searchDepth":172,"depth":172,"links":173},"",2,[],"md","\u002Fshared\u002Fnav",{"title":164,"description":171},"shared\u002Fnav","1dD7ahDRl0SQ4hz53-kKo0tEFrGaLuaztZ3PPfp6a9k",{"id":180,"title":181,"archived":165,"authors":182,"badge":185,"body":187,"date":889,"definedTerm":166,"department":166,"description":890,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":891,"relatedHeading":166,"seo":892,"series":893,"sitemap":131,"status":166,"stem":894,"subhead":166,"tags":895,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":898},"content\u002Fblog\u002Fwhat-is-an-enterprise-agent-harness.md","What is an Enterprise Agent Harness",[183],{"name":184,"to":136},"Nimbus Research",{"label":186},"Explainer",{"type":168,"value":188,"toc":869},[189,198,215,232,237,309,331,335,357,360,377,404,424,428,436,455,475,490,506,516,525,536,542,546,551,558,569,572,586,590,602,609,612,615,628,632,635,641,652,655,659,676,688,692,697,700,704,707,711,714,718,726,730,738,742,756,760,769,773],[190,191,192,193,197],"p",{},"An ",[194,195,196],"strong",{},"enterprise agent harness"," is the outer runtime that lets a model work on company jobs: policy it actually loads, connectors with least privilege, a loop that can stop for a named signer, and a record you can query after the people change.",[190,199,200,201,208,209,214],{},"It is still ",[202,203,207],"a",{"href":204,"rel":205},"https:\u002F\u002Fdocs.langchain.com\u002Foss\u002Fpython\u002Flangchain\u002Fagents",[206],"nofollow","Agent = Model + Harness",". The workspace is not a git root. The sensor is not only pytest. The stop is not only max steps. ",[202,210,213],{"href":211,"rel":212},"https:\u002F\u002Fwww.thoughtworks.com\u002Finsights\u002Farticles\u002Foperating-system-enterprise-ai",[206],"Thoughtworks"," calls the missing piece an organisational harness: identity, ownership, economics, and learning around whatever builder harnesses (Claude Code, Cursor, LangChain graphs) teams already bought. An enterprise agent harness is that layer made operable — whether you assemble it or hire it.",[190,216,217,221,222,226,227,231],{},[202,218,220],{"href":219},"inner-vs-outer-agent-harness","Inner vs outer"," is the cut. This page is the outer object in full. An ",[202,223,225],{"href":224},"what-is-an-enterprise-ai-operating-system","enterprise AI operating system"," is the product category that usually ships it: wiki, ",[202,228,230],{"href":229},"what-is-an-ai-workstream","workstreams",", teams, gates, ledger. You can have OS-class products (Nimbus, Palantir AIP, Salesforce Agentforce) and still fail the harness test if writes are a boolean on an API key. You can assemble an enterprise harness in LangGraph and pass the test. The noun is the runtime properties, not the logo.",[233,234,236],"h2",{"id":235},"words-youll-hear","Words you’ll hear",[238,239,240,251,257,263,273,279,289,299],"ul",{},[241,242,243,246,247,250],"li",{},[194,244,245],{},"Outer harness."," Company workspace. See ",[202,248,249],{"href":219},"inner vs outer",".",[241,252,253,256],{},[194,254,255],{},"Organizational harness."," Thoughtworks’ fourth layer after model, builder harness, and user harness. Governance architecture, not another markdown file.",[241,258,259,262],{},[194,260,261],{},"Workstream."," Isolation domain: roster, connectors, budget, finish line. The job folder. Not a chat title.",[241,264,265,268,269,250],{},[194,266,267],{},"Write quoting."," The human sees the change in the language of the live system before sign-off. ",[202,270,272],{"href":271},"what-is-write-back-governance","Write-back governance",[241,274,275,278],{},[194,276,277],{},"Fail-closed."," Missing approval, detached grant, or down interceptor means nothing mutates. Fail-open is a faster incident.",[241,280,281,284,285,250],{},[194,282,283],{},"Ledger \u002F Lifecycle Graph."," AI operations events: brief, agents, policy version, signer, payload. Distinct from the warehouse’s business events. See ",[202,286,288],{"href":287},"what-is-a-lifecycle-graph","What is a lifecycle graph",[241,290,291,294,295,250],{},[194,292,293],{},"SWE-bench \u002F Terminal-Bench."," Inner evals. Useful for engineering vendors. Not a SOX control. ",[202,296,298],{"href":297},"eval-loops-for-enterprise-agent-harnesses","Eval loops",[241,300,301,304,305,250],{},[194,302,303],{},"Forward-deployed programme."," Vendor engineers for months. AIP at scale. Capability can be real. Time-to-value is staffing. ",[202,306,308],{"href":307},"self-service-vs-forward-deployed-ai-platforms","Self-service vs forward-deployed",[190,310,311,312,315,316,315,318,315,321,315,324,315,327,330],{},"Nimbus is one self-service enterprise harness: ",[202,313,314],{"href":28},"wiki",", ",[202,317,230],{"href":32},[202,319,320],{"href":20},"agent teams",[202,322,323],{"href":40},"governance",[202,325,326],{"href":24},"graph",[202,328,329],{"href":45},"routing",". Score it as an example of the shape, next to AIP and Agentforce, nothe definition of the category.",[233,332,334],{"id":333},"why-you-should-care","Why you should care",[190,336,337,342,343,347,348,351,352,356],{},[202,338,341],{"href":339,"rel":340},"https:\u002F\u002Fwww.mckinsey.com\u002Fcapabilities\u002Fquantumblack\u002Four-insights\u002Fthe-state-of-ai",[206],"McKinsey’s 2025 State of AI"," keeps separating ",[344,345,346],"em",{},"use"," from ",[344,349,350],{},"scale",". Copilots and coding harnesses can produce the first. Enterprise harnesses are how writes to systems of record become the second without becoming ",[202,353,355],{"href":354},"what-is-shadow-ai","shadow AI"," in the CRM.",[190,358,359],{},"It affects you if:",[238,361,362,365,368,371,374],{},[241,363,364],{},"RevOps, Legal, and Finance must share a job, not a Slack channel of screenshots",[241,366,367],{},"Salesforce or NetSuite can change because a model proposed it",[241,369,370],{},"last quarter’s pricing chat is unrecoverable",[241,372,373],{},"security cannot list the AI actors that may write",[241,375,376],{},"the vendor demo is a SWE-bench plot and a “we have MCP”",[190,378,379,380,385,386,391,392,397,398,403],{},"In 2024 Air Canada was held to a chatbot’s invented policy (",[202,381,384],{"href":382,"rel":383},"https:\u002F\u002Fwww.cbc.ca\u002Fnews\u002Fcanada\u002Fbritish-columbia\u002Fair-canada-chatbot-lawsuit-1.7116416",[206],"CBC","). That is an outer-harness failure: a commitment left the building without a quote or a signer. ",[202,387,390],{"href":388,"rel":389},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2016\u002F679\u002Foj",[206],"GDPR"," constrains personal data in payloads. ",[202,393,396],{"href":394,"rel":395},"https:\u002F\u002Fwww.sec.gov\u002Fabout\u002Flaws.shtml",[206],"Sarbanes–Oxley"," constrains who may change revenue truth. ",[202,399,402],{"href":400,"rel":401},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2024\u002F1689\u002Foj",[206],"EU AI Act"," Article 14 wants people who can interpret, interrupt, and leave a record. A coding-agent hook that formats Python does not satisfy those.",[190,405,406,411,412,417,418,423],{},[202,407,410],{"href":408,"rel":409},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework",[206],"NIST AI RMF"," and ",[202,413,416],{"href":414,"rel":415},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F42001",[206],"ISO\u002FIEC 42001"," assume operational controls, not a slide titled governance. ",[202,419,422],{"href":420,"rel":421},"https:\u002F\u002Foecd.ai\u002Fen\u002Fai-principles",[206],"OECD AI Principles"," are a board checklist. They do not implement a gate. The harness does.",[233,425,427],{"id":426},"what-enterprise-adds-to-a-harness","What “enterprise” adds to a harness",[190,429,430,431,435],{},"Start from ",[202,432,434],{"href":433},"what-is-an-agent-harness","what a harness contains"," — loop, tools, memory, permissions, feedback, orchestration — and raise the bar.",[190,437,438,441,442,446,447,450,451,454],{},[194,439,440],{},"Policy that loads."," Inner harnesses inject ",[443,444,445],"code",{},"AGENTS.md",". Enterprise harnesses inject asserted company policy for ",[344,448,449],{},"this"," job, versioned. A Drive dump is not policy. A ",[202,452,314],{"href":453},"what-is-a-company-wiki-for-ai-agents"," that agents cite, with the revision on the run, is. If Legal’s discount cap lives only in a PDF nobody attached, the model will invent a number. That is not hallucination as a personality. That is a missing guide.",[190,456,457,460,461,465,466,470,471,250],{},[194,458,459],{},"Connectors as grants, not a toolbox."," Default read. Write is a separate plane. Least privilege is a workstream property. ",[202,462,464],{"href":463},"connector-and-permissions-architecture","Connector architecture",". MCP may be the plug; it must inherit the grant. ",[202,467,469],{"href":468},"mcp-for-enterprise-integrations","MCP for enterprise integrations",". A Finance team assigned to a GTM-only stream still must not reach ERP “because it is Finance.” ",[202,472,474],{"href":473},"agent-team-architecture","Agent team architecture",[190,476,477,480,481,484,485,489],{},[194,478,479],{},"A hiring object for operators."," Not a folder of personal GPTs. A mandate, required systems, approval triggers — ",[202,482,320],{"href":483},"what-is-multi-agent-ai"," as a roster. ",[202,486,488],{"href":487},"how-to-evaluate-agent-teams-vs-single-agents","How to evaluate agent teams vs single agents",". Nimbus ships functional teams on that roster; AIP and Agentforce have their own packaging. The test is: can an operator inspect the mandate and the required systems before assign.",[190,491,492,495,496,500,501,505],{},[194,493,494],{},"Human wait as a step."," ",[202,497,499],{"href":498},"what-is-human-in-the-loop-ai","HITL"," is not a kill switch in a dashboard. It is quoted payload, named role, fail-closed adapter. ",[202,502,504],{"href":503},"human-in-the-loop-approval-architecture","HITL approval architecture",". Soft \u002F Hard \u002F Critical matched to blast radius. A six-month zero-reject rate on CRM writes is a finding.",[190,507,508,511,512,250],{},[194,509,510],{},"A ledger of AI operations."," Who briefed, which team, which wiki revision, who signed, what executed. Exportable without the vendor in the room. The warehouse is not this ledger. ",[202,513,515],{"href":514},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",[190,517,518,521,522,250],{},[194,519,520],{},"Evals that match the job."," Did the executed write match the signed quote. Can you replay. Inner leaderboards are a vendor quality signal for coding. They are not the enterprise eval. See ",[202,523,524],{"href":297},"eval loops",[190,526,527,530,531,535],{},[194,528,529],{},"Economics of the loop."," Routing compact extract vs frontier judgement. Spend quotes. Seat pricing that includes unlimited flagship is an unengineered cost harness. ",[202,532,534],{"href":533},"what-is-model-routing","Model routing",". Nimbus meters NTUs; copilots meter seats. Different jobs.",[190,537,538,541],{},[194,539,540],{},"Self-service vs programme."," If every new connector is a six-month SOW, you have bought a deployment, not a harness operators can tighten. That can still be the right buy for Ontology-scale complexity. It is the wrong buy for a standard Salesforce write this quarter.",[233,543,545],{"id":544},"what-it-is-not","What it is not",[190,547,548,549,250],{},"A coding harness with SSO. ",[202,550,220],{"href":219},[190,552,553,554,250],{},"A copilot with an admin console. ",[202,555,557],{"href":556},"how-to-choose-between-a-copilot-and-a-work-os","Copilot vs work OS",[190,559,560,561,564,565,250],{},"A framework. LangGraph can ",[344,562,563],{},"host"," an enterprise harness if you build grants, quotes, and a ledger. Out of the box it hosts a graph. ",[202,566,568],{"href":567},"agent-harness-vs-agent-framework","Harness vs framework",[190,570,571],{},"“We integrate with Salesforce.” Integration is a slide. A scoped connector plus a blocked unsigned write is a harness.",[190,573,574,575,411,580,585],{},"SWE-bench-first marketing. ",[202,576,579],{"href":577,"rel":578},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents",[206],"Anthropic",[202,581,584],{"href":582,"rel":583},"https:\u002F\u002Fwww.langchain.com\u002Fblog\u002Fthe-anatomy-of-an-agent-harness",[206],"LangChain"," are writing about coding and general agents. Steal the discipline (stops, artifacts, sensors). Do not steal the benchmark as your control framework.",[233,587,589],{"id":588},"thoughtworks-organisational-harness-in-operator-language","Thoughtworks’ organisational harness, in operator language",[190,591,592,593,597,598,601],{},"The ",[202,594,596],{"href":211,"rel":595},[206],"Thoughtworks OS essay"," (10 July 2026) argues that most AI programmes fail because the organisation never built the operating system around the model: accountability, ownership, measurement, learning. They name four layers. An enterprise agent harness, as this article uses the term, is layers 3–4 made runnable for ",[344,599,600],{},"company jobs"," — not only for coding-agent users.",[190,603,604,605,608],{},"Delegation failures are the tell. The model was fine. The platform ran. Practitioner guides existed. The agent did what it was ",[344,606,607],{},"allowed"," to do. The company still took harm. Layer 4 questions: who approved that autonomy, who owns the policy, what was the escalation, how do we prevent the same miss on another team. Layers 1–3 cannot answer those. A chat product cannot either.",[190,610,611],{},"Thoughtworks’ control matrix is worth stealing even if you never hire them. Use deterministic controls where the boundary is knowable: allowed actions, residency, spend ceilings, blast-radius limits. Use probabilistic controls only where judgement is required. Pair every guide with a sensor. Temporal constraints — consistency across a multi-step workflow, not a single dropdown — are the ones they say teams miss most. A scheduling agent that is locally plausible on each step and globally inconsistent is not a “hallucination.” It is a missing temporal sensor.",[190,613,614],{},"Their public examples (Parloa’s repo-resident rules\u002Fskills\u002Fcommands; Morgan Stanley’s tiered autonomy on CVE triage) are coding-adjacent. Translate them: discount policy as a versioned wiki skill; “what delegation tier does this CRM write require?” instead of “do we trust the agent.” Nimbus’s Soft \u002F Hard \u002F Critical is that tiering in product form. AIP will have a different packaging. The architectural claim is the same.",[190,616,617,411,622,627],{},[202,618,621],{"href":619,"rel":620},"https:\u002F\u002Fwww.databricks.com\u002Fblog\u002Fai-harness",[206],"Databricks",[202,623,626],{"href":624,"rel":625},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FAgent_harness",[206],"Wikipedia"," describe the runtime. Thoughtworks describe why a runtime without ownership still fails at scale. You need both descriptions when you buy.",[233,629,631],{"id":630},"what-good-looks-like-on-a-live-job","What “good” looks like on a live job",[190,633,634],{},"A renewal write: workstream isolation; Salesforce attached read-only until write is enabled; wiki revision with the cap cited on the run; team cannot start if Legal’s connector requirement is missing; model proposes a quote; Hard gate; reject leaves Stage unchanged; export shows signer without a vendor screen-share. That is an enterprise harness. A demo that only answers “what should we do about Acme” is a copilot with a logo.",[190,636,637,638,250],{},"Spend an hour asking where each Thoughtworks layer lives in the vendor’s product. If layer 4 is “our professional services team,” you are buying a programme. That can be the right buy. Name it. ",[202,639,640],{"href":307},"Self-service vs FDE",[190,642,643,644,647,648,651],{},"Operators already know the human version of this harness. Maker-checker on journals. Segregation of duties on payments. Change-advisory on production. The enterprise agent harness is those instincts encoded so a model cannot talk through them. ",[202,645,396],{"href":394,"rel":646},[206]," did not wait for LLMs; it waited for a named signer. ",[202,649,390],{"href":388,"rel":650},[206]," did not wait for MCP; it waits for purpose limitation on the payload. If your AI programme cannot point to the interceptor that enforces those, you have a chatbot with a risk register.",[190,653,654],{},"What failure looks like in the first ninety days: every department clones a GPT with the same Salesforce key; Legal’s cap lives in a slide; the only eval is “the demo was impressive”; coding-agent MCP is pointed at production “just for a spike”; the ledger is Slack. What success looks like: one roster of teams, workstream isolation, default read, a Hard refuse on the first PoV, a wiki revision on the graph, inner harnesses still compiling in repos. Nimbus is built to make the success path a product week rather than a services year. Verify that claim with the refuse. AIP may be the right path when Ontology-scale complexity is real — then the harness is a programme, and you should staff it as one.",[233,656,658],{"id":657},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[190,660,661,662,665,666,669,670,672,673,675],{},"Nimbus’s outer loop is: brief a ",[202,663,664],{"href":32},"workstream"," → assign a ",[202,667,668],{"href":20},"team"," whose connector contract is satisfied → retrieve under scope → draft on the canvas (Conflux) → quote writes → ",[202,671,323],{"href":40}," pause → execute the signed payload → commit to the ",[202,674,23],{"href":24},". Perception orients; it does not silently write. Routing picks model class per step.",[190,677,678,679,411,683,687],{},"That mapping is how we productised harness engineering for operators. It is not a claim that AIP or Agentforce are “not harnesses.” They are different time and scope. ",[202,680,682],{"href":681},"how-to-evaluate-an-enterprise-ai-operating-system","How to evaluate an enterprise AI OS",[202,684,686],{"href":685},"how-to-evaluate-an-agent-harness","how to evaluate an agent harness"," are the two sheets; use both.",[233,689,691],{"id":690},"questions-people-actually-ask","Questions people actually ask",[693,694,696],"h3",{"id":695},"do-we-need-this-if-we-already-have-claude-code","Do we need this if we already have Claude Code?",[190,698,699],{},"You need it for jobs whose workspace is the company. Keep Claude Code for repos. Do not share production SoR write tokens into the inner harness.",[693,701,703],{"id":702},"is-palantir-aip-an-enterprise-harness","Is Palantir AIP an enterprise harness?",[190,705,706],{},"It can be, as a programme-shaped outer runtime. Ask deployment time, who sets a gate without vendor engineers, and whether the ledger is yours. Category yes; evaluation still required.",[693,708,710],{"id":709},"is-agentforce-enough","Is Agentforce enough?",[190,712,713],{},"If the job is CRM-anchored and stays there, maybe. Cross-system jobs with Legal on the canvas usually need a harness that is not only Salesforce. Clear scopes; avoid two writers.",[693,715,717],{"id":716},"can-we-build-this-on-langchain","Can we build this on LangChain?",[190,719,720,721,725],{},"Yes, with time. You will rebuild grants, quoting, roster, and replay. ",[202,722,724],{"href":723},"build-vs-buy-an-enterprise-ai-os","Build vs buy",". Frameworks assemble loops; operators still need a loop they can hire.",[693,727,729],{"id":728},"whats-the-first-proof","What’s the first proof?",[190,731,732,733,737],{},"A real cross-department write: operator attaches OAuth; unsigned payload blocked; reject leaves SoR unchanged; export shows signer. ",[202,734,736],{"href":735},"how-to-run-an-enterprise-ai-proof-of-value","Proof of value",". A chat demo is not this.",[693,739,741],{"id":740},"what-should-i-read-next","What should I read next?",[190,743,744,747,748,747,752,250],{},[202,745,746],{"href":685},"How to evaluate an agent harness",". ",[202,749,751],{"href":750},"agent-harness-architecture","Agent harness architecture",[202,753,755],{"href":754},"what-is-harness-engineering","What is harness engineering",[233,757,759],{"id":758},"related-reading","Related reading",[190,761,762,411,765,250],{},[202,763,764],{"href":271},"What is write-back governance",[202,766,768],{"href":767},"rfp-questions-for-enterprise-ai-agents","RFP questions for enterprise AI agents",[233,770,772],{"id":771},"sources","Sources",[238,774,775,781,787,793,799,806,812,818,823,828,833,838,843,849,855,862],{},[241,776,777],{},[202,778,780],{"href":204,"rel":779},[206],"LangChain, Agents",[241,782,783],{},[202,784,786],{"href":619,"rel":785},[206],"Databricks, What is an AI agent harness?",[241,788,789],{},[202,790,792],{"href":624,"rel":791},[206],"Wikipedia, Agent harness",[241,794,795],{},[202,796,798],{"href":211,"rel":797},[206],"Thoughtworks, The operating system for enterprise AI",[241,800,801],{},[202,802,805],{"href":803,"rel":804},"https:\u002F\u002Fwww.thoughtworks.com\u002Finsights\u002Fpodcasts\u002Ftechnology-podcasts\u002Fscaling-the-enterprise-harness--how-to-achieve-ai-agent-controll",[206],"Thoughtworks, Scaling the enterprise harness",[241,807,808],{},[202,809,811],{"href":577,"rel":810},[206],"Anthropic, Building effective agents",[241,813,814],{},[202,815,817],{"href":339,"rel":816},[206],"McKinsey, The state of AI in 2025",[241,819,820],{},[202,821,410],{"href":408,"rel":822},[206],[241,824,825],{},[202,826,416],{"href":414,"rel":827},[206],[241,829,830],{},[202,831,422],{"href":420,"rel":832},[206],[241,834,835],{},[202,836,402],{"href":400,"rel":837},[206],[241,839,840],{},[202,841,390],{"href":388,"rel":842},[206],[241,844,845],{},[202,846,848],{"href":394,"rel":847},[206],"SEC, Sarbanes–Oxley",[241,850,851],{},[202,852,854],{"href":382,"rel":853},[206],"CBC, Air Canada chatbot lawsuit",[241,856,857],{},[202,858,861],{"href":859,"rel":860},"https:\u002F\u002Fmodelcontextprotocol.io\u002Fspecification\u002F2025-11-25\u002Findex",[206],"Model Context Protocol specification",[241,863,864],{},[202,865,868],{"href":866,"rel":867},"https:\u002F\u002Fwww.swebench.com\u002F",[206],"SWE-bench",{"title":171,"searchDepth":172,"depth":172,"links":870},[871,872,873,874,875,876,877,878,887,888],{"id":235,"depth":172,"text":236},{"id":333,"depth":172,"text":334},{"id":426,"depth":172,"text":427},{"id":544,"depth":172,"text":545},{"id":588,"depth":172,"text":589},{"id":630,"depth":172,"text":631},{"id":657,"depth":172,"text":658},{"id":690,"depth":172,"text":691,"children":879},[880,882,883,884,885,886],{"id":695,"depth":881,"text":696},3,{"id":702,"depth":881,"text":703},{"id":709,"depth":881,"text":710},{"id":716,"depth":881,"text":717},{"id":728,"depth":881,"text":729},{"id":740,"depth":881,"text":741},{"id":758,"depth":172,"text":759},{"id":771,"depth":172,"text":772},"2026-08-24","An enterprise agent harness is the outer runtime for operators: wiki, scoped connectors, agent teams, write gates, and a ledger — not a SWE-bench score and not a chat with every production login.","\u002Fblog\u002Fwhat-is-an-enterprise-agent-harness",{"title":181,"description":890},"explainer","blog\u002Fwhat-is-an-enterprise-agent-harness",[893,896,897,323],"agent-harness","enterprise-ai","0h1cwEnohtRYy2Fn_kGzjyy-5uzKmo2jJAZspPrhhPw",{"hero":900,"id":902,"title":903,"archived":165,"authors":166,"badge":166,"body":904,"date":166,"definedTerm":166,"department":166,"description":908,"extension":174,"eyebrow":909,"faqHeader":166,"faqs":166,"footerBand":910,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":60,"relatedHeading":916,"seo":917,"series":166,"sitemap":131,"status":166,"stem":918,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":919},{"filename":901},"u2221455217_Flat_design_of_a_futuristic_minimalist_landscape__5d589295-cdea-4ea9-a262-be766881accf_1.png","content\u002Fblog\u002Findex.md","Exploring the future of intelligence.",{"type":168,"value":905,"toc":906},[],{"title":171,"searchDepth":172,"depth":172,"links":907},[],"Deep dives into pre-cognitive intelligence, sentient enterprises, and the evolving landscape of AI-driven business transformation.","Latest Research",{"headline":911,"description":912,"primaryLabel":913,"primaryTo":914,"secondaryLabel":915,"secondaryTo":12},"Stay at the frontier.","Subscribe for product updates and new insights.","Subscribe","\u002Fnewsletter","Explore the platform","More research",{"title":903,"description":908},"blog\u002Findex","BFSWGYO9bcTlaulivKYWyg08_DJHsdGg3OC6g_CG1Hw",[921,1558],{"id":922,"title":923,"archived":165,"authors":924,"badge":926,"body":927,"date":889,"definedTerm":166,"department":166,"description":1550,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1551,"relatedHeading":166,"seo":1552,"series":893,"sitemap":131,"status":166,"stem":1553,"subhead":166,"tags":1554,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1557},"content\u002Fblog\u002Fwhat-is-harness-engineering.md","What is Harness Engineering",[925],{"name":184,"to":136},{"label":186},{"type":168,"value":928,"toc":1532},[929,935,962,965,983,985,1067,1073,1075,1083,1085,1105,1119,1133,1136,1140,1150,1170,1183,1197,1211,1223,1226,1230,1236,1242,1255,1266,1275,1279,1286,1292,1298,1304,1318,1321,1333,1337,1340,1354,1356,1360,1363,1367,1374,1378,1385,1389,1396,1400,1417,1419,1432,1434,1443,1445],[190,930,931,934],{},[194,932,933],{},"Harness engineering"," is the practice of treating the runtime around a model as the system you design, test, and tighten — so that when an agent fails, you change the environment, not only the prompt.",[190,936,937,940,941,944,945,950,951,956,957,961],{},[202,938,584],{"href":204,"rel":939},[206]," defines the object: Agent = Model + Harness. Harness engineering is what you ",[344,942,943],{},"do"," to that object. ",[202,946,949],{"href":947,"rel":948},"https:\u002F\u002Faddyosmani.com\u002Fblog\u002Fagent-harness-engineering\u002F",[206],"Addy Osmani"," puts the payoff in one line: a decent model with a great harness beats a great model with a bad harness. ",[202,952,955],{"href":953,"rel":954},"https:\u002F\u002Fmartinfowler.com\u002Farticles\u002Fharness-engineering.html",[206],"Birgitta Böckeler’s article on martinfowler.com"," is the user’s-side map for coding agents: guides in, sensors back. Thoughtworks then asked the organisational question: ",[202,958,960],{"href":803,"rel":959},[206],"how you scale that harness across a company"," without turning every team into a snowflake of markdown files.",[190,963,964],{},"The practice showed up because prompt engineering hit a wall that everyone could see and nobody wanted to name. You can spend a week on a system prompt. The agent will still skip the test, ignore the style guide, or report the task finished. The model is non-deterministic. The prompt is interpreted, not executed. The harness is code. That is the whole discipline.",[190,966,967,968,972,973,978,979,982],{},"This is not a replacement for ",[202,969,971],{"href":577,"rel":970},[206],"prompt"," or ",[202,974,977],{"href":975,"rel":976},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-harnesses-for-long-running-agents",[206],"context"," work. Those live ",[344,980,981],{},"inside"," the harness. Harness engineering is the wider loop: every failure becomes a rule, a hook, a test, or a denied tool — the ratchet Osmani describes — so the same mistake is cheaper the second time and impossible the tenth.",[233,984,236],{"id":235},[238,986,987,993,1013,1025,1037,1043,1049,1058],{},[241,988,989,992],{},[194,990,991],{},"Ratchet."," A failure updates the harness. Commented-out test → pre-commit hook and a reviewer check. Invented CRM field → schema quote and a Hard gate. If you only fix the artefact by hand, you did operations. You did not do harness engineering.",[241,994,995,998,999,1002,1003,315,1005,1008,1009,1012],{},[194,996,997],{},"Guides (feed-forward)."," Context the agent gets ",[344,1000,1001],{},"before"," it acts: ",[443,1004,445],{},[443,1006,1007],{},"CLAUDE.md",", architecture notes, ",[202,1010,1011],{"href":453},"company wiki"," playbooks. Böckeler’s term. Advice. Necessary. Not a stop.",[241,1014,1015,1018,1019,1024],{},[194,1016,1017],{},"Sensors (feedback)."," Deterministic checks (compiler, linter, schema, pytest) and inferential checks (LLM reviewer, specialist critic). ",[202,1020,1023],{"href":1021,"rel":1022},"https:\u002F\u002Fwww.thoughtworks.com\u002Fen-us\u002Finsights\u002Fblog\u002Fgenerative-ai\u002Fharness-engineering-agent-feedback-exploring-ai-coding-sensors",[206],"Thoughtworks on sensors",". Without sensors the agent grades its own homework.",[241,1026,1027,1030,1031,1036],{},[194,1028,1029],{},"Hooks."," Lifecycle intercepts that always run. ",[202,1032,1035],{"href":1033,"rel":1034},"https:\u002F\u002Fcode.claude.com\u002Fdocs\u002Fen\u002Fhooks",[206],"Claude Code"," can block a tool with exit code 2. LangChain middleware is the library form. A guide that says “never run rm -rf” is not a hook.",[241,1038,1039,1042],{},[194,1040,1041],{},"Harness-as-a-service."," Osmani’s HaaS framing: you used to build on completion APIs; you now build on runtime APIs (Claude Agent SDK, Codex SDK, OpenAI Agents SDK) that already own the loop, sandbox, and hooks. You configure; you do not re-implement ReAct.",[241,1044,1045,1048],{},[194,1046,1047],{},"Skill issue."," HumanLayer’s joke with a serious edge: most agent failures are configuration. Blaming the model first is how teams wait for the next release instead of adding a sensor.",[241,1050,1051,495,1053,1057],{},[194,1052,255],{},[202,1054,1056],{"href":211,"rel":1055},[206],"Thoughtworks’ enterprise layer",": who may build which harness, how exceptions work, identity, economics, learning. The gap after builder harnesses (Claude Code, Cursor) and user harnesses (guides and sensors on a repo).",[241,1059,1060,1063,1064,250],{},[194,1061,1062],{},"Eval loop."," Independent verification that does not take the model’s word. SWE-bench and Terminal-Bench for code. Quoted payload vs executed write for operations. See ",[202,1065,1066],{"href":297},"eval loops for enterprise agent harnesses",[190,1068,1069,1070,1072],{},"In Nimbus, harness engineering for operators looks like: wiki revisions as guides, connector scopes as tool policy, Soft \u002F Hard \u002F Critical as hooks on the write plane, and the ",[202,1071,23],{"href":24}," as the sensor log you can query. That is the same discipline as adding a linter. The artefact is a signed CRM change rather than a green CI job.",[233,1074,334],{"id":333},[190,1076,1077,1078,1082],{},"If you only tune prompts, every incident is a conversation. If you engineer the harness, incidents become tests. ",[202,1079,1081],{"href":408,"rel":1080},[206],"NIST’s AI RMF"," Measure and Manage steps assume you can change controls after you observe harm. A prompt history is not a control change. A hook that now fires is.",[190,1084,359],{},[238,1086,1087,1090,1096,1099,1102],{},[241,1088,1089],{},"agents already write code or propose writes to live systems",[241,1091,1092,1093,1095],{},"two teams have two ",[443,1094,1007],{}," files that contradict Legal",[241,1097,1098],{},"you cannot say which harness version ran last Tuesday",[241,1100,1101],{},"spend is “the model was verbose” rather than “the loop had no budget”",[241,1103,1104],{},"auditors ask who could have stopped the action, and the answer is “the model was supposed to ask”",[190,1106,1107,1110,1111,1114,1115,1118],{},[202,1108,341],{"href":339,"rel":1109},[206]," keeps showing usage without redesign. Harness engineering ",[344,1112,1113],{},"is"," the redesign for agentic work: not a new department named AI, a runtime with stops. ",[202,1116,416],{"href":414,"rel":1117},[206]," wants named AI actors and documented operational controls. You cannot name actors if every operator’s personal GPT is a different harness.",[190,1120,1121,1122,1125,1126,1128,1129,1132],{},"Coding teams already have half of this and do not always notice. Types, tests, CI, CODEOWNERS — Böckeler’s point is that those ",[344,1123,1124],{},"are"," sensors. The work is to point the agent at them and to add the ones that are missing (architecture fitness, behaviour: did it do what was asked). Operations teams usually have the human version — maker-checker, SoD, SOX — and have not yet wired those instincts into a loop. ",[202,1127,272],{"href":271}," is that wiring. ",[202,1130,1131],{"href":503},"Human-in-the-loop approval architecture"," is the state machine.",[190,1134,1135],{},"Air Canada’s chatbot and the sanctioned ChatGPT brief are what happens when generation reaches a system of record with no ratchet. The fix is not a sterner system prompt. The fix is a harness that cannot emit a commitment or a filing until a named person has seen the artefact.",[233,1137,1139],{"id":1138},"the-practice-not-the-slogan","The practice, not the slogan",[190,1141,1142,1145,1146,1149],{},[194,1143,1144],{},"1. Work backward from the behaviour you cannot afford to miss once."," Inner loop: never merge without tests; never ",[443,1147,1148],{},"git push --force"," to main. Outer loop: never PATCH Opportunity.Amount without a Hard quote. Write those as hooks, not as paragraphs.",[190,1151,1152,495,1155,1160,1161,1163,1164,1166,1167,1169],{},[194,1153,1154],{},"2. Separate advice from invariants.",[202,1156,1159],{"href":1157,"rel":1158},"https:\u002F\u002Fclaude.com\u002Fblog\u002Fsteering-claude-code-skills-hooks-rules-subagents-and-more",[206],"Anthropic’s steering note for Claude Code"," is unusually clear: ",[443,1162,1007],{}," is always-on context; hooks fire on events and can block. If a rule must hold when the model is tired, it graduates from markdown to a hook. Enterprise equivalent: playbooks in the ",[202,1165,314],{"href":28}," versus the interceptor in ",[202,1168,323],{"href":40},". If they conflict, the interceptor wins.",[190,1171,1172,1175,1176,1179,1180,250],{},[194,1173,1174],{},"3. Put verification outside the generator."," Anthropic’s long-running harness uses incremental commits and end-to-end checks so later sessions cannot declare victory by vibes. Coding sensors: pytest, tsc, lint. Enterprise sensors: schema of the quote, identity of the signer, hash of the payload that executed, connector grant still attached. The model may ",[344,1177,1178],{},"propose"," that it is done. The harness ",[344,1181,1182],{},"decides",[190,1184,1185,1188,1189,1191,1192,1196],{},[194,1186,1187],{},"4. Version the harness."," Which ",[443,1190,445],{},", which wiki revision, which team contract, which approval tier ran. ",[202,1193,1195],{"href":1194},"what-is-an-agentic-workflow","What is an agentic workflow"," already treats workflow version as an input. Harness engineering extends that to tools and gates. Hot-patching production prompts without a change record is how Tuesday becomes unexplained.",[190,1198,1199,1202,1203,411,1206,1210],{},[194,1200,1201],{},"5. Budget the loop."," Max steps and a cost cap that do not depend on the model’s judgement. Seat licences hide this; metered work makes it visible. See ",[202,1204,1205],{"href":533},"What is model routing",[202,1207,1209],{"href":1208},"ai-cost-control-architecture","AI cost control architecture",". Always-flagship is not careful. It is an unengineered harness.",[190,1212,1213,1216,1217,1219,1220,1222],{},[194,1214,1215],{},"6. Do not fork a harness per person."," User-owned bots are how mandates drift. Org-level ",[202,1218,320],{"href":473}," assigned to ",[202,1221,230],{"href":229}," is the enterprise form of “one CI config per repo, not one per intern.” Thoughtworks’ organisational harness is this ownership question: who is allowed to add a write tool.",[190,1224,1225],{},"Nimbus encodes several of these as product defaults — read-only connectors until you enable write, quoted payloads, graph on the way out — because operators should not have to re-implement ReAct to get a ratchet. You can still fail the practice: a wiki that is never updated, a Critical tier nobody uses, a graph nobody queries. The product is not the practice. The practice is whether last month’s incident produced a new gate.",[233,1227,1229],{"id":1228},"how-this-differs-from-adjacent-crafts","How this differs from adjacent crafts",[190,1231,1232,1235],{},[194,1233,1234],{},"Prompt engineering"," improves a single call. Necessary for tone, tool descriptions, and “what good looks like.” Insufficient for tool dispatch, identity, and replay.",[190,1237,1238,1241],{},[194,1239,1240],{},"Context engineering"," governs what the model sees this turn: compaction, retrieval, files. Anthropic’s initializer agent is context engineering in a harness. It is not permission to write NetSuite.",[190,1243,1244,1247,1248,1251,1252,250],{},[194,1245,1246],{},"Platform \u002F DevOps."," CI, sandboxes, secrets. Harness engineering ",[344,1249,1250],{},"reuses"," thosensors and execution environments. It adds the fact that the component in the loop is non-deterministic, so “the job returned zero” is not enough: you need independent tests of the ",[344,1253,1254],{},"claim",[190,1256,1257,1260,1261,1265],{},[194,1258,1259],{},"Governance-as-PDF."," Policy. Harness engineering is whether the tool call is reachable. ",[202,1262,1264],{"href":1263},"how-to-evaluate-ai-governance-platforms","How to evaluate AI governance platforms"," is the buying cousin.",[190,1267,1268,1271,1272,250],{},[194,1269,1270],{},"Framework assembly."," Writing LangGraph nodes is building a harness in code. Harness engineering is the ongoing discipline after the graph exists: sensors, ownership, eval. See ",[202,1273,1274],{"href":567},"agent harness vs agent framework",[233,1276,1278],{"id":1277},"four-layers-one-ratchet","Four layers, one ratchet",[190,1280,1281,1285],{},[202,1282,1284],{"href":211,"rel":1283},[206],"Thoughtworks’ July 2026 essay"," is the organisational map most engineering blogs skip. They split enterprise AI into four harness layers. Most companies have built one, maybe two. The gap is not a smarter model.",[190,1287,1288,1291],{},[194,1289,1290],{},"Layer 1 — the model."," Substrate. Choice still matters for cost, residency, and task fit. It is the wrong unit of analysis for a programme. Teams that prototype, hit a failure, and buy the next flagship are looping on layer 1.",[190,1293,1294,1297],{},[194,1295,1296],{},"Layer 2 — the builder harness."," Frameworks, tool access, memory, where inference runs. LangChain, Claude Agent SDK, AIP-style platforms, Nimbus’s hosted loop. Without layer 3, every team invents naming and review. Without layer 4, nobody owns failure.",[190,1299,1300,1303],{},[194,1301,1302],{},"Layer 3 — the user harness."," Guides and sensors on the job. Böckeler’s taxonomy lives here. Thoughtworks add a useful matrix: feed-forward vs feedback, crossed with deterministic vs probabilistic. Deterministic feed-forward is a whitelist and a spend ceiling — cheap, auditable, default. Probabilistic feed-forward is a runbook retrieved at decision time. Deterministic feedback is schema validation after the act. Probabilistic feedback is an eval model on a rubric — expensive, use on critical paths only. A guide with no sensor is theatre.",[190,1305,1306,1309,1310,1313,1314,1317],{},[194,1307,1308],{},"Layer 4 — the organisational harness."," Who may grant which autonomy, escalation, accountability when layers 1–3 all “worked” and the company still took harm. Thoughtworks’ public cases: Parloa, where versioned rules, skills, commands, and helpers lived ",[344,1311,1312],{},"in the repo"," (they report p95 latency drops they attribute to harness architecture, not a new model); Morgan Stanley, where hygiene and CVE triage used a ",[344,1315,1316],{},"delegation tier"," instead of a yes\u002Fno “do we trust the agent.” You do not need those vendors to accept the lesson: governance that is not versioned next to the work decays.",[190,1319,1320],{},"Harness engineering is the steering loop across those layers. Sensor data reveals a miss. Guides update. Hooks graduate. Templates change. The next job is cheaper. An organisation with that loop has a compounding harness. An organisation without one has markdown that rots while models improve.",[190,1322,1323,1324,1326,1327,1329,1330,1332],{},"A concrete week: Monday the agent comments out a flaky test (inner) or proposes Amount without CloseDate (outer). Tuesday a human fixes the artefact. That is operations. Harness engineering is Tuesday’s hook or schema sensor, Wednesday’s wiki or ",[443,1325,445],{}," line, Thursday’s replay that the new control fired. Friday you run the job ten times and count refuses. Nimbus makes the outer version of that week a product surface — ",[202,1328,323],{"href":40}," queues, ",[202,1331,326],{"href":24}," export — so operators are not waiting on a platform sprint to add the sensor. You still have to look at the refuse count. A product without a steering cadence is layer 2 with a nicer UI.",[233,1334,1336],{"id":1335},"what-good-looks-like","What good looks like",[190,1338,1339],{},"Good: a named owner for the harness (not “AI working group”), a cadence that turns incidents into controls, deterministic gates on knowable bounds, inferential checks only where judgement is required, versioned guides, exportable traces. Failure: a new system prompt after every incident; sensors the agent can skip; no owner; SWE-bench as the only score for a CRM job; layer 4 as a PDF.",[190,1341,1342,1346,1347,1349,1350,1353],{},[202,1343,1345],{"href":947,"rel":1344},[206],"Osmani’s ratchet"," and Thoughtworks’ steering loop are the same instinct. ",[202,1348,746],{"href":685}," asks whether your vendor lets you ",[344,1351,1352],{},"run"," that instinct.",[233,1355,691],{"id":690},[693,1357,1359],{"id":1358},"who-coined-harness-engineering","Who coined “harness engineering”?",[190,1361,1362],{},"The phrase circulated in early 2026 across OpenAI engineering notes (Ryan Lopopolo’s line of work), LangChain’s anatomy posts, Böckeler at Thoughtworks, and Osmani’s synthesis. Treat it as a shared 2026 name for work teams were already doing, not a trademarked method.",[693,1364,1366],{"id":1365},"is-this-only-for-coding-agents","Is this only for coding agents?",[190,1368,1369,1370,1373],{},"The literature is densest there because tests already exist. The discipline is the same for RevOps and Finance: independent sensors, fail-closed writes, versioned context. An ",[202,1371,196],{"href":1372},"what-is-an-enterprise-agent-harness"," is that application.",[693,1375,1377],{"id":1376},"do-we-wait-for-a-better-model-instead","Do we wait for a better model instead?",[190,1379,1380,1381,1384],{},"You still buy better models. You do not pause the ratchet. Stronger models attempt larger jobs and fail in new ways. Anthropic’s long-running work exists ",[344,1382,1383],{},"because"," models got good enough to outlast a window.",[693,1386,1388],{"id":1387},"how-do-we-start-this-quarter","How do we start this quarter?",[190,1390,1391,1392,1395],{},"Pick one job that already has a finish line. Encode guides. Attach one deterministic sensor. Add one hook that can refuse. Run it ten times. Every failure updates the harness. That is a ",[202,1393,1394],{"href":735},"proof of value"," for the practice, not a chat demo.",[693,1397,1399],{"id":1398},"how-does-nimbus-fit-without-becoming-the-definition","How does Nimbus fit without becoming the definition?",[190,1401,1402,1403,315,1405,315,1408,315,1411,1413,1414,1416],{},"Nimbus is an outer harness you can hire: ",[202,1404,230],{"href":32},[202,1406,1407],{"href":20},"teams",[202,1409,1410],{"href":40},"gates",[202,1412,326],{"href":24},". Score it the way you score Claude Code: can you add a sensor, refuse a write, and replay who signed. ",[202,1415,746],{"href":685}," is the sheet.",[693,1418,741],{"id":740},[190,1420,1421,1424,1425,1427,1428,1431],{},[202,1422,1423],{"href":219},"Inner vs outer agent harness"," for the repo\u002Fcompany cut. ",[202,1426,751],{"href":750}," for the parts. ",[202,1429,1430],{"href":433},"What is an agent harness"," if you still need the noun.",[233,1433,759],{"id":758},[190,1435,1436,411,1439,250],{},[202,1437,1438],{"href":224},"What is an enterprise AI operating system",[202,1440,1442],{"href":1441},"how-to-solve-ai-that-cannot-write-back-safely","How to solve AI that cannot write back safely",[233,1444,772],{"id":771},[238,1446,1447,1452,1458,1464,1470,1476,1481,1487,1494,1499,1505,1511,1517,1522,1527],{},[241,1448,1449],{},[202,1450,780],{"href":204,"rel":1451},[206],[241,1453,1454],{},[202,1455,1457],{"href":582,"rel":1456},[206],"LangChain, The anatomy of an agent harness",[241,1459,1460],{},[202,1461,1463],{"href":953,"rel":1462},[206],"Böckeler, Harness engineering for coding agent users",[241,1465,1466],{},[202,1467,1469],{"href":1021,"rel":1468},[206],"Thoughtworks, Harness engineering and agent feedback",[241,1471,1472],{},[202,1473,1475],{"href":803,"rel":1474},[206],"Thoughtworks, Scaling the enterprise harness (podcast)",[241,1477,1478],{},[202,1479,798],{"href":211,"rel":1480},[206],[241,1482,1483],{},[202,1484,1486],{"href":947,"rel":1485},[206],"Addy Osmani, Agent harness engineering",[241,1488,1489],{},[202,1490,1493],{"href":1491,"rel":1492},"https:\u002F\u002Fwww.oreilly.com\u002Fradar\u002Fagent-harness-engineering\u002F",[206],"O’Reilly Radar, Agent harness engineering",[241,1495,1496],{},[202,1497,811],{"href":577,"rel":1498},[206],[241,1500,1501],{},[202,1502,1504],{"href":975,"rel":1503},[206],"Anthropic, Effective harnesses for long-running agents",[241,1506,1507],{},[202,1508,1510],{"href":1157,"rel":1509},[206],"Anthropic, Steering Claude Code",[241,1512,1513],{},[202,1514,1516],{"href":1033,"rel":1515},[206],"Claude Code, Hooks",[241,1518,1519],{},[202,1520,817],{"href":339,"rel":1521},[206],[241,1523,1524],{},[202,1525,410],{"href":408,"rel":1526},[206],[241,1528,1529],{},[202,1530,416],{"href":414,"rel":1531},[206],{"title":171,"searchDepth":172,"depth":172,"links":1533},[1534,1535,1536,1537,1538,1539,1540,1548,1549],{"id":235,"depth":172,"text":236},{"id":333,"depth":172,"text":334},{"id":1138,"depth":172,"text":1139},{"id":1228,"depth":172,"text":1229},{"id":1277,"depth":172,"text":1278},{"id":1335,"depth":172,"text":1336},{"id":690,"depth":172,"text":691,"children":1541},[1542,1543,1544,1545,1546,1547],{"id":1358,"depth":881,"text":1359},{"id":1365,"depth":881,"text":1366},{"id":1376,"depth":881,"text":1377},{"id":1387,"depth":881,"text":1388},{"id":1398,"depth":881,"text":1399},{"id":740,"depth":881,"text":741},{"id":758,"depth":172,"text":759},{"id":771,"depth":172,"text":772},"Harness engineering is the 2026 practice of fixing the environment when an agent fails — tools, hooks, tests, and stops — instead of rewriting the prompt and hoping the next model call behaves.","\u002Fblog\u002Fwhat-is-harness-engineering",{"title":923,"description":1550},"blog\u002Fwhat-is-harness-engineering",[893,1555,896,1556],"harness-engineering","evaluation","LzRecSKU7dxHQQ6Y36MmPShlEZizNveN9Zt-cBPNPl0",{"id":1559,"title":1560,"archived":165,"authors":1561,"badge":1563,"body":1564,"date":889,"definedTerm":166,"department":166,"description":2182,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2183,"relatedHeading":166,"seo":2184,"series":893,"sitemap":131,"status":166,"stem":2185,"subhead":166,"tags":2186,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2188},"content\u002Fblog\u002Finner-vs-outer-agent-harness.md","Inner vs Outer Agent Harness",[1562],{"name":184,"to":136},{"label":186},{"type":168,"value":1565,"toc":2165},[1566,1580,1599,1618,1633,1635,1707,1728,1730,1733,1735,1749,1765,1768,1790,1800,1804,1821,1831,1842,1851,1861,1871,1888,1897,1911,1915,1918,1928,1936,1945,1949,1958,1968,1977,1983,1988,1999,2002,2004,2008,2011,2015,2022,2026,2037,2041,2046,2050,2056,2058,2070,2072,2078,2080],[190,1567,192,1568,1571,1572,1575,1576,1579],{},[194,1569,1570],{},"inner agent harness"," is the runtime around a model for a developer and a codebase. An ",[194,1573,1574],{},"outer agent harness"," is the runtime around a model for operators and live business systems. Same equation — ",[202,1577,207],{"href":204,"rel":1578},[206]," — different workspace, different sensors, different stop.",[190,1581,1582,1586,1587,1590,1591,1595,1596,250],{},[202,1583,1585],{"href":953,"rel":1584},[206],"Böckeler"," already uses “outer harness” for the controls ",[344,1588,1589],{},"users"," add around a coding agent (guides, sensors) as distinct from the vendor’s built-in loop. ",[202,1592,949],{"href":1593,"rel":1594},"https:\u002F\u002Faddyosmani.com\u002Fblog\u002Fown-the-outer-loop\u002F",[206]," tells engineers to own the outer loop of investigate → implement → verify so accountability does not dissolve into the model. This article borrows those words and draws the cut enterprises actually buy: ",[194,1597,1598],{},"repo versus company",[190,1600,1601,1602,315,1605,1607,1608,315,1610,1614,1615,1617],{},"Claude Code, Cursor, and Codex are excellent inner harnesses. They sandboxes, ",[443,1603,1604],{},"apply_patch",[443,1606,1007],{}," \u002F ",[443,1609,445],{},[202,1611,1613],{"href":1033,"rel":1612},[206],"hooks",", and tests. Palantir AIP, Salesforce Agentforce, and OS-class products such as Nimbus are outer harnesses: ",[202,1616,230],{"href":229},", connectors, named signers, a decision record. Confusing them is how Legal is asked to “just use Cursor on the Salesforce repo” and how engineering is asked to “approve CRM writes in a coding agent.”",[190,1619,1620,1624,1625,1628,1629,1632],{},[202,1621,1623],{"href":1622},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose between a coding harness and an enterprise harness"," is the buying version of this page. ",[202,1626,1627],{"href":556},"How to choose between a copilot and a work OS"," is the adjacent cut (personal assistant versus departmental work). Inner\u002Fouter is about ",[344,1630,1631],{},"which loop you are hiring",", not whether the UI looks like chat.",[233,1634,236],{"id":235},[238,1636,1637,1643,1649,1659,1671,1680,1693],{},[241,1638,1639,1642],{},[194,1640,1641],{},"Inner loop (classic SE)."," Edit, build, test on a developer’s machine. Fast. Local. The coding-agent inner harness lives here: shell, files, compiler.",[241,1644,1645,1648],{},[194,1646,1647],{},"Outer loop (classic SE)."," PR, CI, review, release. Osmani’s “own the outer loop” is this accountability layer for agentic coding. Still software.",[241,1650,1651,1654,1655,1658],{},[194,1652,1653],{},"Inner harness (this article)."," Vendor + user controls for a ",[194,1656,1657],{},"repository workspace",": Claude Code, Cursor, Codex. Eval: tests, Terminal-Bench, SWE-bench.",[241,1660,1661,1664,1665,1668,1669,250],{},[194,1662,1663],{},"Outer harness (this article)."," Controls for a ",[194,1666,1667],{},"company workspace",": jobs, systems of record, people who may sign. Eval: quoted write, identity, ledger. An ",[202,1670,196],{"href":1372},[241,1672,1673,1676,1677,250],{},[194,1674,1675],{},"Guides vs sensors."," Feed-forward markdown versus feedback from tools. Inner: lint and pytest. Outer: schema of a Salesforce payload and a Hard gate. See ",[202,1678,1679],{"href":754},"what is harness engineering",[241,1681,1682,1685,1686,1689,1690,250],{},[194,1683,1684],{},"CLAUDE.md \u002F AGENTS.md."," Inner guides. ",[202,1687,579],{"href":1157,"rel":1688},[206]," is explicit: files are context; hooks are deterministic. A company wiki is the outer analogue of those files — asserted policy, not a repo README. See ",[202,1691,1692],{"href":453},"What is a company wiki for AI agents",[241,1694,1695,1698,1699,1702,1703,1706],{},[194,1696,1697],{},"Write gate."," Inner: hook denies ",[443,1700,1701],{},"rm"," or force-push. Outer: ",[202,1704,1705],{"href":271},"write-back governance"," — adapter cannot mutate CRM until a named role signs the quote.",[190,1708,1709,1710,1712,1713,315,1715,1718,1719,1721,1722,1712,1724,1727],{},"Nimbus is built as an outer harness: ",[202,1711,314],{"href":28}," instead of only ",[443,1714,445],{},[202,1716,1717],{"href":51},"connectors"," instead of only a local shell, ",[202,1720,323],{"href":40}," instead of only a pre-commit hook, ",[202,1723,23],{"href":24},[443,1725,1726],{},"git log",". Engineering should still run Claude Code. Those products should not share a write path to NetSuite.",[233,1729,334],{"id":333},[190,1731,1732],{},"Demos collapse the cut. Both products answer a question. Both call tools. Both show a transcript. The evaluation is the workspace.",[190,1734,359],{},[238,1736,1737,1740,1743,1746],{},[241,1738,1739],{},"Security asks whether the coding agent’s MCP server can reach production Salesforce",[241,1741,1742],{},"RevOps wants “an agent” and is shown a SWE-bench slide",[241,1744,1745],{},"Engineering wants Cursor and is told to wait for the enterprise OS",[241,1747,1748],{},"You already have both, and they silently write to the same object",[190,1750,1751,1754,1755,1760,1761,1764],{},[202,1752,341],{"href":339,"rel":1753},[206]," describes agentic systems as an organisational design problem. Inner harnesses scale developer throughput. They do not, by themselves, scale governed operations. ",[202,1756,1759],{"href":1757,"rel":1758},"https:\u002F\u002Fhai.stanford.edu\u002Fai-index\u002F2025-ai-index-report",[206],"Stanford HAI’s 2025 AI Index"," maps how fast coding-agent tooling moved. Speed in the repo is not a substitute for ",[202,1762,402],{"href":400,"rel":1763},[206]," oversight on systems that affect customers and money.",[190,1766,1767],{},"Two failure modes:",[1769,1770,1771,1781],"ol",{},[241,1772,1773,1776,1777,1780],{},[194,1774,1775],{},"Outer job, inner harness."," A pricing change drafted in Cursor with an MCP Salesforce tool. Tests pass on a fixture. Production Amount changes. ",[443,1778,1779],{},"git blame"," does not name the signer. You used a repo loop on a company record.",[241,1782,1783,1786,1787,1789],{},[194,1784,1785],{},"Inner job, outer harness."," “Rewrite this function” opened as a cross-department ",[202,1788,664],{"href":32}," with a Critical gate. Engineers will route around it. You used a company loop on a compile.",[190,1791,1792,1795,1796,1799],{},[202,1793,410],{"href":408,"rel":1794},[206]," Map step: know the context of use. Inner and outer are different contexts. ",[202,1797,416],{"href":414,"rel":1798},[206]," wants controls matched to that context. One harness policy for “all AI” is how both jobs get the wrong stop.",[233,1801,1803],{"id":1802},"what-each-harness-actually-owns","What each harness actually owns",[190,1805,1806,1809,1810,1814,1815,1817,1818,1820],{},[194,1807,1808],{},"Workspace."," Inner: a checkout, often sandboxed. Anthropic’s ",[202,1811,1813],{"href":975,"rel":1812},[206],"long-running harness"," keeps progress in git and files because the workspace ",[344,1816,1113],{}," the filesystem. Outer: a job folder with people, budget, and attached systems — a ",[202,1819,664],{"href":229},". Files may appeartefacts. They are not the system of record.",[190,1822,1823,1826,1827,1830],{},[194,1824,1825],{},"Identity."," Inner: the developer’s machine credentials, a repo token, maybe a sandbox role. Outer: org roster, workstream membership, named approver. The model is not the principal. ",[202,1828,1829],{"href":463},"Connector and permissions architecture"," is the outer identity plane.",[190,1832,1833,1836,1837,1841],{},[194,1834,1835],{},"Tools."," Inner: shell, editor, tests, browser, maybe MCP to docs. Outer: CRM, ERP, warehouse, ticket systems, mail — default read, write as a separate plane. ",[202,1838,1840],{"href":1839},"what-is-model-context-protocol","MCP"," can sit under both. The grant must not.",[190,1843,1844,1847,1848,1850],{},[194,1845,1846],{},"Guides."," Inner: ",[443,1849,445],{},", skills, directory-local rules. Outer: company wiki, playbooks versioned with the run. Mixing them is useful (engineering conventions in the repo; discount policy in the wiki). Collapsing them is how a style guide becomes “legal approval.”",[190,1852,1853,1856,1857,1860],{},[194,1854,1855],{},"Sensors."," Inner: typechecker, unit tests, CI, architecture tests. Böckeler and ",[202,1858,1023],{"href":1021,"rel":1859},[206],". Outer: payload schema, blast-radius cardinality, maker-checker, exportable ledger. A passing pytest does not mean Opportunity.Stage was authorised.",[190,1862,1863,1866,1867,1870],{},[194,1864,1865],{},"Stop."," Inner: tests red, hook exit 2, max steps, human in the IDE. Outer: wait-for-named-signer, missing connector, budget, reject. ",[202,1868,1869],{"href":498},"Human-in-the-loop"," in a coding agent is “the developer kept going.” HITL in an outer harness is a first-class step with identity.",[190,1872,1873,1847,1876,315,1879,1884,1885,1887],{},[194,1874,1875],{},"Eval.",[202,1877,868],{"href":866,"rel":1878},[206],[202,1880,1883],{"href":1881,"rel":1882},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.11868",[206],"Terminal-Bench",", your suite. Outer: replay the signer; compare quote to SoR; see ",[202,1886,524],{"href":297},". Leaderboard scores are not a SOX control.",[190,1889,1890,1893,1894,1896],{},[194,1891,1892],{},"Memory."," Inner: files, commits, session transcripts, memory files the next coding session loads. Outer: wiki + ",[202,1895,23],{"href":287}," so next quarter’s operator can ask why a field changed. Chat logs of a coding session are not institutional memory for RevOps.",[190,1898,1899,1900,1902,1903,1905,1906,1910],{},"Nimbus’s ",[202,1901,320],{"href":20}," sit on the outer side: mandates, required connectors, approval triggers. You can still ",[344,1904,346],{}," an inner harness as a bounded tool behind a connector (for example a coding agent that only opens a draft PR). Do not let that inner harness become the orchestrator of record for a CRM write. ",[202,1907,1909],{"href":1908},"multi-agent-ai-architecture","Multi-agent architecture"," says the same thing with specialists: hands are not roles.",[233,1912,1914],{"id":1913},"how-they-should-sit-together","How they should sit together",[190,1916,1917],{},"Most companies need both. That is not a hedge. It is how software and operations already split.",[190,1919,1920,1923,1924,1927],{},[194,1921,1922],{},"Pattern that works."," Engineers use Cursor or Claude Code on application repos. CI remains the merge sensor. Separately, RevOps and Finance run outer-harness jobs on Salesforce and NetSuite. If a coding agent must touch a live business system, it proposes an artefact; the outer harness quotes and gates the write. Two writers to the same object without a single quote is the failure ",[202,1925,1926],{"href":1908},"multi-agent architecture"," already names.",[190,1929,1930,1933,1934,250],{},[194,1931,1932],{},"Pattern that fails."," One MCP mesh with production tokens, used from the IDE and from the chatbot and from the OS. Confused deputy. ",[202,1935,469],{"href":468},[190,1937,1938,1941,1942,250],{},[194,1939,1940],{},"Thoughtworks’ four layers"," — model, builder harness, user harness, organisational harness — map cleanly: Claude Code is builder + user on the inner side; the organisational layer is the outer operating model. Nimbus is one productisation of that outer layer, not the only one. AIP is a programme-shaped outer harness. Agentforce is CRM-anchored. Score scope and time-to-value separately. See ",[202,1943,1944],{"href":307},"self-service vs forward-deployed",[233,1946,1948],{"id":1947},"a-week-that-uses-both","A week that uses both",[190,1950,1951,1952,1954,1955,1957],{},"Monday an engineer uses Cursor to fix a pricing calculator in the billing service. ",[443,1953,445],{}," says no raw SQL in the request path. A hook blocks ",[443,1956,1148],{},". CI runs the unit suite. The PR is the artefact. CODEOWNERS signs the merge. That is a complete inner story. SWE-bench is relevant only as a vendor quality signal for the coding tool, not as a control.",[190,1959,1960,1961,1963,1964,1967],{},"Tuesday RevOps needs the list price on twenty renewals updated after Legal changed the cap in the playbook. The artefact is Salesforce. The signer is a named RevOps lead. The sensor is: quoted fields, hash, read-back. If Tuesday’s job is opened as a Cursor session with an MCP Salesforce server using a shared integration user, you have imported Monday’s workspace into Tuesday’s system of record. ",[443,1962,1726],{}," will not name the RevOps lead. ",[202,1965,1966],{"href":271},"Write-back"," did not fire because the inner harness does not have that interceptor.",[190,1969,1970,1971,1973,1974,1976],{},"Wednesday someone proposes “one agent for everything.” The honest architecture is: Monday’s harness stays. Tuesday’s job runs on an outer harness — in Nimbus, a ",[202,1972,664],{"href":32}," with the CRM connector, the wiki revision that contains the new cap, a Hard gate. If the calculator ",[344,1975,443],{}," must change as well, the outer job can spawn a bounded inner step that opens a draft PR. Two artefacts, two sensors, one company rule: unsigned SoR writes are impossible from either loop.",[190,1978,1979,1980,250],{},"Thursday Security reviews MCP. The question is not “is MCP approved.” It is “which workspace may this server mutate.” Inner: sandbox and repo. Outer: workstream grant. Same protocol, different identity box. ",[202,1981,1982],{"href":468},"MCP for enterprise",[190,1984,1985,1986,250],{},"Friday you look at evals. Engineering posts a Terminal-Bench plot for the coding vendor. Finance asks who signed Amount. Those are not competing dashboards. They are different oracles. ",[202,1987,298],{"href":297},[190,1989,1990,1993,1994,1998],{},[202,1991,213],{"href":211,"rel":1992},[206]," would call Monday layers 2–3 on a builder harness, Tuesday a delegation question on layer 4, and “one agent” a way to skip layer 4. ",[202,1995,1997],{"href":1593,"rel":1996},[206],"Osmani"," would say engineering still owns verify-and-merge on Monday. Neither author is selling Nimbus. Both are describing why the cut exists.",[190,2000,2001],{},"If you only fund inner harnesses, Tuesday happens in paste and Slack. If you only fund outer harnesses, Monday happens in unsanctioned Cursor anyway. Fund both. Bind writes.",[233,2003,691],{"id":690},[693,2005,2007],{"id":2006},"is-cursor-an-enterprise-harness-if-we-sso-it","Is Cursor an enterprise harness if we SSO it?",[190,2009,2010],{},"SSO is admin control. It does not quote a NetSuite journal or bind a Finance signer. Cursor can be an inner harness in an enterprise. That is not the same as an outer harness.",[693,2012,2014],{"id":2013},"can-claude-code-hooks-replace-write-back-governance","Can Claude Code hooks replace write-back governance?",[190,2016,2017,2018,2021],{},"They can replace ",[344,2019,2020],{},"some"," inner invariants (dangerous bash). They do not give you a payload in the language of Salesforce, a roster-aware approver, or an exportable operations ledger. Different workspace.",[693,2023,2025],{"id":2024},"should-we-ban-coding-agents-until-the-os-is-live","Should we ban coding agents until the OS is live?",[190,2027,2028,2029,2032,2033,250],{},"Usually no. Ban unsigned writes to systems of record from ",[344,2030,2031],{},"any"," agent, inner or outer. Let inner harnesses keep compiling. ",[202,2034,2036],{"href":2035},"how-to-solve-unapproved-crm-writes-from-ai","How to solve unapproved CRM writes from AI",[693,2038,2040],{"id":2039},"where-does-a-copilot-fit","Where does a copilot fit?",[190,2042,2043,2044,250],{},"A copilot is often not a full inner harness — no repo loop, no tests. Personal throughput. Keep it for mail. Do not give it the CRM write token. ",[202,2045,557],{"href":556},[693,2047,2049],{"id":2048},"is-nimbus-trying-to-replace-claude-code","Is Nimbus trying to replace Claude Code?",[190,2051,2052,2053,2055],{},"No. Different workspace. Nimbus is the company loop; Claude Code is the repo loop. ",[202,2054,11],{"href":12}," is the product map. This page is the architectural cut.",[693,2057,741],{"id":740},[190,2059,2060,747,2063,2066,2067,2069],{},[202,2061,2062],{"href":1372},"What is an enterprise agent harness",[202,2064,2065],{"href":567},"Agent harness vs agent framework"," if you are assembling rather than hiring. ",[202,2068,1430],{"href":433}," for the base noun.",[233,2071,759],{"id":758},[190,2073,2074,411,2076,250],{},[202,2075,755],{"href":754},[202,2077,751],{"href":750},[233,2079,772],{"id":771},[238,2081,2082,2087,2092,2098,2103,2108,2113,2118,2123,2128,2134,2139,2145,2150,2155,2160],{},[241,2083,2084],{},[202,2085,780],{"href":204,"rel":2086},[206],[241,2088,2089],{},[202,2090,1463],{"href":953,"rel":2091},[206],[241,2093,2094],{},[202,2095,2097],{"href":1593,"rel":2096},[206],"Addy Osmani, Own the outer loop",[241,2099,2100],{},[202,2101,1486],{"href":947,"rel":2102},[206],[241,2104,2105],{},[202,2106,798],{"href":211,"rel":2107},[206],[241,2109,2110],{},[202,2111,1504],{"href":975,"rel":2112},[206],[241,2114,2115],{},[202,2116,1510],{"href":1157,"rel":2117},[206],[241,2119,2120],{},[202,2121,1516],{"href":1033,"rel":2122},[206],[241,2124,2125],{},[202,2126,868],{"href":866,"rel":2127},[206],[241,2129,2130],{},[202,2131,2133],{"href":1881,"rel":2132},[206],"Terminal-Bench (arXiv:2601.11868)",[241,2135,2136],{},[202,2137,817],{"href":339,"rel":2138},[206],[241,2140,2141],{},[202,2142,2144],{"href":1757,"rel":2143},[206],"Stanford HAI, 2025 AI Index",[241,2146,2147],{},[202,2148,410],{"href":408,"rel":2149},[206],[241,2151,2152],{},[202,2153,416],{"href":414,"rel":2154},[206],[241,2156,2157],{},[202,2158,402],{"href":400,"rel":2159},[206],[241,2161,2162],{},[202,2163,861],{"href":859,"rel":2164},[206],{"title":171,"searchDepth":172,"depth":172,"links":2166},[2167,2168,2169,2170,2171,2172,2180,2181],{"id":235,"depth":172,"text":236},{"id":333,"depth":172,"text":334},{"id":1802,"depth":172,"text":1803},{"id":1913,"depth":172,"text":1914},{"id":1947,"depth":172,"text":1948},{"id":690,"depth":172,"text":691,"children":2173},[2174,2175,2176,2177,2178,2179],{"id":2006,"depth":881,"text":2007},{"id":2013,"depth":881,"text":2014},{"id":2024,"depth":881,"text":2025},{"id":2039,"depth":881,"text":2040},{"id":2048,"depth":881,"text":2049},{"id":740,"depth":881,"text":741},{"id":758,"depth":172,"text":759},{"id":771,"depth":172,"text":772},"An inner agent harness runs a developer and a repository — CLAUDE.md, hooks, tests. An outer harness runs the company — wiki, connectors, write gates, and a ledger. Most enterprises need both.","\u002Fblog\u002Finner-vs-outer-agent-harness",{"title":1560,"description":2182},"blog\u002Finner-vs-outer-agent-harness",[893,896,2187,897],"coding-agents","4SP_cgHXWiZDlDtRNzYXDjhIozAbo0cYRkC1ey6jz_k",{"enabled":165,"message":2190,"linkLabel":79,"linkHref":80,"id":2191,"title":2192,"archived":165,"authors":166,"badge":166,"body":2193,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2197,"relatedHeading":166,"seo":2198,"series":166,"sitemap":165,"status":166,"stem":2199,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2200},"We're hiring! Join the team building the Sentient Enterprise.","content\u002Fshared\u002Fhiring.md","Hiring banner",{"type":168,"value":2194,"toc":2195},[],{"title":171,"searchDepth":172,"depth":172,"links":2196},[],"\u002Fshared\u002Fhiring",{"title":2192,"description":171},"shared\u002Fhiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",{"fold":2202,"id":2206,"title":2207,"archived":165,"authors":166,"badge":166,"body":2208,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":2212,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2216,"relatedHeading":166,"seo":2217,"series":166,"sitemap":165,"status":166,"stem":2218,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2219},{"headline":2203,"description":2204,"primaryLabel":8,"primaryTo":2205,"secondaryLabel":915,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Start on Free.","\u002Fsignup?plan=free","content\u002Fshared\u002Fcta.md","Site CTAs",{"type":168,"value":2209,"toc":2210},[],{"title":171,"searchDepth":172,"depth":172,"links":2211},[],{"headline":2213,"description":2214,"primaryLabel":8,"primaryTo":2205,"secondaryLabel":2215,"secondaryTo":85},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger. Start on Free.","Talk to our team","\u002Fshared\u002Fcta",{"title":2207,"description":171},"shared\u002Fcta","PS2VPJsszmUpMBZT6nEp8cWXCdeiN6zDRl-p8d0uY2k",1790215700881]