[{"data":1,"prerenderedAt":1820},["ShallowReactive",2],{"site-nav-content":3,"blog:/blog/what-is-harness-engineering":178,"blog-index-copy":889,"blog:/blog/what-is-harness-engineering:surround":910,"hiring-banner-content":1789,"site-cta-content":1801},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":131,"id":162,"title":163,"archived":164,"authors":165,"badge":165,"body":166,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":174,"relatedHeading":165,"seo":175,"series":165,"sitemap":164,"status":165,"stem":176,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":177},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","/overview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","/product/conflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","/product/agent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","/product/lifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","/product/wiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","/product/workstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","/product/perception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","/product/governance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","/models",{"label":47,"to":48},"Pricing","/pricing",{"label":50,"to":51},"Integrations","/integrations",{"label":53,"to":54},"Security","/security",{"label":56,"to":57},"Partners","/partners",{"label":59,"to":60},"Insights","/blog",{"productHeading":5,"companyHeading":62,"legalHeading":63,"docsLabel":64,"docsUrl":65,"statementLines":66,"copyright":69,"companyLinks":70,"legalLinks":100,"socialLinks":110,"bottomLinks":120},"Company","Legal","Docs","https://docs.gonimbus.ai",[67,68],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[71,72,73,74,75,78,81,84,87,90,92,95,98],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":76,"to":77},"Glossary","/glossary",{"label":79,"to":80},"Compare","/compare",{"label":82,"to":83},"Evaluate","/evaluate",{"label":85,"to":86},"Problems","/problems",{"label":88,"to":89},"Use cases","/use-cases",{"label":91,"to":57},"Partner Program",{"label":93,"to":94},"Careers","/careers",{"label":96,"to":97},"System status","/status",{"label":7,"to":99},"/contact",[101,104,107],{"label":102,"to":103},"Terms of Service","/terms",{"label":105,"to":106},"Privacy Policy","/privacy",{"label":108,"to":109},"Compliance","/compliance",[111,114,117],{"label":112,"href":113},"LinkedIn","https://www.linkedin.com/company/gonimbusai/",{"label":115,"href":116},"X","https://x.com/gonimbusai",{"label":118,"href":119},"Instagram","https://www.instagram.com/gonimbus_ai/",[121,123,125,126,127],{"label":122,"to":103},"Terms",{"label":124,"to":106},"Privacy",{"label":108,"to":109},{"label":96,"to":97},{"label":128,"to":129,"external":130},"LLMs.txt","/llms.txt",true,{"text":132,"prompt":133},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":134,"platforms":136},{" Summarize the highlights from Nimbus's website":135},"https://gonimbus.ai",[137,142,147,152,157],{"name":138,"label":139,"icon":140,"hrefPrefix":141},"chatgpt","ChatGPT","simple-icons:openai","https://chatgpt.com/?prompt=",{"name":143,"label":144,"icon":145,"hrefPrefix":146},"perplexity","Perplexity","mdi:magnify","https://www.perplexity.ai/search/new?q=",{"name":148,"label":149,"icon":150,"hrefPrefix":151},"grok","Grok","simple-icons:x","https://x.com/i/grok?text=",{"name":153,"label":154,"icon":155,"hrefPrefix":156},"claude","Claude","simple-icons:anthropic","https://claude.ai/new?q=",{"name":158,"label":159,"icon":160,"hrefPrefix":161},"google-ai","Google AI","simple-icons:google","https://www.google.com/search?udm=50&aep=11&q=","content/shared/nav.md","Site navigation",false,null,{"type":167,"value":168,"toc":169},"minimark",[],{"title":170,"searchDepth":171,"depth":171,"links":172},"",2,[],"md","/shared/nav",{"title":163,"description":170},"shared/nav","rDEv5cVG6P2l9ATcdQv2n6VOQiSL5ioOutyfvGevcD0",{"id":179,"title":180,"archived":164,"authors":181,"badge":184,"body":186,"date":878,"definedTerm":165,"department":165,"description":879,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":880,"relatedHeading":165,"seo":881,"series":882,"sitemap":130,"status":165,"stem":883,"subhead":165,"tags":884,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":888},"content/blog/what-is-harness-engineering.md","What is Harness Engineering",[182],{"name":183,"to":135},"Nimbus Research",{"label":185},"Explainer",{"type":167,"value":187,"toc":859},[188,196,229,232,251,256,349,355,359,368,371,391,409,426,429,433,443,465,478,492,508,524,527,531,537,543,556,567,577,581,588,594,600,606,620,623,636,640,643,659,663,668,671,675,683,687,694,698,706,710,727,731,748,752,762,766],[189,190,191,195],"p",{},[192,193,194],"strong",{},"Harness engineering"," is the practice of treating the runtime around a model as the system you design, test, and tighten — so that when an agent fails, you change the environment, not only the prompt.",[189,197,198,205,206,210,211,216,217,222,223,228],{},[199,200,204],"a",{"href":201,"rel":202},"https://docs.langchain.com/oss/python/langchain/agents",[203],"nofollow","LangChain"," defines the object: Agent = Model + Harness. Harness engineering is what you ",[207,208,209],"em",{},"do"," to that object. ",[199,212,215],{"href":213,"rel":214},"https://addyosmani.com/blog/agent-harness-engineering/",[203],"Addy Osmani"," puts the payoff in one line: a decent model with a great harness beats a great model with a bad harness. ",[199,218,221],{"href":219,"rel":220},"https://martinfowler.com/articles/harness-engineering.html",[203],"Birgitta Böckeler’s article on martinfowler.com"," is the user’s-side map for coding agents: guides in, sensors back. Thoughtworks then asked the organisational question: ",[199,224,227],{"href":225,"rel":226},"https://www.thoughtworks.com/insights/podcasts/technology-podcasts/scaling-the-enterprise-harness--how-to-achieve-ai-agent-controll",[203],"how you scale that harness across a company"," without turning every team into a snowflake of markdown files.",[189,230,231],{},"The practice showed up because prompt engineering hit a wall that everyone could see and nobody wanted to name. You can spend a week on a system prompt. The agent will still skip the test, ignore the style guide, or report the task finished. The model is non-deterministic. The prompt is interpreted, not executed. The harness is code. That is the whole discipline.",[189,233,234,235,240,241,246,247,250],{},"This is not a replacement for ",[199,236,239],{"href":237,"rel":238},"https://www.anthropic.com/engineering/building-effective-agents",[203],"prompt"," or ",[199,242,245],{"href":243,"rel":244},"https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents",[203],"context"," work. Those live ",[207,248,249],{},"inside"," the harness. Harness engineering is the wider loop: every failure becomes a rule, a hook, a test, or a denied tool — the ratchet Osmani describes — so the same mistake is cheaper the second time and impossible the tenth.",[252,253,255],"h2",{"id":254},"words-youll-hear","Words you’ll hear",[257,258,259,266,290,302,314,320,326,338],"ul",{},[260,261,262,265],"li",{},[192,263,264],{},"Ratchet."," A failure updates the harness. Commented-out test → pre-commit hook and a reviewer check. Invented CRM field → schema quote and a Hard gate. If you only fix the artefact by hand, you did operations. You did not do harness engineering.",[260,267,268,271,272,275,276,280,281,284,285,289],{},[192,269,270],{},"Guides (feed-forward)."," Context the agent gets ",[207,273,274],{},"before"," it acts: ",[277,278,279],"code",{},"AGENTS.md",", ",[277,282,283],{},"CLAUDE.md",", architecture notes, ",[199,286,288],{"href":287},"what-is-a-company-wiki-for-ai-agents","company wiki"," playbooks. Böckeler’s term. Advice. Necessary. Not a stop.",[260,291,292,295,296,301],{},[192,293,294],{},"Sensors (feedback)."," Deterministic checks (compiler, linter, schema, pytest) and inferential checks (LLM reviewer, specialist critic). ",[199,297,300],{"href":298,"rel":299},"https://www.thoughtworks.com/en-us/insights/blog/generative-ai/harness-engineering-agent-feedback-exploring-ai-coding-sensors",[203],"Thoughtworks on sensors",". Without sensors the agent grades its own homework.",[260,303,304,307,308,313],{},[192,305,306],{},"Hooks."," Lifecycle intercepts that always run. ",[199,309,312],{"href":310,"rel":311},"https://code.claude.com/docs/en/hooks",[203],"Claude Code"," can block a tool with exit code 2. LangChain middleware is the library form. A guide that says “never run rm -rf” is not a hook.",[260,315,316,319],{},[192,317,318],{},"Harness-as-a-service."," Osmani’s HaaS framing: you used to build on completion APIs; you now build on runtime APIs (Claude Agent SDK, Codex SDK, OpenAI Agents SDK) that already own the loop, sandbox, and hooks. You configure; you do not re-implement ReAct.",[260,321,322,325],{},[192,323,324],{},"Skill issue."," HumanLayer’s joke with a serious edge: most agent failures are configuration. Blaming the model first is how teams wait for the next release instead of adding a sensor.",[260,327,328,331,332,337],{},[192,329,330],{},"Organizational harness."," ",[199,333,336],{"href":334,"rel":335},"https://www.thoughtworks.com/insights/articles/operating-system-enterprise-ai",[203],"Thoughtworks’ enterprise layer",": who may build which harness, how exceptions work, identity, economics, learning. The gap after builder harnesses (Claude Code, Cursor) and user harnesses (guides and sensors on a repo).",[260,339,340,343,344,348],{},[192,341,342],{},"Eval loop."," Independent verification that does not take the model’s word. SWE-bench and Terminal-Bench for code. Quoted payload vs executed write for operations. See ",[199,345,347],{"href":346},"eval-loops-for-enterprise-agent-harnesses","eval loops for enterprise agent harnesses",".",[189,350,351,352,354],{},"In Nimbus, harness engineering for operators looks like: wiki revisions as guides, connector scopes as tool policy, Soft / Hard / Critical as hooks on the write plane, and the ",[199,353,23],{"href":24}," as the sensor log you can query. That is the same discipline as adding a linter. The artefact is a signed CRM change rather than a green CI job.",[252,356,358],{"id":357},"why-you-should-care","Why you should care",[189,360,361,362,367],{},"If you only tune prompts, every incident is a conversation. If you engineer the harness, incidents become tests. ",[199,363,366],{"href":364,"rel":365},"https://www.nist.gov/itl/ai-risk-management-framework",[203],"NIST’s AI RMF"," Measure and Manage steps assume you can change controls after you observe harm. A prompt history is not a control change. A hook that now fires is.",[189,369,370],{},"It affects you if:",[257,372,373,376,382,385,388],{},[260,374,375],{},"agents already write code or propose writes to live systems",[260,377,378,379,381],{},"two teams have two ",[277,380,283],{}," files that contradict Legal",[260,383,384],{},"you cannot say which harness version ran last Tuesday",[260,386,387],{},"spend is “the model was verbose” rather than “the loop had no budget”",[260,389,390],{},"auditors ask who could have stopped the action, and the answer is “the model was supposed to ask”",[189,392,393,398,399,402,403,408],{},[199,394,397],{"href":395,"rel":396},"https://www.mckinsey.com/capabilities/quantumblack/our-insights/the-state-of-ai",[203],"McKinsey’s 2025 State of AI"," keeps showing usage without redesign. Harness engineering ",[207,400,401],{},"is"," the redesign for agentic work: not a new department named AI, a runtime with stops. ",[199,404,407],{"href":405,"rel":406},"https://www.iso.org/standard/42001",[203],"ISO/IEC 42001"," wants named AI actors and documented operational controls. You cannot name actors if every operator’s personal GPT is a different harness.",[189,410,411,412,415,416,420,421,425],{},"Coding teams already have half of this and do not always notice. Types, tests, CI, CODEOWNERS — Böckeler’s point is that those ",[207,413,414],{},"are"," sensors. The work is to point the agent at them and to add the ones that are missing (architecture fitness, behaviour: did it do what was asked). Operations teams usually have the human version — maker-checker, SoD, SOX — and have not yet wired those instincts into a loop. ",[199,417,419],{"href":418},"what-is-write-back-governance","Write-back governance"," is that wiring. ",[199,422,424],{"href":423},"human-in-the-loop-approval-architecture","Human-in-the-loop approval architecture"," is the state machine.",[189,427,428],{},"Air Canada’s chatbot and the sanctioned ChatGPT brief are what happens when generation reaches a system of record with no ratchet. The fix is not a sterner system prompt. The fix is a harness that cannot emit a commitment or a filing until a named person has seen the artefact.",[252,430,432],{"id":431},"the-practice-not-the-slogan","The practice, not the slogan",[189,434,435,438,439,442],{},[192,436,437],{},"1. Work backward from the behaviour you cannot afford to miss once."," Inner loop: never merge without tests; never ",[277,440,441],{},"git push --force"," to main. Outer loop: never PATCH Opportunity.Amount without a Hard quote. Write those as hooks, not as paragraphs.",[189,444,445,331,448,453,454,456,457,460,461,464],{},[192,446,447],{},"2. Separate advice from invariants.",[199,449,452],{"href":450,"rel":451},"https://claude.com/blog/steering-claude-code-skills-hooks-rules-subagents-and-more",[203],"Anthropic’s steering note for Claude Code"," is unusually clear: ",[277,455,283],{}," is always-on context; hooks fire on events and can block. If a rule must hold when the model is tired, it graduates from markdown to a hook. Enterprise equivalent: playbooks in the ",[199,458,459],{"href":28},"wiki"," versus the interceptor in ",[199,462,463],{"href":40},"governance",". If they conflict, the interceptor wins.",[189,466,467,470,471,474,475,348],{},[192,468,469],{},"3. Put verification outside the generator."," Anthropic’s long-running harness uses incremental commits and end-to-end checks so later sessions cannot declare victory by vibes. Coding sensors: pytest, tsc, lint. Enterprise sensors: schema of the quote, identity of the signer, hash of the payload that executed, connector grant still attached. The model may ",[207,472,473],{},"propose"," that it is done. The harness ",[207,476,477],{},"decides",[189,479,480,483,484,486,487,491],{},[192,481,482],{},"4. Version the harness."," Which ",[277,485,279],{},", which wiki revision, which team contract, which approval tier ran. ",[199,488,490],{"href":489},"what-is-an-agentic-workflow","What is an agentic workflow"," already treats workflow version as an input. Harness engineering extends that to tools and gates. Hot-patching production prompts without a change record is how Tuesday becomes unexplained.",[189,493,494,497,498,502,503,507],{},[192,495,496],{},"5. Budget the loop."," Max steps and a cost cap that do not depend on the model’s judgement. Seat licences hide this; metered work makes it visible. See ",[199,499,501],{"href":500},"what-is-model-routing","What is model routing"," and ",[199,504,506],{"href":505},"ai-cost-control-architecture","AI cost control architecture",". Always-flagship is not careful. It is an unengineered harness.",[189,509,510,513,514,518,519,523],{},[192,511,512],{},"6. Do not fork a harness per person."," User-owned bots are how mandates drift. Org-level ",[199,515,517],{"href":516},"agent-team-architecture","agent teams"," assigned to ",[199,520,522],{"href":521},"what-is-an-ai-workstream","workstreams"," is the enterprise form of “one CI config per repo, not one per intern.” Thoughtworks’ organisational harness is this ownership question: who is allowed to add a write tool.",[189,525,526],{},"Nimbus encodes several of these as product defaults — read-only connectors until you enable write, quoted payloads, graph on the way out — because operators should not have to re-implement ReAct to get a ratchet. You can still fail the practice: a wiki that is never updated, a Critical tier nobody uses, a graph nobody queries. The product is not the practice. The practice is whether last month’s incident produced a new gate.",[252,528,530],{"id":529},"how-this-differs-from-adjacent-crafts","How this differs from adjacent crafts",[189,532,533,536],{},[192,534,535],{},"Prompt engineering"," improves a single call. Necessary for tone, tool descriptions, and “what good looks like.” Insufficient for tool dispatch, identity, and replay.",[189,538,539,542],{},[192,540,541],{},"Context engineering"," governs what the model sees this turn: compaction, retrieval, files. Anthropic’s initializer agent is context engineering in a harness. It is not permission to write NetSuite.",[189,544,545,548,549,552,553,348],{},[192,546,547],{},"Platform / DevOps."," CI, sandboxes, secrets. Harness engineering ",[207,550,551],{},"reuses"," those as sensors and execution environments. It adds the fact that the component in the loop is non-deterministic, so “the job returned zero” is not enough: you need independent tests of the ",[207,554,555],{},"claim",[189,557,558,561,562,566],{},[192,559,560],{},"Governance-as-PDF."," Policy. Harness engineering is whether the tool call is reachable. ",[199,563,565],{"href":564},"how-to-evaluate-ai-governance-platforms","How to evaluate AI governance platforms"," is the buying cousin.",[189,568,569,572,573,348],{},[192,570,571],{},"Framework assembly."," Writing LangGraph nodes is building a harness in code. Harness engineering is the ongoing discipline after the graph exists: sensors, ownership, eval. See ",[199,574,576],{"href":575},"agent-harness-vs-agent-framework","agent harness vs agent framework",[252,578,580],{"id":579},"four-layers-one-ratchet","Four layers, one ratchet",[189,582,583,587],{},[199,584,586],{"href":334,"rel":585},[203],"Thoughtworks’ July 2026 essay"," is the organisational map most engineering blogs skip. They split enterprise AI into four harness layers. Most companies have built one, maybe two. The gap is not a smarter model.",[189,589,590,593],{},[192,591,592],{},"Layer 1 — the model."," Substrate. Choice still matters for cost, residency, and task fit. It is the wrong unit of analysis for a programme. Teams that prototype, hit a failure, and buy the next flagship are looping on layer 1.",[189,595,596,599],{},[192,597,598],{},"Layer 2 — the builder harness."," Frameworks, tool access, memory, where inference runs. LangChain, Claude Agent SDK, AIP-style platforms, Nimbus’s hosted loop. Without layer 3, every team invents naming and review. Without layer 4, nobody owns failure.",[189,601,602,605],{},[192,603,604],{},"Layer 3 — the user harness."," Guides and sensors on the job. Böckeler’s taxonomy lives here. Thoughtworks add a useful matrix: feed-forward vs feedback, crossed with deterministic vs probabilistic. Deterministic feed-forward is a whitelist and a spend ceiling — cheap, auditable, default. Probabilistic feed-forward is a runbook retrieved at decision time. Deterministic feedback is schema validation after the act. Probabilistic feedback is an eval model on a rubric — expensive, use on critical paths only. A guide with no sensor is theatre.",[189,607,608,611,612,615,616,619],{},[192,609,610],{},"Layer 4 — the organisational harness."," Who may grant which autonomy, escalation, accountability when layers 1–3 all “worked” and the company still took harm. Thoughtworks’ public cases: Parloa, where versioned rules, skills, commands, and helpers lived ",[207,613,614],{},"in the repo"," (they report p95 latency drops they attribute to harness architecture, not a new model); Morgan Stanley, where hygiene and CVE triage used a ",[207,617,618],{},"delegation tier"," instead of a yes/no “do we trust the agent.” You do not need those vendors to accept the lesson: governance that is not versioned next to the work decays.",[189,621,622],{},"Harness engineering is the steering loop across those layers. Sensor data reveals a miss. Guides update. Hooks graduate. Templates change. The next job is cheaper. An organisation with that loop has a compounding harness. An organisation without one has markdown that rots while models improve.",[189,624,625,626,628,629,631,632,635],{},"A concrete week: Monday the agent comments out a flaky test (inner) or proposes Amount without CloseDate (outer). Tuesday a human fixes the artefact. That is operations. Harness engineering is Tuesday’s hook or schema sensor, Wednesday’s wiki or ",[277,627,279],{}," line, Thursday’s replay that the new control fired. Friday you run the job ten times and count refuses. Nimbus makes the outer version of that week a product surface — ",[199,630,463],{"href":40}," queues, ",[199,633,634],{"href":24},"graph"," export — so operators are not waiting on a platform sprint to add the sensor. You still have to look at the refuse count. A product without a steering cadence is layer 2 with a nicer UI.",[252,637,639],{"id":638},"what-good-looks-like","What good looks like",[189,641,642],{},"Good: a named owner for the harness (not “AI working group”), a cadence that turns incidents into controls, deterministic gates on knowable bounds, inferential checks only where judgement is required, versioned guides, exportable traces. Failure: a new system prompt after every incident; sensors the agent can skip; no owner; SWE-bench as the only score for a CRM job; layer 4 as a PDF.",[189,644,645,649,650,654,655,658],{},[199,646,648],{"href":213,"rel":647},[203],"Osmani’s ratchet"," and Thoughtworks’ steering loop are the same instinct. ",[199,651,653],{"href":652},"how-to-evaluate-an-agent-harness","How to evaluate an agent harness"," asks whether your vendor lets you ",[207,656,657],{},"run"," that instinct.",[252,660,662],{"id":661},"questions-people-actually-ask","Questions people actually ask",[664,665,667],"h3",{"id":666},"who-coined-harness-engineering","Who coined “harness engineering”?",[189,669,670],{},"The phrase circulated in early 2026 across OpenAI engineering notes (Ryan Lopopolo’s line of work), LangChain’s anatomy posts, Böckeler at Thoughtworks, and Osmani’s synthesis. Treat it as a shared 2026 name for work teams were already doing, not a trademarked method.",[664,672,674],{"id":673},"is-this-only-for-coding-agents","Is this only for coding agents?",[189,676,677,678,682],{},"The literature is densest there because tests already exist. The discipline is the same for RevOps and Finance: independent sensors, fail-closed writes, versioned context. An ",[199,679,681],{"href":680},"what-is-an-enterprise-agent-harness","enterprise agent harness"," is that application.",[664,684,686],{"id":685},"do-we-wait-for-a-better-model-instead","Do we wait for a better model instead?",[189,688,689,690,693],{},"You still buy better models. You do not pause the ratchet. Stronger models attempt larger jobs and fail in new ways. Anthropic’s long-running work exists ",[207,691,692],{},"because"," models got good enough to outlast a window.",[664,695,697],{"id":696},"how-do-we-start-this-quarter","How do we start this quarter?",[189,699,700,701,705],{},"Pick one job that already has a finish line. Encode guides. Attach one deterministic sensor. Add one hook that can refuse. Run it ten times. Every failure updates the harness. That is a ",[199,702,704],{"href":703},"how-to-run-an-enterprise-ai-proof-of-value","proof of value"," for the practice, not a chat demo.",[664,707,709],{"id":708},"how-does-nimbus-fit-without-becoming-the-definition","How does Nimbus fit without becoming the definition?",[189,711,712,713,280,715,280,718,280,721,723,724,726],{},"Nimbus is an outer harness you can hire: ",[199,714,522],{"href":32},[199,716,717],{"href":20},"teams",[199,719,720],{"href":40},"gates",[199,722,634],{"href":24},". Score it the way you score Claude Code: can you add a sensor, refuse a write, and replay who signed. ",[199,725,653],{"href":652}," is the sheet.",[664,728,730],{"id":729},"what-should-i-read-next","What should I read next?",[189,732,733,737,738,742,743,747],{},[199,734,736],{"href":735},"inner-vs-outer-agent-harness","Inner vs outer agent harness"," for the repo/company cut. ",[199,739,741],{"href":740},"agent-harness-architecture","Agent harness architecture"," for the parts. ",[199,744,746],{"href":745},"what-is-an-agent-harness","What is an agent harness"," if you still need the noun.",[252,749,751],{"id":750},"related-reading","Related reading",[189,753,754,502,758,348],{},[199,755,757],{"href":756},"what-is-an-enterprise-ai-operating-system","What is an enterprise AI operating system",[199,759,761],{"href":760},"how-to-solve-ai-that-cannot-write-back-safely","How to solve AI that cannot write back safely",[252,763,765],{"id":764},"sources","Sources",[257,767,768,774,781,787,793,799,805,811,818,824,830,836,842,848,854],{},[260,769,770],{},[199,771,773],{"href":201,"rel":772},[203],"LangChain, Agents",[260,775,776],{},[199,777,780],{"href":778,"rel":779},"https://www.langchain.com/blog/the-anatomy-of-an-agent-harness",[203],"LangChain, The anatomy of an agent harness",[260,782,783],{},[199,784,786],{"href":219,"rel":785},[203],"Böckeler, Harness engineering for coding agent users",[260,788,789],{},[199,790,792],{"href":298,"rel":791},[203],"Thoughtworks, Harness engineering and agent feedback",[260,794,795],{},[199,796,798],{"href":225,"rel":797},[203],"Thoughtworks, Scaling the enterprise harness (podcast)",[260,800,801],{},[199,802,804],{"href":334,"rel":803},[203],"Thoughtworks, The operating system for enterprise AI",[260,806,807],{},[199,808,810],{"href":213,"rel":809},[203],"Addy Osmani, Agent harness engineering",[260,812,813],{},[199,814,817],{"href":815,"rel":816},"https://www.oreilly.com/radar/agent-harness-engineering/",[203],"O’Reilly Radar, Agent harness engineering",[260,819,820],{},[199,821,823],{"href":237,"rel":822},[203],"Anthropic, Building effective agents",[260,825,826],{},[199,827,829],{"href":243,"rel":828},[203],"Anthropic, Effective harnesses for long-running agents",[260,831,832],{},[199,833,835],{"href":450,"rel":834},[203],"Anthropic, Steering Claude Code",[260,837,838],{},[199,839,841],{"href":310,"rel":840},[203],"Claude Code, Hooks",[260,843,844],{},[199,845,847],{"href":395,"rel":846},[203],"McKinsey, The state of AI in 2025",[260,849,850],{},[199,851,853],{"href":364,"rel":852},[203],"NIST AI RMF",[260,855,856],{},[199,857,407],{"href":405,"rel":858},[203],{"title":170,"searchDepth":171,"depth":171,"links":860},[861,862,863,864,865,866,867,876,877],{"id":254,"depth":171,"text":255},{"id":357,"depth":171,"text":358},{"id":431,"depth":171,"text":432},{"id":529,"depth":171,"text":530},{"id":579,"depth":171,"text":580},{"id":638,"depth":171,"text":639},{"id":661,"depth":171,"text":662,"children":868},[869,871,872,873,874,875],{"id":666,"depth":870,"text":667},3,{"id":673,"depth":870,"text":674},{"id":685,"depth":870,"text":686},{"id":696,"depth":870,"text":697},{"id":708,"depth":870,"text":709},{"id":729,"depth":870,"text":730},{"id":750,"depth":171,"text":751},{"id":764,"depth":171,"text":765},"2026-08-24","Harness engineering is the 2026 practice of fixing the environment when an agent fails — tools, hooks, tests, and stops — instead of rewriting the prompt and hoping the next model call behaves.","/blog/what-is-harness-engineering",{"title":180,"description":879},"explainer","blog/what-is-harness-engineering",[882,885,886,887],"harness-engineering","agent-harness","evaluation","J7WO8MGgOG9nFjnGjs25nfHH5sz-UYeKtt3NvK0eueE",{"hero":890,"id":892,"title":893,"archived":164,"authors":165,"badge":165,"body":894,"date":165,"definedTerm":165,"department":165,"description":898,"extension":173,"eyebrow":899,"faqHeader":165,"faqs":165,"footerBand":900,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":60,"relatedHeading":906,"seo":907,"series":165,"sitemap":130,"status":165,"stem":908,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":909},{"filename":891},"u2221455217_Flat_design_of_a_futuristic_minimalist_landscape__5d589295-cdea-4ea9-a262-be766881accf_1.png","content/blog/index.md","Exploring the future of intelligence.",{"type":167,"value":895,"toc":896},[],{"title":170,"searchDepth":171,"depth":171,"links":897},[],"Deep dives into pre-cognitive intelligence, sentient enterprises, and the evolving landscape of AI-driven business transformation.","Latest Research",{"headline":901,"description":902,"primaryLabel":903,"primaryTo":904,"secondaryLabel":905,"secondaryTo":12},"Stay at the frontier.","Subscribe for product updates and new insights.","Subscribe","/newsletter","Explore the platform","More research",{"title":893,"description":898},"blog/index","BFSWGYO9bcTlaulivKYWyg08_DJHsdGg3OC6g_CG1Hw",[911,1145],{"id":912,"title":913,"archived":164,"authors":914,"badge":916,"body":917,"date":1123,"definedTerm":165,"department":165,"description":1124,"extension":173,"eyebrow":165,"faqHeader":1125,"faqs":1128,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1138,"relatedHeading":165,"seo":1139,"series":882,"sitemap":130,"status":165,"stem":1140,"subhead":165,"tags":1141,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1144},"content/blog/multiplayer-ai-and-multi-agent-ai.md","Multiplayer AI vs multi-agent AI: what is the difference?",[915],{"name":183,"to":135},{"label":185},{"type":167,"value":918,"toc":1116},[919,922,925,932,936,939,942,945,948,956,960,963,966,969,972,982,989,993,996,999,1010,1013,1021,1024,1028,1031,1034,1051,1061,1070,1073,1077,1080,1098,1106,1113],[189,920,921],{},"Multiplayer AI is people and AI on the same job at the same time. Multi-agent AI is more than one model handing work to another. They are not the same product, and they fail in different places. You can have both — several models staffing steps inside one shared room — but buying a swarm is not the same as buying a room.",[189,923,924],{},"You should care if a demo shows agents passing tickets to each other and you still cannot name who would refuse a write to a live system. This is a useful distinction, not a verdict on agent platforms. Plenty of teams will keep specialists for retrieval or checks. The question is whether people still share the job.",[189,926,927,931],{},[199,928,930],{"href":929},"what-is-multi-agent-ai","What is multi-agent AI"," is the cast-of-models definition. This page keeps that word apart from multiplayer: people and AI on the same job at the same time.",[252,933,935],{"id":934},"what-is-the-difference-between-multiplayer-ai-and-multi-agent-ai","What is the difference between multiplayer AI and multi-agent AI?",[189,937,938],{},"Multiplayer answers: who is in the room, what they can see, and who can halt a change. The unit is the job. Finance and sales can open the same brief while the model drafts.",[189,940,941],{},"Multi-agent answers: how work is split between models. One specialist retrieves. Another drafts. A third “reviews.” The unit is the graph — the sequence of model calls.",[189,943,944],{},"A simple check: if you remove every extra model and two departments still cannot share the files and the stop, you never had multiplayer. If you remove the second human and the run still completes in private, you had a personal tool with extra model calls.",[189,946,947],{},"Write-back is when AI changes a live system. In a multiplayer setup, one job holds one payload — the exact change — and a named signer. In a multi-agent setup, several writers can exist unless you bind them to that same stop. Fail-closed means if nobody approves, nothing happens. That rule belongs to a person on the roster, not to the orchestrator.",[189,949,950,951,955],{},"McKinsey’s ",[199,952,954],{"href":395,"rel":953},[203],"State of AI"," (2025) found that most organisations using AI are still piloting. A common pilot is either one copilot or a small agent demo. Neither automatically creates a shared job.",[252,957,959],{"id":958},"why-does-that-distinction-matter","Why does that distinction matter?",[189,961,962],{},"It matters when something goes out wrong and you need a name.",[189,964,965],{},"In multiplayer AI, a named person owns the finish line: the model drafts, and a human on the roster signs or rejects. If the artefact is wrong, you can say who was on the job, including which AI role, and who was allowed to stop it.",[189,967,968],{},"In multi-agent AI, accountability is easy to lose. Each specialist did “its step.” The human who started the run may not have seen the intermediate draft. A log can show that agent B called agent C at 14:03. It does not show that finance agreed.",[189,970,971],{},"Orchestration decides sequence. Accountability is a person with a duty who can refuse at the moment a live system is about to change. A node labelled “human review” is not a name until you can say whose name, on this job, for which class of write.",[189,973,974,978,979,981],{},[199,975,977],{"href":976},"rbac-for-enterprise-ai","RBAC for enterprise AI"," is that list. ",[199,980,419],{"href":418}," is the companion for the write itself.",[189,983,984,985,988],{},"The harness — the tools, stops, and checks around the model — is how a cast of specialists stays bounded. ",[199,986,194],{"href":987},"what-is-harness-engineering"," is the guide to that environment.",[252,990,992],{"id":991},"when-do-you-need-several-people-versus-several-models","When do you need several people versus several models?",[189,994,995],{},"You need several people when more than one owner must stand on the result, or when a handover will happen, or when a customer-facing sentence can leave.",[189,997,998],{},"You need several models when the hand-off already exists between human roles and you want a narrower tool for each step. Useful examples:",[257,1000,1001,1004,1007],{},[260,1002,1003],{},"A research pass that must not share an identity with the agent drafting customer email.",[260,1005,1006],{},"A finance check that should not be able to send mail, even by accident.",[260,1008,1009],{},"A long retrieval over many files that a person will then judge on the job.",[189,1011,1012],{},"Separation of duties is the useful idea. The specialist that recommends a CRM update is not the principal that executes it. Multiplayer AI still puts a human on the execute step.",[189,1014,1015,1016,1020],{},"You do not need a swarm to summarise your own notes. That is a ",[199,1017,1019],{"href":1018},"collaborative-ai-and-personal-assistants","personal assistant",". You do not need a second department on a private brainstorm. You do need both people and a stop when the output can change CRM, a journal, or a message a customer will keep.",[189,1022,1023],{},"A disagreement is a good test. Sales’ specialist wants to send. Legal’s specialist wants to hold. If the orchestrator averages them, or picks the last speaker, you do not have a stop. You have a race. Multiplayer AI makes the human with the duty the one who decides.",[252,1025,1027],{"id":1026},"how-do-you-talk-about-this-with-a-vendor","How do you talk about this with a vendor?",[189,1029,1030],{},"Ask to see the room and the cast as two demos, not one slide.",[189,1032,1033],{},"Useful questions:",[257,1035,1036,1039,1042,1045,1048],{},[260,1037,1038],{},"Can a second department join live, see the same brief, and reject a proposal?",[260,1040,1041],{},"If we remove the person who started the run, can someone else still refuse a write?",[260,1043,1044],{},"When two specialists disagree, who decides — a person with a name, or the graph?",[260,1046,1047],{},"Can we open the intermediate draft tomorrow, including a stored no?",[260,1049,1050],{},"Is the write identity a named human role, or a shared service credential?",[189,1052,1053,1057,1058,1060],{},[199,1054,1056],{"href":1055},"what-auditors-are-asking-for","What auditors are asking for"," is the evidence cut. A common first rule is: do not give the swarm a production write token so the demo looks complete. ",[199,1059,419],{"href":418}," is that checklist.",[189,1062,1063,1064,1069],{},"Stanford HAI’s ",[199,1065,1068],{"href":1066,"rel":1067},"https://hai.stanford.edu/ai-index",[203],"AI Index"," (2025) tracks adoption, investment, and incident reporting. Incident stories are easier to learn from when you can name the job and the signer, not only the model family.",[189,1071,1072],{},"If the vendor can only show a happy path of agents completing a ticket, ask for a specialist disagreement and a human rejection. That is a fair request. You may still buy the swarm for staffing. You will know whether you also bought a workplace.",[252,1074,1076],{"id":1075},"how-do-you-start-without-buying-a-new-stack","How do you start without buying a new stack?",[189,1078,1079],{},"Bind what you already have to one job.",[1081,1082,1083,1086,1089,1092,1095],"ol",{},[260,1084,1085],{},"Pick a recurring job that already has two owners (a weekly exception, a clause check, a forecast update).",[260,1087,1088],{},"Put the brief and two files in one place those people can both open.",[260,1090,1091],{},"If you already run specialists, let them draft into that place. Keep the intermediate draft visible.",[260,1093,1094],{},"Name who can sign a write. Keep the connection read-only until that name exists.",[260,1096,1097],{},"After two cycles, ask: did we fail because we needed another model, or because the second person could not see the file?",[189,1099,1100,1101,502,1103,1105],{},"Nimbus’s ",[199,1102,522],{"href":32},[199,1104,463],{"href":40}," are one attempt at that shape. You can start with a shared folder, a ticket, and a written stop if that is what you have.",[189,1107,1108,1112],{},[199,1109,1111],{"href":1110},"nimbus-vs-paperclip","Nimbus vs Paperclip"," is a vendor-shaped version of the same cut: governing what agents do inside one platform is not the same as two departments finishing a signed forecast in your CRM.",[189,1114,1115],{},"Keep the words apart because they help you buy the right next thing. Multiplayer is the room. Multi-agent is the cast. Adding to the cast is a staffing decision. The room is what owns the result.",{"title":170,"searchDepth":171,"depth":171,"links":1117},[1118,1119,1120,1121,1122],{"id":934,"depth":171,"text":935},{"id":958,"depth":171,"text":959},{"id":991,"depth":171,"text":992},{"id":1026,"depth":171,"text":1027},{"id":1075,"depth":171,"text":1076},"2026-08-17","Multiplayer AI is people and AI on one job. Multi-agent AI is models coordinating. A guide to the distinction, when you need each, and how to talk about it with a vendor.",{"eyebrow":1126,"title":1127},"Short answers","People in the room, or models in a loop?",[1129,1132,1135],{"question":1130,"answer":1131},"Can we have both multiplayer AI and multi-agent AI?","Yes. Several models can staff steps on one shared job. The distinction is whether people share the job, not how many models you run.",{"question":1133,"answer":1134},"Does more agents mean more accountability?","Not by itself. Accountability is a named person who can refuse a change. Extra models without that name make the trail harder to read.",{"question":1136,"answer":1137},"Is a human-in-the-loop node enough?","Only if you can say whose name, on this job, for which class of write. A node labelled “review” is not a roster until it is a person.","/blog/multiplayer-ai-and-multi-agent-ai",{"title":913,"description":1124},"blog/multiplayer-ai-and-multi-agent-ai",[882,1142,1143],"multiplayer AI","multi-agent AI","-n6tINQ6Df9I6LG9Or7a_b3lFTjNqMd-AQcySlZwiZE",{"id":1146,"title":1147,"archived":164,"authors":1148,"badge":1150,"body":1151,"date":878,"definedTerm":165,"department":165,"description":1782,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1783,"relatedHeading":165,"seo":1784,"series":882,"sitemap":130,"status":165,"stem":1785,"subhead":165,"tags":1786,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1788},"content/blog/what-is-an-enterprise-agent-harness.md","What is an Enterprise Agent Harness",[1149],{"name":183,"to":135},{"label":185},{"type":167,"value":1152,"toc":1763},[1153,1159,1172,1185,1187,1252,1269,1271,1290,1292,1309,1336,1351,1355,1362,1378,1397,1411,1425,1435,1444,1454,1460,1464,1469,1476,1486,1489,1500,1504,1516,1523,1526,1529,1542,1546,1549,1555,1566,1569,1573,1590,1601,1603,1607,1610,1614,1617,1621,1624,1628,1636,1640,1647,1649,1659,1661,1670,1672],[189,1154,1155,1156,1158],{},"An ",[192,1157,681],{}," is the outer runtime that lets a model work on company jobs: policy it actually loads, connectors with least privilege, a loop that can stop for a named signer, and a record you can query after the people change.",[189,1160,1161,1162,1166,1167,1171],{},"It is still ",[199,1163,1165],{"href":201,"rel":1164},[203],"Agent = Model + Harness",". The workspace is not a git root. The sensor is not only pytest. The stop is not only max steps. ",[199,1168,1170],{"href":334,"rel":1169},[203],"Thoughtworks"," calls the missing piece an organisational harness: identity, ownership, economics, and learning around whatever builder harnesses (Claude Code, Cursor, LangChain graphs) teams already bought. An enterprise agent harness is that layer made operable — whether you assemble it or hire it.",[189,1173,1174,1177,1178,1181,1182,1184],{},[199,1175,1176],{"href":735},"Inner vs outer"," is the cut. This page is the outer object in full. An ",[199,1179,1180],{"href":756},"enterprise AI operating system"," is the product category that usually ships it: wiki, ",[199,1183,522],{"href":521},", teams, gates, ledger. You can have OS-class products (Nimbus, Palantir AIP, Salesforce Agentforce) and still fail the harness test if writes are a boolean on an API key. You can assemble an enterprise harness in LangGraph and pass the test. The noun is the runtime properties, not the logo.",[252,1186,255],{"id":254},[257,1188,1189,1198,1203,1209,1217,1223,1233,1242],{},[260,1190,1191,1194,1195,348],{},[192,1192,1193],{},"Outer harness."," Company workspace. See ",[199,1196,1197],{"href":735},"inner vs outer",[260,1199,1200,1202],{},[192,1201,330],{}," Thoughtworks’ fourth layer after model, builder harness, and user harness. Governance architecture, not another markdown file.",[260,1204,1205,1208],{},[192,1206,1207],{},"Workstream."," Isolation domain: roster, connectors, budget, finish line. The job folder. Not a chat title.",[260,1210,1211,1214,1215,348],{},[192,1212,1213],{},"Write quoting."," The human sees the change in the language of the live system before sign-off. ",[199,1216,419],{"href":418},[260,1218,1219,1222],{},[192,1220,1221],{},"Fail-closed."," Missing approval, detached grant, or down interceptor means nothing mutates. Fail-open is a faster incident.",[260,1224,1225,1228,1229,348],{},[192,1226,1227],{},"Ledger / Lifecycle Graph."," AI operations events: brief, agents, policy version, signer, payload. Distinct from the warehouse’s business events. See ",[199,1230,1232],{"href":1231},"what-is-a-lifecycle-graph","What is a lifecycle graph",[260,1234,1235,1238,1239,348],{},[192,1236,1237],{},"SWE-bench / Terminal-Bench."," Inner evals. Useful for engineering vendors. Not a SOX control. ",[199,1240,1241],{"href":346},"Eval loops",[260,1243,1244,1247,1248,348],{},[192,1245,1246],{},"Forward-deployed programme."," Vendor engineers for months. AIP at scale. Capability can be real. Time-to-value is staffing. ",[199,1249,1251],{"href":1250},"self-service-vs-forward-deployed-ai-platforms","Self-service vs forward-deployed",[189,1253,1254,1255,280,1257,280,1259,280,1261,280,1263,280,1265,1268],{},"Nimbus is one self-service enterprise harness: ",[199,1256,459],{"href":28},[199,1258,522],{"href":32},[199,1260,517],{"href":20},[199,1262,463],{"href":40},[199,1264,634],{"href":24},[199,1266,1267],{"href":45},"routing",". Score it as an example of the shape, next to AIP and Agentforce, not as the definition of the category.",[252,1270,358],{"id":357},[189,1272,1273,1276,1277,1280,1281,1284,1285,1289],{},[199,1274,397],{"href":395,"rel":1275},[203]," keeps separating ",[207,1278,1279],{},"use"," from ",[207,1282,1283],{},"scale",". Copilots and coding harnesses can produce the first. Enterprise harnesses are how writes to systems of record become the second without becoming ",[199,1286,1288],{"href":1287},"what-is-shadow-ai","shadow AI"," in the CRM.",[189,1291,370],{},[257,1293,1294,1297,1300,1303,1306],{},[260,1295,1296],{},"RevOps, Legal, and Finance must share a job, not a Slack channel of screenshots",[260,1298,1299],{},"Salesforce or NetSuite can change because a model proposed it",[260,1301,1302],{},"last quarter’s pricing chat is unrecoverable",[260,1304,1305],{},"security cannot list the AI actors that may write",[260,1307,1308],{},"the vendor demo is a SWE-bench plot and a “we have MCP”",[189,1310,1311,1312,1317,1318,1323,1324,1329,1330,1335],{},"In 2024 Air Canada was held to a chatbot’s invented policy (",[199,1313,1316],{"href":1314,"rel":1315},"https://www.cbc.ca/news/canada/british-columbia/air-canada-chatbot-lawsuit-1.7116416",[203],"CBC","). That is an outer-harness failure: a commitment left the building without a quote or a signer. ",[199,1319,1322],{"href":1320,"rel":1321},"https://eur-lex.europa.eu/eli/reg/2016/679/oj",[203],"GDPR"," constrains personal data in payloads. ",[199,1325,1328],{"href":1326,"rel":1327},"https://www.sec.gov/about/laws.shtml",[203],"Sarbanes–Oxley"," constrains who may change revenue truth. ",[199,1331,1334],{"href":1332,"rel":1333},"https://eur-lex.europa.eu/eli/reg/2024/1689/oj",[203],"EU AI Act"," Article 14 wants people who can interpret, interrupt, and leave a record. A coding-agent hook that formats Python does not satisfy those.",[189,1337,1338,502,1341,1344,1345,1350],{},[199,1339,853],{"href":364,"rel":1340},[203],[199,1342,407],{"href":405,"rel":1343},[203]," assume operational controls, not a slide titled governance. ",[199,1346,1349],{"href":1347,"rel":1348},"https://oecd.ai/en/ai-principles",[203],"OECD AI Principles"," are a board checklist. They do not implement a gate. The harness does.",[252,1352,1354],{"id":1353},"what-enterprise-adds-to-a-harness","What “enterprise” adds to a harness",[189,1356,1357,1358,1361],{},"Start from ",[199,1359,1360],{"href":745},"what a harness contains"," — loop, tools, memory, permissions, feedback, orchestration — and raise the bar.",[189,1363,1364,1367,1368,1370,1371,1374,1375,1377],{},[192,1365,1366],{},"Policy that loads."," Inner harnesses inject ",[277,1369,279],{},". Enterprise harnesses inject asserted company policy for ",[207,1372,1373],{},"this"," job, versioned. A Drive dump is not policy. A ",[199,1376,459],{"href":287}," that agents cite, with the revision on the run, is. If Legal’s discount cap lives only in a PDF nobody attached, the model will invent a number. That is not hallucination as a personality. That is a missing guide.",[189,1379,1380,1383,1384,1388,1389,1393,1394,348],{},[192,1381,1382],{},"Connectors as grants, not a toolbox."," Default read. Write is a separate plane. Least privilege is a workstream property. ",[199,1385,1387],{"href":1386},"connector-and-permissions-architecture","Connector architecture",". MCP may be the plug; it must inherit the grant. ",[199,1390,1392],{"href":1391},"mcp-for-enterprise-integrations","MCP for enterprise integrations",". A Finance team assigned to a GTM-only stream still must not reach ERP “because it is Finance.” ",[199,1395,1396],{"href":516},"Agent team architecture",[189,1398,1399,1402,1403,1405,1406,1410],{},[192,1400,1401],{},"A hiring object for operators."," Not a folder of personal GPTs. A mandate, required systems, approval triggers — ",[199,1404,517],{"href":929}," as a roster. ",[199,1407,1409],{"href":1408},"how-to-evaluate-agent-teams-vs-single-agents","How to evaluate agent teams vs single agents",". Nimbus ships functional teams on that roster; AIP and Agentforce have their own packaging. The test is: can an operator inspect the mandate and the required systems before assign.",[189,1412,1413,331,1416,1420,1421,1424],{},[192,1414,1415],{},"Human wait as a step.",[199,1417,1419],{"href":1418},"what-is-human-in-the-loop-ai","HITL"," is not a kill switch in a dashboard. It is quoted payload, named role, fail-closed adapter. ",[199,1422,1423],{"href":423},"HITL approval architecture",". Soft / Hard / Critical matched to blast radius. A six-month zero-reject rate on CRM writes is a finding.",[189,1426,1427,1430,1431,348],{},[192,1428,1429],{},"A ledger of AI operations."," Who briefed, which team, which wiki revision, who signed, what executed. Exportable without the vendor in the room. The warehouse is not this ledger. ",[199,1432,1434],{"href":1433},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",[189,1436,1437,1440,1441,348],{},[192,1438,1439],{},"Evals that match the job."," Did the executed write match the signed quote. Can you replay. Inner leaderboards are a vendor quality signal for coding. They are not the enterprise eval. See ",[199,1442,1443],{"href":346},"eval loops",[189,1445,1446,1449,1450,1453],{},[192,1447,1448],{},"Economics of the loop."," Routing compact extract vs frontier judgement. Spend quotes. Seat pricing that includes unlimited flagship is an unengineered cost harness. ",[199,1451,1452],{"href":500},"Model routing",". Nimbus meters NTUs; copilots meter seats. Different jobs.",[189,1455,1456,1459],{},[192,1457,1458],{},"Self-service vs programme."," If every new connector is a six-month SOW, you have bought a deployment, not a harness operators can tighten. That can still be the right buy for Ontology-scale complexity. It is the wrong buy for a standard Salesforce write this quarter.",[252,1461,1463],{"id":1462},"what-it-is-not","What it is not",[189,1465,1466,1467,348],{},"A coding harness with SSO. ",[199,1468,1176],{"href":735},[189,1470,1471,1472,348],{},"A copilot with an admin console. ",[199,1473,1475],{"href":1474},"how-to-choose-between-a-copilot-and-a-work-os","Copilot vs work OS",[189,1477,1478,1479,1482,1483,348],{},"A framework. LangGraph can ",[207,1480,1481],{},"host"," an enterprise harness if you build grants, quotes, and a ledger. Out of the box it hosts a graph. ",[199,1484,1485],{"href":575},"Harness vs framework",[189,1487,1488],{},"“We integrate with Salesforce.” Integration is a slide. A scoped connector plus a blocked unsigned write is a harness.",[189,1490,1491,1492,502,1496,1499],{},"SWE-bench-first marketing. ",[199,1493,1495],{"href":237,"rel":1494},[203],"Anthropic",[199,1497,204],{"href":778,"rel":1498},[203]," are writing about coding and general agents. Steal the discipline (stops, artifacts, sensors). Do not steal the benchmark as your control framework.",[252,1501,1503],{"id":1502},"thoughtworks-organisational-harness-in-operator-language","Thoughtworks’ organisational harness, in operator language",[189,1505,1506,1507,1511,1512,1515],{},"The ",[199,1508,1510],{"href":334,"rel":1509},[203],"Thoughtworks OS essay"," (10 July 2026) argues that most AI programmes fail because the organisation never built the operating system around the model: accountability, ownership, measurement, learning. They name four layers. An enterprise agent harness, as this article uses the term, is layers 3–4 made runnable for ",[207,1513,1514],{},"company jobs"," — not only for coding-agent users.",[189,1517,1518,1519,1522],{},"Delegation failures are the tell. The model was fine. The platform ran. Practitioner guides existed. The agent did what it was ",[207,1520,1521],{},"allowed"," to do. The company still took harm. Layer 4 questions: who approved that autonomy, who owns the policy, what was the escalation, how do we prevent the same miss on another team. Layers 1–3 cannot answer those. A chat product cannot either.",[189,1524,1525],{},"Thoughtworks’ control matrix is worth stealing even if you never hire them. Use deterministic controls where the boundary is knowable: allowed actions, residency, spend ceilings, blast-radius limits. Use probabilistic controls only where judgement is required. Pair every guide with a sensor. Temporal constraints — consistency across a multi-step workflow, not a single dropdown — are the ones they say teams miss most. A scheduling agent that is locally plausible on each step and globally inconsistent is not a “hallucination.” It is a missing temporal sensor.",[189,1527,1528],{},"Their public examples (Parloa’s repo-resident rules/skills/commands; Morgan Stanley’s tiered autonomy on CVE triage) are coding-adjacent. Translate them: discount policy as a versioned wiki skill; “what delegation tier does this CRM write require?” instead of “do we trust the agent.” Nimbus’s Soft / Hard / Critical is that tiering in product form. AIP will have a different packaging. The architectural claim is the same.",[189,1530,1531,502,1536,1541],{},[199,1532,1535],{"href":1533,"rel":1534},"https://www.databricks.com/blog/ai-harness",[203],"Databricks",[199,1537,1540],{"href":1538,"rel":1539},"https://en.wikipedia.org/wiki/Agent_harness",[203],"Wikipedia"," describe the runtime. Thoughtworks describe why a runtime without ownership still fails at scale. You need both descriptions when you buy.",[252,1543,1545],{"id":1544},"what-good-looks-like-on-a-live-job","What “good” looks like on a live job",[189,1547,1548],{},"A renewal write: workstream isolation; Salesforce attached read-only until write is enabled; wiki revision with the cap cited on the run; team cannot start if Legal’s connector requirement is missing; model proposes a quote; Hard gate; reject leaves Stage unchanged; export shows signer without a vendor screen-share. That is an enterprise harness. A demo that only answers “what should we do about Acme” is a copilot with a logo.",[189,1550,1551,1552,348],{},"Spend an hour asking where each Thoughtworks layer lives in the vendor’s product. If layer 4 is “our professional services team,” you are buying a programme. That can be the right buy. Name it. ",[199,1553,1554],{"href":1250},"Self-service vs FDE",[189,1556,1557,1558,1561,1562,1565],{},"Operators already know the human version of this harness. Maker-checker on journals. Segregation of duties on payments. Change-advisory on production. The enterprise agent harness is those instincts encoded so a model cannot talk through them. ",[199,1559,1328],{"href":1326,"rel":1560},[203]," did not wait for LLMs; it waited for a named signer. ",[199,1563,1322],{"href":1320,"rel":1564},[203]," did not wait for MCP; it waits for purpose limitation on the payload. If your AI programme cannot point to the interceptor that enforces those, you have a chatbot with a risk register.",[189,1567,1568],{},"What failure looks like in the first ninety days: every department clones a GPT with the same Salesforce key; Legal’s cap lives in a slide; the only eval is “the demo was impressive”; coding-agent MCP is pointed at production “just for a spike”; the ledger is Slack. What success looks like: one roster of teams, workstream isolation, default read, a Hard refuse on the first PoV, a wiki revision on the graph, inner harnesses still compiling in repos. Nimbus is built to make the success path a product week rather than a services year. Verify that claim with the refuse. AIP may be the right path when Ontology-scale complexity is real — then the harness is a programme, and you should staff it as one.",[252,1570,1572],{"id":1571},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[189,1574,1575,1576,1579,1580,1583,1584,1586,1587,1589],{},"Nimbus’s outer loop is: brief a ",[199,1577,1578],{"href":32},"workstream"," → assign a ",[199,1581,1582],{"href":20},"team"," whose connector contract is satisfied → retrieve under scope → draft on the canvas (Conflux) → quote writes → ",[199,1585,463],{"href":40}," pause → execute the signed payload → commit to the ",[199,1588,23],{"href":24},". Perception orients; it does not silently write. Routing picks model class per step.",[189,1591,1592,1593,502,1597,1600],{},"That mapping is how we productised harness engineering for operators. It is not a claim that AIP or Agentforce are “not harnesses.” They are different time and scope. ",[199,1594,1596],{"href":1595},"how-to-evaluate-an-enterprise-ai-operating-system","How to evaluate an enterprise AI OS",[199,1598,1599],{"href":652},"how to evaluate an agent harness"," are the two sheets; use both.",[252,1602,662],{"id":661},[664,1604,1606],{"id":1605},"do-we-need-this-if-we-already-have-claude-code","Do we need this if we already have Claude Code?",[189,1608,1609],{},"You need it for jobs whose workspace is the company. Keep Claude Code for repos. Do not share production SoR write tokens into the inner harness.",[664,1611,1613],{"id":1612},"is-palantir-aip-an-enterprise-harness","Is Palantir AIP an enterprise harness?",[189,1615,1616],{},"It can be, as a programme-shaped outer runtime. Ask deployment time, who sets a gate without vendor engineers, and whether the ledger is yours. Category yes; evaluation still required.",[664,1618,1620],{"id":1619},"is-agentforce-enough","Is Agentforce enough?",[189,1622,1623],{},"If the job is CRM-anchored and stays there, maybe. Cross-system jobs with Legal on the canvas usually need a harness that is not only Salesforce. Clear scopes; avoid two writers.",[664,1625,1627],{"id":1626},"can-we-build-this-on-langchain","Can we build this on LangChain?",[189,1629,1630,1631,1635],{},"Yes, with time. You will rebuild grants, quoting, roster, and replay. ",[199,1632,1634],{"href":1633},"build-vs-buy-an-enterprise-ai-os","Build vs buy",". Frameworks assemble loops; operators still need a loop they can hire.",[664,1637,1639],{"id":1638},"whats-the-first-proof","What’s the first proof?",[189,1641,1642,1643,1646],{},"A real cross-department write: operator attaches OAuth; unsigned payload blocked; reject leaves SoR unchanged; export shows signer. ",[199,1644,1645],{"href":703},"Proof of value",". A chat demo is not this.",[664,1648,730],{"id":729},[189,1650,1651,1653,1654,1653,1656,348],{},[199,1652,653],{"href":652},". ",[199,1655,741],{"href":740},[199,1657,1658],{"href":987},"What is harness engineering",[252,1660,751],{"id":750},[189,1662,1663,502,1666,348],{},[199,1664,1665],{"href":418},"What is write-back governance",[199,1667,1669],{"href":1668},"rfp-questions-for-enterprise-ai-agents","RFP questions for enterprise AI agents",[252,1671,765],{"id":764},[257,1673,1674,1679,1685,1691,1696,1702,1707,1712,1717,1722,1727,1732,1737,1743,1749,1756],{},[260,1675,1676],{},[199,1677,773],{"href":201,"rel":1678},[203],[260,1680,1681],{},[199,1682,1684],{"href":1533,"rel":1683},[203],"Databricks, What is an AI agent harness?",[260,1686,1687],{},[199,1688,1690],{"href":1538,"rel":1689},[203],"Wikipedia, Agent harness",[260,1692,1693],{},[199,1694,804],{"href":334,"rel":1695},[203],[260,1697,1698],{},[199,1699,1701],{"href":225,"rel":1700},[203],"Thoughtworks, Scaling the enterprise harness",[260,1703,1704],{},[199,1705,823],{"href":237,"rel":1706},[203],[260,1708,1709],{},[199,1710,847],{"href":395,"rel":1711},[203],[260,1713,1714],{},[199,1715,853],{"href":364,"rel":1716},[203],[260,1718,1719],{},[199,1720,407],{"href":405,"rel":1721},[203],[260,1723,1724],{},[199,1725,1349],{"href":1347,"rel":1726},[203],[260,1728,1729],{},[199,1730,1334],{"href":1332,"rel":1731},[203],[260,1733,1734],{},[199,1735,1322],{"href":1320,"rel":1736},[203],[260,1738,1739],{},[199,1740,1742],{"href":1326,"rel":1741},[203],"SEC, Sarbanes–Oxley",[260,1744,1745],{},[199,1746,1748],{"href":1314,"rel":1747},[203],"CBC, Air Canada chatbot lawsuit",[260,1750,1751],{},[199,1752,1755],{"href":1753,"rel":1754},"https://modelcontextprotocol.io/specification/2025-11-25/index",[203],"Model Context Protocol specification",[260,1757,1758],{},[199,1759,1762],{"href":1760,"rel":1761},"https://www.swebench.com/",[203],"SWE-bench",{"title":170,"searchDepth":171,"depth":171,"links":1764},[1765,1766,1767,1768,1769,1770,1771,1772,1780,1781],{"id":254,"depth":171,"text":255},{"id":357,"depth":171,"text":358},{"id":1353,"depth":171,"text":1354},{"id":1462,"depth":171,"text":1463},{"id":1502,"depth":171,"text":1503},{"id":1544,"depth":171,"text":1545},{"id":1571,"depth":171,"text":1572},{"id":661,"depth":171,"text":662,"children":1773},[1774,1775,1776,1777,1778,1779],{"id":1605,"depth":870,"text":1606},{"id":1612,"depth":870,"text":1613},{"id":1619,"depth":870,"text":1620},{"id":1626,"depth":870,"text":1627},{"id":1638,"depth":870,"text":1639},{"id":729,"depth":870,"text":730},{"id":750,"depth":171,"text":751},{"id":764,"depth":171,"text":765},"An enterprise agent harness is the outer runtime for operators: wiki, scoped connectors, agent teams, write gates, and a ledger — not a SWE-bench score and not a chat with every production login.","/blog/what-is-an-enterprise-agent-harness",{"title":1147,"description":1782},"blog/what-is-an-enterprise-agent-harness",[882,886,1787,463],"enterprise-ai","VCI6uaS6V4vGana_omswQGrFpUW2iH4qg59Oz11-AgA",{"enabled":164,"message":1790,"linkLabel":93,"linkHref":94,"id":1791,"title":1792,"archived":164,"authors":165,"badge":165,"body":1793,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1797,"relatedHeading":165,"seo":1798,"series":165,"sitemap":164,"status":165,"stem":1799,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1800},"We're hiring! Join the team building the Sentient Enterprise.","content/shared/hiring.md","Hiring banner",{"type":167,"value":1794,"toc":1795},[],{"title":170,"searchDepth":171,"depth":171,"links":1796},[],"/shared/hiring",{"title":1792,"description":170},"shared/hiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",{"fold":1802,"id":1806,"title":1807,"archived":164,"authors":165,"badge":165,"body":1808,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":1812,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1816,"relatedHeading":165,"seo":1817,"series":165,"sitemap":164,"status":165,"stem":1818,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1819},{"headline":1803,"description":1804,"primaryLabel":8,"primaryTo":1805,"secondaryLabel":905,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Free 7-day trial.","/checkout","content/shared/cta.md","Site CTAs",{"type":167,"value":1809,"toc":1810},[],{"title":170,"searchDepth":171,"depth":171,"links":1811},[],{"headline":1813,"description":1814,"primaryLabel":8,"primaryTo":1805,"secondaryLabel":1815,"secondaryTo":99},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger - free for 7 days.","Talk to our team","/shared/cta",{"title":1807,"description":170},"shared/cta","wz4AdRHnaYH021WMdWcHnZvHmkZJNKaNG4XGZfnFBtw",1788985847672]