[{"data":1,"prerenderedAt":12312},["ShallowReactive",2],{"site-nav-content":3,"hub:glossary:posts":178,"site-cta-content":12280,"hiring-banner-content":12300},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":131,"id":162,"title":163,"archived":164,"authors":165,"badge":165,"body":166,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":174,"relatedHeading":165,"seo":175,"series":165,"sitemap":164,"status":165,"stem":176,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":177},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","/overview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","/product/conflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","/product/agent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","/product/lifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","/product/wiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","/product/workstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","/product/perception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","/product/governance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","/models",{"label":47,"to":48},"Pricing","/pricing",{"label":50,"to":51},"Integrations","/integrations",{"label":53,"to":54},"Security","/security",{"label":56,"to":57},"Partners","/partners",{"label":59,"to":60},"Insights","/blog",{"productHeading":5,"companyHeading":62,"legalHeading":63,"docsLabel":64,"docsUrl":65,"statementLines":66,"copyright":69,"companyLinks":70,"legalLinks":100,"socialLinks":110,"bottomLinks":120},"Company","Legal","Docs","https://docs.gonimbus.ai",[67,68],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[71,72,73,74,75,78,81,84,87,90,92,95,98],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":76,"to":77},"Glossary","/glossary",{"label":79,"to":80},"Compare","/compare",{"label":82,"to":83},"Evaluate","/evaluate",{"label":85,"to":86},"Problems","/problems",{"label":88,"to":89},"Use cases","/use-cases",{"label":91,"to":57},"Partner Program",{"label":93,"to":94},"Careers","/careers",{"label":96,"to":97},"System status","/status",{"label":7,"to":99},"/contact",[101,104,107],{"label":102,"to":103},"Terms of Service","/terms",{"label":105,"to":106},"Privacy Policy","/privacy",{"label":108,"to":109},"Compliance","/compliance",[111,114,117],{"label":112,"href":113},"LinkedIn","https://www.linkedin.com/company/gonimbusai/",{"label":115,"href":116},"X","https://x.com/gonimbusai",{"label":118,"href":119},"Instagram","https://www.instagram.com/gonimbus_ai/",[121,123,125,126,127],{"label":122,"to":103},"Terms",{"label":124,"to":106},"Privacy",{"label":108,"to":109},{"label":96,"to":97},{"label":128,"to":129,"external":130},"LLMs.txt","/llms.txt",true,{"text":132,"prompt":133},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":134,"platforms":136},{" Summarize the highlights from Nimbus's website":135},"https://gonimbus.ai",[137,142,147,152,157],{"name":138,"label":139,"icon":140,"hrefPrefix":141},"chatgpt","ChatGPT","simple-icons:openai","https://chatgpt.com/?prompt=",{"name":143,"label":144,"icon":145,"hrefPrefix":146},"perplexity","Perplexity","mdi:magnify","https://www.perplexity.ai/search/new?q=",{"name":148,"label":149,"icon":150,"hrefPrefix":151},"grok","Grok","simple-icons:x","https://x.com/i/grok?text=",{"name":153,"label":154,"icon":155,"hrefPrefix":156},"claude","Claude","simple-icons:anthropic","https://claude.ai/new?q=",{"name":158,"label":159,"icon":160,"hrefPrefix":161},"google-ai","Google AI","simple-icons:google","https://www.google.com/search?udm=50&aep=11&q=","content/shared/nav.md","Site navigation",false,null,{"type":167,"value":168,"toc":169},"minimark",[],{"title":170,"searchDepth":171,"depth":171,"links":172},"",2,[],"md","/shared/nav",{"title":163,"description":170},"shared/nav","rDEv5cVG6P2l9ATcdQv2n6VOQiSL5ioOutyfvGevcD0",[179,1018,1612,2271,2894,3123,3590,4049,4597,5047,5651,6066,6497,7082,7539,7970,8423,9022,9462,9890,10339,10737,11154,11354,11786],{"id":180,"title":181,"archived":164,"authors":182,"badge":185,"body":187,"date":1008,"definedTerm":165,"department":165,"description":1009,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1010,"relatedHeading":165,"seo":1011,"series":1012,"sitemap":130,"status":165,"stem":1013,"subhead":165,"tags":1014,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1017},"content/blog/agent-harness-architecture.md","Agent Harness Architecture",[183],{"name":184,"to":135},"Nimbus Research",{"label":186},"Architecture",{"type":167,"value":188,"toc":990},[189,197,231,254,259,321,349,353,379,387,390,394,397,415,426,446,452,476,485,495,506,516,527,533,539,542,546,595,603,607,614,723,726,738,744,752,759,767,784,790,794,804,808,813,816,820,827,831,839,843,850,854,866,870,880,884],[190,191,192,196],"p",{},[193,194,195],"strong",{},"Agent harness architecture"," is the design of the runtime around a model: who owns the loop, how tools run, what context is injected, which hooks can refuse, which identity the tools use, and how “done” is checked without taking the model’s word.",[190,198,199,206,207,212,213,217,218,217,222,217,226,230],{},[200,201,205],"a",{"href":202,"rel":203},"https://www.langchain.com/blog/the-anatomy-of-an-agent-harness",[204],"nofollow","LangChain’s anatomy"," is the public parts list: prompts, tools and MCP, bundled infrastructure (filesystem, sandbox, browser), orchestration (subagents, routing), hooks and middleware (compaction, lint, continuation). ",[200,208,211],{"href":209,"rel":210},"https://www.databricks.com/blog/ai-harness",[204],"Databricks"," groups the same into tools, memory, workspace, guardrails. This article is that list as an architecture you can inspect — then the mapping onto company jobs: ",[200,214,216],{"href":215},"what-is-an-ai-workstream","workstreams",", ",[200,219,221],{"href":220},"agent-team-architecture","agent teams",[200,223,225],{"href":224},"connector-and-permissions-architecture","connectors",[200,227,229],{"href":228},"what-is-write-back-governance","write-back",".",[190,232,233,234,238,239,243,244,248,249,253],{},"It is not a novel about kernels. It is not ",[200,235,237],{"href":236},"multi-agent-ai-architecture","multi-agent protocol"," (hand-offs between specialists) and not ",[200,240,242],{"href":241},"human-in-the-loop-approval-architecture","HITL state machines"," (quote → sign → execute), though a complete outer harness contains both. Start from ",[200,245,247],{"href":246},"what-is-an-agent-harness","what is an agent harness",". Use ",[200,250,252],{"href":251},"how-to-evaluate-an-agent-harness","how to evaluate"," as the test of this diagram.",[255,256,258],"h2",{"id":257},"words-youll-hear","Words you’ll hear",[260,261,262,278,288,299,311],"ul",{},[263,264,265,268,269,273,274,277],"li",{},[193,266,267],{},"Control plane vs data plane."," Control: grants, budgets, gates, routing policy — known independently of the model. Data: tokens, tool results, artefacts. If the orchestrator is only a system prompt, a jailbreak ",[270,271,272],"em",{},"is"," a privilege escalation. ",[200,275,276],{"href":236},"Multi-agent architecture"," already said this; it is a harness invariant.",[263,279,280,283,284,230],{},[193,281,282],{},"Workspace."," Inner: checkout / sandbox. Outer: workstream. ",[200,285,287],{"href":286},"inner-vs-outer-agent-harness","Inner vs outer",[263,289,290,293,294,230],{},[193,291,292],{},"Tool plane vs write plane."," Reads default on. Mutations fail-closed. MCP may implement both; architecture must split them. ",[200,295,298],{"href":296,"rel":297},"https://modelcontextprotocol.io/specification/2025-11-25/index",[204],"MCP spec",[263,300,301,304,305,310],{},[193,302,303],{},"Compaction."," Harness-owned context management so the window does not become the only memory. Anthropic’s ",[200,306,309],{"href":307,"rel":308},"https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents",[204],"long-running harness"," offloads state to files and git.",[263,312,313,316,317,230],{},[193,314,315],{},"Routing."," Model class per step, not a user-picked mascot. ",[200,318,320],{"href":319},"model-routing-architecture","Model routing architecture",[190,322,323,324,327,328,331,332,334,335,337,338,341,342,344,345,348],{},"Nimbus maps this architecture onto product objects rather than asking operators to draw LangGraph: ",[200,325,326],{"href":28},"wiki"," (guides), ",[200,329,330],{"href":51},"integrations"," (tool plane), ",[200,333,221],{"href":20}," (orchestration contract), ",[200,336,216],{"href":32}," (workspace), ",[200,339,340],{"href":40},"governance"," (write plane), ",[200,343,23],{"href":24}," (eval and memory), ",[200,346,347],{"href":45},"models"," (routing). Other vendors map the same boxes differently. Score the boxes.",[255,350,352],{"id":351},"why-architecture-not-a-bigger-prompt","Why architecture (not a bigger prompt)",[190,354,355,356,360,361,366,367,372,373,378],{},"A prompt cannot own tool execution, identity, or a stop that survives a tired model. ",[200,357,359],{"href":358},"what-is-harness-engineering","Harness engineering"," is the practice; this page is the structure the practice edits. ",[200,362,365],{"href":363,"rel":364},"https://www.nist.gov/itl/ai-risk-management-framework",[204],"NIST AI RMF"," Govern/Map need a system you can point to. ",[200,368,371],{"href":369,"rel":370},"https://www.iso.org/standard/42001",[204],"ISO 42001"," needs operational controls. ",[200,374,377],{"href":375,"rel":376},"https://genai.owasp.org/llm-top-10/",[204],"OWASP LLM Top 10"," excessive agency is what happens when the tool plane has no architecture.",[190,380,381,386],{},[200,382,385],{"href":383,"rel":384},"https://www.mckinsey.com/capabilities/quantumblack/our-insights/the-state-of-ai",[204],"McKinsey 2025"," treats agentic value as organisational. Architecture is how you stop “every team’s unofficial loop” from becoming the estate.",[190,388,389],{},"It affects you if you are combining MCP servers, a coding agent, a copilot, and a CRM writer without a single grant and quote rule. Two writers to one object is an architecture bug, not a training issue.",[255,391,393],{"id":392},"the-pieces","The pieces",[190,395,396],{},"Keep these as inspectable contracts.",[190,398,399,402,403,408,409,414],{},[193,400,401],{},"1. Loop runtime."," Plan → act → observe, with max steps and a cost budget the model cannot waive. Frameworks (",[200,404,407],{"href":405,"rel":406},"https://docs.langchain.com/oss/python/langchain/agents",[204],"create_agent",", LangGraph, CrewAI) implement this in process. Product harnesses implement it as a hosted run. ",[200,410,413],{"href":411,"rel":412},"https://www.anthropic.com/engineering/building-effective-agents",[204],"Anthropic’s effective agents"," is still the best short note on bounding the loop. “The model says it is done” is an input to the runtime, not the runtime.",[190,416,417,420,421,425],{},[193,418,419],{},"2. Workspace and filesystem."," Inner harnesses treat the directory as externalised memory — Manus-style and Anthropic-style artefacts. Outer harnesses treat the workstream as the directory analogue: artefacts on a canvas, not a hidden ",[422,423,424],"code",{},"/tmp"," on a laptop. Do not store approved discounts only in a coding agent’s memory file.",[190,427,428,431,432,435,436,440,441,445],{},[193,429,430],{},"3. Context assembly."," System prompt, skills, ",[422,433,434],{},"AGENTS.md"," / wiki slices, retrieved records, prior graph nodes. Guides in Böckeler’s sense. Compaction and retrieval belong here. ",[200,437,439],{"href":438},"what-is-enterprise-rag","Enterprise RAG"," is a pattern inside this box, not the architecture. ",[200,442,444],{"href":443},"what-is-a-company-wiki-for-ai-agents","Company wiki"," is asserted policy; do not collapse it into a private vector bucket per agent.",[190,447,448,451],{},[193,449,450],{},"4. Tool dispatch."," Host executes; model proposes. Sandbox for shell. Adapters for SaaS. Timeouts, retries, structured errors back into the loop. Generic HTTP with a production token is not this box. It is a confused deputy.",[190,453,454,457,458,463,464,467,468,471,472,475],{},[193,455,456],{},"5. Hooks / middleware."," Deterministic intercepts: ",[200,459,462],{"href":460,"rel":461},"https://code.claude.com/docs/en/hooks",[204],"Claude Code"," ",[422,465,466],{},"PreToolUse"," / ",[422,469,470],{},"PostToolUse","; LangChain middleware; outer interceptor that never exposes the write API unsigned. ",[200,473,474],{"href":228},"Write-back governance",". Advice in markdown does not live in this box.",[190,477,478,481,482,230],{},[193,479,480],{},"6. Permissions and identity."," Who the harness authenticates as, per tool, per object, per job. Roster and workstream membership on the outer side. Repo and sandbox roles on the inner side. Teams declare required connectors; the workspace still grants. ",[200,483,484],{"href":220},"Agent team architecture",[190,486,487,490,491,230],{},[193,488,489],{},"7. Orchestration."," Subagents, specialist hand-offs, stop on gate. Optional until duties already split. Orchestrator in the product, not a manager persona with every login. ",[200,492,494],{"href":493},"what-is-multi-agent-ai","What is multi-agent AI",[190,496,497,500,501,505],{},[193,498,499],{},"8. Sensors and eval."," Compiler, tests, schema, quote-hash, SoR read-back, human review. Independent of the generator. ",[200,502,504],{"href":503},"eval-loops-for-enterprise-agent-harnesses","Eval loops",". SWE-bench / Terminal-Bench measure inner coding harnesses; they do not close this box for GL posts.",[190,507,508,511,512,230],{},[193,509,510],{},"9. Durable memory of operations."," Files and git (inner). Wiki + Lifecycle Graph (outer). Session transcripts are a debug aid. They are not the ledger. ",[200,513,515],{"href":514},"causal-memory-architecture-for-enterprise-ai","Causal memory",[190,517,518,521,522,526],{},[193,519,520],{},"10. Routing and spend."," Step classes → model classes. Caps on the run. ",[200,523,525],{"href":524},"ai-cost-control-architecture","AI cost control",". Seat-unlimited flagship is an architectural choice (always-frontier), not a missing feature.",[190,528,529,532],{},[193,530,531],{},"Flow (outer)."," Brief on a workstream → satisfy connector contract → plan → retrieve (logged, scoped) → draft on canvas → quote if write in scope → gate → execute signed payload only → commit graph. If steps 5–7 live only in a prompt, jailbreaks and tired operators fall through the same hole.",[190,534,535,538],{},[193,536,537],{},"Flow (inner)."," Session start loads guides → loop with shell/editor tools → hooks on tool events → tests as sensor → commit / PR → CI as outer-loop sensor in Osmani’s sense. Anthropic’s initializer vs coding agent is a two-role inner architecture for work that outlasts one window.",[190,540,541],{},"Nimbus’s hosted flow is the outer sequence. Perception and Conflux sit on retrieve/draft; they must not skip the quote. That is architecture, not brand.",[255,543,545],{"id":544},"failure-modes-the-diagram-exists-to-prevent","Failure modes the diagram exists to prevent",[547,548,549,555,561,567,573,579,585],"ol",{},[263,550,551,554],{},[193,552,553],{},"Orchestrator-in-the-model."," Jailbreak equals admin.",[263,556,557,560],{},[193,558,559],{},"Shared toolbox."," Every specialist has every write.",[263,562,563,566],{},[193,564,565],{},"Context as only memory."," Compaction deletes the approval.",[263,568,569,572],{},[193,570,571],{},"MCP as control plane."," Plug without grants.",[263,574,575,578],{},[193,576,577],{},"Eval = transcript."," The model graded itself.",[263,580,581,584],{},[193,582,583],{},"Two harnesses, one SoR writer."," IDE MCP and OS both PATCH.",[263,586,587,590,591,230],{},[193,588,589],{},"Framework mistaken for architecture."," Nodes without identity. ",[200,592,594],{"href":593},"agent-harness-vs-agent-framework","Harness vs framework",[190,596,597,602],{},[200,598,601],{"href":599,"rel":600},"https://eur-lex.europa.eu/eli/reg/2024/1689/oj",[204],"EU AI Act"," oversight needs interrupt and record. Those are boxes 5, 6, and 9.",[255,604,606],{"id":605},"mapping-langchains-anatomy-onto-company-objects","Mapping LangChain’s anatomy onto company objects",[190,608,609,613],{},[200,610,612],{"href":202,"rel":611},[204],"LangChain’s parts list"," is built from coding and general agents. Translate, do not copy:",[615,616,617,633],"table",{},[618,619,620],"thead",{},[621,622,623,627,630],"tr",{},[624,625,626],"th",{},"Anatomy piece",[624,628,629],{},"Inner binding",[624,631,632],{},"Outer binding",[634,635,636,651,662,673,687,698,712],"tbody",{},[621,637,638,642,648],{},[639,640,641],"td",{},"System prompts / skills",[639,643,644,647],{},[422,645,646],{},"CLAUDE.md",", skills",[639,649,650],{},"Wiki playbooks, versioned with the run",[621,652,653,656,659],{},[639,654,655],{},"Tools + MCP",[639,657,658],{},"Shell, apply_patch, browser",[639,660,661],{},"Connectors; MCP behind the same grant",[621,663,664,667,670],{},[639,665,666],{},"Filesystem / sandbox",[639,668,669],{},"Checkout, container",[639,671,672],{},"Workstream canvas + isolated grants",[621,674,675,678,681],{},[639,676,677],{},"Orchestration",[639,679,680],{},"Subagents in the IDE",[639,682,683,686],{},[200,684,685],{"href":220},"Agent teams"," on a roster",[621,688,689,692,695],{},[639,690,691],{},"Hooks / middleware",[639,693,694],{},"PreToolUse, lint",[639,696,697],{},"Write interceptor, spend cap",[621,699,700,703,706],{},[639,701,702],{},"Memory",[639,704,705],{},"Files, git, memory md",[639,707,708,709],{},"Wiki + ",[200,710,23],{"href":711},"what-is-a-lifecycle-graph",[621,713,714,717,720],{},[639,715,716],{},"Eval",[639,718,719],{},"Tests, Terminal-Bench",[639,721,722],{},"Quote hash, SoR read-back",[190,724,725],{},"If a vendor cannot fill the outer column, they are an inner (or framework) product. That is allowed. Do not invent the column in a slide.",[190,727,728,731,732,735,736,230],{},[193,729,730],{},"Control plane independence."," Whatever sits in the Orchestration row must know grants, budget, and gates ",[270,733,734],{},"without"," asking the model. LangGraph can do that if the nodes are code. A “manager agent” with every tool cannot. Nimbus’s orchestrator is product-hosted for that reason; you should still ask it to refuse when NetSuite is missing. ",[200,737,82],{"href":251},[190,739,740,743],{},[193,741,742],{},"Thoughtworks’ four combinations"," (deterministic/probabilistic × feed-forward/feedback) overlay this table. Whitelists and spend ceilings are box 5/6 deterministic feed-forward. Schema validation is box 8 deterministic feedback. Wiki retrieval is probabilistic feed-forward. LLM critic is probabilistic feedback — never the only item in box 8 for a GL post.",[190,745,746,749,750,230],{},[193,747,748],{},"Two harnesses, one SoR rule."," Draw both columns on one whiteboard. Draw one write plane. If two arrows reach Salesforce, you have an architecture incident waiting. ",[200,751,287],{"href":286},[190,753,754,755,230],{},"Version the diagram when you add a tool. A new MCP server is a change to boxes 4 and 6, not a chat plugin. ",[200,756,758],{"href":757},"what-is-model-context-protocol","MCP",[190,760,761,762,766],{},"Implementation order for a company that has none of this: (1) split write plane from read plane — even if the “harness” is still a single agent; (2) pin policy version on the run; (3) add one deterministic sensor on the artefact you cannot get wrong; (4) host the orchestrator’s grants outside the prompt; (5) only then add specialists. Reversing that order is how shared-toolbox swarms ship. ",[200,763,765],{"href":411,"rel":764},[204],"Anthropic"," starts with bounding tools and defining done for a reason.",[190,768,769,770,217,772,217,774,217,776,779,780,783],{},"Framework teams should draw the ten boxes on the README of the graph repo and tick which are code, which are still prompts, which are missing. Product teams should map each box to a screen an operator can see. If box 8 is “the model reflects,” you do not have eval architecture. If box 6 is “the service account,” you do not have identity architecture. Nimbus’s screens are ",[200,771,216],{"href":32},[200,773,340],{"href":40},[200,775,326],{"href":28},[200,777,778],{"href":24},"graph"," — use them as a checklist, not as proof that the boxes exist in ",[270,781,782],{},"your"," configuration.",[190,785,786,789],{},[200,787,211],{"href":209,"rel":788},[204]," calls the model the brain and the harness the body. Architecture is the anatomy of that body so Security can review it. If the diagram is only “LLM in the middle, tools around it,” you have a marketing poster. Add identity, the write split, the sensor that does not trust the brain, and the ledger. Then the poster is a design.",[255,791,793],{"id":792},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[190,795,796,797,799,800,803],{},"The product is a particular binding of the ten boxes for operators: hosted loop, workstream workspace, wiki context, connector dispatch, governance hooks, team orchestration, graph memory, NTU routing. ",[200,798,11],{"href":12},". Inspect each box in a PoV the way you would inspect Claude Code’s hooks and sandbox for an inner buy. ",[200,801,802],{"href":251},"How to evaluate",". AIP and Agentforce bind the same boxes to Ontology or CRM; the architecture still applies.",[255,805,807],{"id":806},"questions-people-actually-ask","Questions people actually ask",[809,810,812],"h3",{"id":811},"do-we-need-all-ten-boxes-on-day-one","Do we need all ten boxes on day one?",[190,814,815],{},"You need loop, tools, a stop, and a sensor for the job you are running. Add orchestration when duties split. Add graph when people leave. Do not add every MCP server first.",[809,817,819],{"id":818},"is-this-the-same-as-an-enterprise-ai-os-architecture","Is this the same as an enterprise AI OS architecture?",[190,821,822,826],{},[200,823,825],{"href":824},"enterprise-ai-operating-system-architecture","OS architecture"," is the product category (collaboration, gates, ledger, routing). Harness architecture is the runtime idea that also covers Claude Code. Overlap on the outer side is expected.",[809,828,830],{"id":829},"where-do-skills-fit","Where do skills fit?",[190,832,833,834,230],{},"Reusable procedures in the context box. Not a substitute for hooks. ",[200,835,838],{"href":836,"rel":837},"https://claude.com/blog/steering-claude-code-skills-hooks-rules-subagents-and-more",[204],"Anthropic on steering",[809,840,842],{"id":841},"can-langgraph-implement-this","Can LangGraph implement this?",[190,844,845,846,230],{},"Yes. You will implement boxes 5, 6, and 9 yourself for enterprise writes. That is ",[200,847,849],{"href":848},"build-vs-buy-an-enterprise-ai-os","build vs buy",[809,851,853],{"id":852},"what-should-i-read-next","What should I read next?",[190,855,856,858,859,858,862,230],{},[200,857,504],{"href":503},". ",[200,860,861],{"href":358},"What is harness engineering",[200,863,865],{"href":864},"what-is-an-enterprise-agent-harness","What is an enterprise agent harness",[255,867,869],{"id":868},"related-reading","Related reading",[190,871,872,876,877,230],{},[200,873,875],{"href":874},"workstream-architecture","Workstream architecture"," and ",[200,878,879],{"href":224},"Connector and permissions architecture",[255,881,883],{"id":882},"sources","Sources",[260,885,886,892,898,905,911,918,924,930,936,942,949,956,961,967,972,978,984],{},[263,887,888],{},[200,889,891],{"href":202,"rel":890},[204],"LangChain, The anatomy of an agent harness",[263,893,894],{},[200,895,897],{"href":405,"rel":896},[204],"LangChain, Agents",[263,899,900],{},[200,901,904],{"href":902,"rel":903},"https://www.langchain.com/blog/how-to-build-a-custom-agent-harness",[204],"LangChain, How to build a custom agent harness",[263,906,907],{},[200,908,910],{"href":209,"rel":909},[204],"Databricks, What is an AI agent harness?",[263,912,913],{},[200,914,917],{"href":915,"rel":916},"https://en.wikipedia.org/wiki/Agent_harness",[204],"Wikipedia, Agent harness",[263,919,920],{},[200,921,923],{"href":411,"rel":922},[204],"Anthropic, Building effective agents",[263,925,926],{},[200,927,929],{"href":307,"rel":928},[204],"Anthropic, Effective harnesses for long-running agents",[263,931,932],{},[200,933,935],{"href":836,"rel":934},[204],"Anthropic, Steering Claude Code",[263,937,938],{},[200,939,941],{"href":460,"rel":940},[204],"Claude Code, Hooks",[263,943,944],{},[200,945,948],{"href":946,"rel":947},"https://martinfowler.com/articles/harness-engineering.html",[204],"Böckeler, Harness engineering for coding agent users",[263,950,951],{},[200,952,955],{"href":953,"rel":954},"https://addyosmani.com/blog/agent-harness-engineering/",[204],"Addy Osmani, Agent harness engineering",[263,957,958],{},[200,959,365],{"href":363,"rel":960},[204],[263,962,963],{},[200,964,966],{"href":369,"rel":965},[204],"ISO/IEC 42001",[263,968,969],{},[200,970,601],{"href":599,"rel":971},[204],[263,973,974],{},[200,975,977],{"href":375,"rel":976},[204],"OWASP Top 10 for LLM applications",[263,979,980],{},[200,981,983],{"href":383,"rel":982},[204],"McKinsey, The state of AI in 2025",[263,985,986],{},[200,987,989],{"href":296,"rel":988},[204],"Model Context Protocol specification",{"title":170,"searchDepth":171,"depth":171,"links":991},[992,993,994,995,996,997,998,1006,1007],{"id":257,"depth":171,"text":258},{"id":351,"depth":171,"text":352},{"id":392,"depth":171,"text":393},{"id":544,"depth":171,"text":545},{"id":605,"depth":171,"text":606},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":999},[1000,1002,1003,1004,1005],{"id":811,"depth":1001,"text":812},3,{"id":818,"depth":1001,"text":819},{"id":829,"depth":1001,"text":830},{"id":841,"depth":1001,"text":842},{"id":852,"depth":1001,"text":853},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"2026-08-24","Agent harness architecture is the runtime around a model: loop, tools, context, hooks, permissions, and eval — mapped, for company jobs, onto workstreams, agent teams, connectors, and write gates.","/blog/agent-harness-architecture",{"title":181,"description":1009},"architecture","blog/agent-harness-architecture",[1012,1015,1016,340],"agent-harness","orchestration","NbF7f7oegzZJ7kbYT2wy7RoWBPoIrmmPE1R8WlUetFA",{"id":1019,"title":1020,"archived":164,"authors":1021,"badge":1023,"body":1025,"date":1008,"definedTerm":165,"department":165,"description":1603,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1604,"relatedHeading":165,"seo":1605,"series":1606,"sitemap":130,"status":165,"stem":1607,"subhead":165,"tags":1608,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1611},"content/blog/agent-harness-vs-agent-framework.md","Agent Harness vs Agent Framework",[1022],{"name":184,"to":135},{"label":1024},"Explainer",{"type":167,"value":1026,"toc":1586},[1027,1038,1067,1085,1088,1090,1143,1157,1161,1169,1172,1186,1201,1207,1211,1217,1231,1249,1264,1270,1280,1291,1305,1312,1316,1345,1351,1355,1365,1376,1382,1390,1401,1417,1424,1429,1432,1438,1440,1444,1451,1455,1462,1466,1469,1473,1476,1480,1487,1491,1499,1501,1509,1511],[190,1028,1029,1030,1033,1034,1037],{},"An ",[193,1031,1032],{},"agent framework"," is a library for composing models, tools, and control flow. An ",[193,1035,1036],{},"agent harness"," is the running environment around a model: the loop, the tools as they are actually granted, the stops, the sensors, and the identity that production will use.",[190,1039,1040,1044,1045,463,1048,1050,1051,1056,1057,1062,1063,1066],{},[200,1041,1043],{"href":405,"rel":1042},[204],"LangChain’s own docs"," are careful with the words. ",[193,1046,1047],{},"Agent = Model + Harness.",[422,1049,407],{}," is “a highly configurable harness.” ",[200,1052,1055],{"href":1053,"rel":1054},"https://github.com/langchain-ai/deepagents",[204],"Deep Agents"," is “the batteries-included agent harness.” ",[200,1058,1061],{"href":1059,"rel":1060},"https://docs.langchain.com/oss/python/langgraph/overview",[204],"LangGraph"," is the low-level orchestration framework when the built-in loop is the wrong shape. That taxonomy is the whole article: a framework can ",[270,1064,1065],{},"implement"," a harness. Shipping the pip package does not mean you have one operators can hire.",[190,1068,1069,1073,1074,1076,1077,1080,1081,1084],{},[200,1070,1072],{"href":902,"rel":1071},[204],"LangChain’s custom-harness post"," says the same from the other side. Pre-assembled harnesses (Deep Agents, Claude Agent SDK) get you to a working agent fast. ",[422,1075,407],{}," is minimal on purpose: core loop plus middleware. You still choose tools, guardrails, and business logic. CrewAI, Semantic Kernel, AutoGen, and Pydantic AI live in this neighbourhood. They are how engineers assemble loops. They are not a substitute for ",[200,1078,1079],{"href":228},"write-back governance",", a ",[200,1082,1083],{"href":215},"workstream",", or a ledger.",[190,1086,1087],{},"Claude Code and Cursor are harnesses you run, not frameworks you import. Nimbus, Palantir AIP, and Agentforce are (different) harnesses you run for company jobs. Confusing “we use LangGraph” with “we have an enterprise harness” is the 2026 version of “we use Kubernetes” meaning “we have a product.”",[255,1089,258],{"id":257},[260,1091,1092,1098,1111,1121,1127,1133],{},[263,1093,1094,1097],{},[193,1095,1096],{},"Framework."," SDKs and graphs: LangChain, LangGraph, CrewAI, AutoGen, Semantic Kernel, Pydantic AI. You write code. You own production identity unless you add it.",[263,1099,1100,1103,1104,1107,1108,1110],{},[193,1101,1102],{},"Harness."," Runtime around the model. ",[200,1105,1106],{"href":246},"What is an agent harness",". May be a product (Claude Code) or a configured framework (your ",[422,1109,407],{}," plus hooks plus IdP).",[263,1112,1113,1116,1117,230],{},[193,1114,1115],{},"Middleware / hooks."," Framework primitive that becomes harness behaviour when it always runs. LangChain middleware; ",[200,1118,1120],{"href":460,"rel":1119},[204],"Claude Code hooks",[263,1122,1123,1126],{},[193,1124,1125],{},"Batteries-included harness."," Deep Agents, Claude Agent SDK, Codex SDK. Opinionated loop, filesystem, subagents, compaction. Still not your CRM grant model.",[263,1128,1129,1132],{},[193,1130,1131],{},"Orchestration framework."," LangGraph when you need deterministic nodes mixed with agentic ones. Powerful. Easy to put the orchestrator in a system prompt and call it done.",[263,1134,1135,1138,1139,1142],{},[193,1136,1137],{},"MCP."," Plug. ",[200,1140,1141],{"href":757},"What is Model Context Protocol",". Works behind frameworks and products. Does not choose the framework/harness cut.",[190,1144,1145,1146,1149,1150,1152,1153,230],{},"In Nimbus you do not import a graph to start a job. You assign an ",[200,1147,1148],{"href":20},"agent team"," on a ",[200,1151,1083],{"href":32},". Under the hood there is still a loop, tools, and stops — a harness. The product choice is whether operators must be graph authors. ",[200,1154,1156],{"href":1155},"self-service-vs-forward-deployed-ai-platforms","Self-service vs forward-deployed",[255,1158,1160],{"id":1159},"why-you-should-care","Why you should care",[190,1162,1163,1164,1168],{},"Engineers will prefer frameworks. They should. Control, portability, tests in CI. Operators and Legal will prefer a harness they can inspect without a pull request. ",[200,1165,1167],{"href":383,"rel":1166},[204],"McKinsey"," keeps showing isolated technical use without operating-model change. A beautiful LangGraph in a platform team’s repo is still isolated use if RevOps cannot attach Salesforce or refuse a write.",[190,1170,1171],{},"It affects you if:",[260,1173,1174,1177,1180,1183],{},[263,1175,1176],{},"the RFP says “must support LangChain” as if that were a control",[263,1178,1179],{},"a vendor says “model-agnostic framework” and prices seats on one flagship",[263,1181,1182],{},"you are asked to rebuild quoting and SoD because “we already have agents in Python”",[263,1184,1185],{},"security reviews the GitHub org and never reviews who can call PATCH",[190,1187,1188,1192,1193,1196,1197,1200],{},[200,1189,1191],{"href":375,"rel":1190},[204],"OWASP’s LLM Top 10"," excessive agency shows up in both: a framework that exposes every tool by default, or a product that does. The cut is not safety vs convenience. It is ",[270,1194,1195],{},"who can change the harness when it fails"," — ",[200,1198,1199],{"href":358},"harness engineering"," — and whether a fail-closed write exists.",[190,1202,1203,1206],{},[200,1204,365],{"href":363,"rel":1205},[204]," Map/Measure need a system boundary. “Our framework” is not a boundary. A named runtime with grants and logs is.",[255,1208,1210],{"id":1209},"the-practical-differences","The practical differences",[190,1212,1213,1216],{},[193,1214,1215],{},"Who authors the loop."," Framework: software engineers. Product harness: operators (and maybe SE for custom tools). If only engineers can add a sensor, you will wait on a sprint for a Legal rule.",[190,1218,1219,1222,1223,1226,1227,1230],{},[193,1220,1221],{},"Where identity lives."," Framework default: service account in ",[422,1224,1225],{},".env",". Product harness: org roster, workstream membership, OAuth grants. You ",[270,1228,1229],{},"can"," do the latter in LangGraph. You must build it.",[190,1232,1233,1236,1237,876,1241,1244,1245,1248],{},[193,1234,1235],{},"What “done” means."," Framework: your node returned. Inner product harness: tests / hook. Outer product harness: signer. Anthropic’s ",[200,1238,1240],{"href":411,"rel":1239},[204],"effective agents",[200,1242,309],{"href":307,"rel":1243},[204]," notes are about encoding done in the ",[270,1246,1247],{},"environment",". Frameworks give you the primitives; they do not know your done.",[190,1250,1251,1254,1255,1258,1259,1263],{},[193,1252,1253],{},"Portability."," Frameworks win on model swap ",[270,1256,1257],{},"if"," tools and middleware stay. Product harnesses win if they actually route and do not bury a flagship default in a seat. ",[200,1260,1262],{"href":1261},"what-is-model-routing","Model routing",". “We wrap LangChain” is not routing.",[190,1265,1266,1269],{},[193,1267,1268],{},"Eval."," Frameworks shine in unit tests of nodes. Inner harnesses shine on SWE-bench / Terminal-Bench. Enterprise harnesses shine when quote hash equals SoR row. Different CI.",[190,1271,1272,1275,1276,230],{},[193,1273,1274],{},"Time-to-first-governed-write."," Framework: months unless you already built the interceptor. Forward-deployed OS: months of people. Self-service outer harness: the product’s week-one claim — verify it. ",[200,1277,1279],{"href":1278},"how-to-run-an-enterprise-ai-proof-of-value","Proof of value",[190,1281,1282,1285,1286,1290],{},[193,1283,1284],{},"Lock-in."," Framework lock-in is code and patterns. Product lock-in is data, graph, and operating habits. Both are real. ",[200,1287,1289],{"href":1288},"how-to-solve-model-lock-in","How to solve model lock-in"," is the model slice; harness lock-in is the loop slice. Prefer quoted payloads and exportable ledgers either way.",[190,1292,1293,1294,1298,1299,1302,1303,230],{},"LangChain is not the villain. Their ",[200,1295,1297],{"href":202,"rel":1296},[204],"anatomy post"," is one of the clearer public derivations of harness parts. Use it. Then ask whether your ",[270,1300,1301],{},"deployment"," has those parts for the job you are buying — repo or company. ",[200,1304,287],{"href":286},[190,1306,1307,1308,1311],{},"Nimbus’s bet is that most operators should not author LangGraph to update a discount cap. The wiki and the gate should move. Teams that ",[270,1309,1310],{},"should"," author graphs (unique simulation, exotic tools) can still sit behind a connector. Framework inside a harness. Not a framework instead of one.",[255,1313,1315],{"id":1314},"a-decision-rule","A decision rule",[260,1317,1318,1324,1330,1339],{},[263,1319,1320,1323],{},[193,1321,1322],{},"Building a product or a unique workflow in code, with engineers on the hook:"," framework (or SDK harness) plus your own grants and evals.",[263,1325,1326,1329],{},[193,1327,1328],{},"Hiring a loop for a repository:"," inner product harness (Claude Code, Cursor, Codex). Optionally extend with a framework for custom tools.",[263,1331,1332,463,1335,1338],{},[193,1333,1334],{},"Hiring a loop for CRM/ERP/cross-department work:",[200,1336,1337],{"href":864},"enterprise agent harness"," / OS-class product. A framework is a build programme.",[263,1340,1341,1344],{},[193,1342,1343],{},"Vendor says “we are a framework and an OS”:"," make them show a failed unsigned write and an operator-attached connector. Words are cheap.",[190,1346,1347,1350],{},[200,1348,1349],{"href":848},"Build vs buy an enterprise AI OS"," is the longer form of the third bullet.",[255,1352,1354],{"id":1353},"what-each-layer-of-the-stack-is-for","What each layer of the stack is for",[190,1356,1357,1358,1361,1362,1364],{},"LangChain’s own split is the cleanest vendor-native map: use Deep Agents when you want a batteries-included ",[270,1359,1360],{},"harness","; use ",[422,1363,407],{}," when you want a minimal harness you customise with middleware; drop to LangGraph when the agent loop is the wrong shape and you need deterministic nodes mixed with agentic ones; use LangSmith to trace whatever you built. That is a builder’s menu. It does not decide whether RevOps can refuse a write.",[190,1366,1367,1368,1370,1371,1375],{},"CrewAI is a role-and-task framework. AutoGen is a conversation-of-agents framework. Semantic Kernel is Microsoft’s orchestration SDK. Pydantic AI moved toward a “harness-first” design in 2026 (capabilities as tools + hooks + instructions). None of these are wrong. All of them leave identity, SoR quoting, and operator self-service as ",[270,1369,782],{}," story unless you add them. ",[200,1372,1374],{"href":375,"rel":1373},[204],"OWASP"," will still fail you if the first graph you merge attaches every production tool “so the demo looks alive.”",[190,1377,1378,1379,1381],{},"Product harnesses fail the other way: they hide the graph so operators can work, then surprise engineers who wanted to unit-test a node. Demand an escape hatch — export traces, typed payloads, maybe a documented tool SDK — without requiring every discount cap to be a pull request. Nimbus’s bet is that the cap lives in the ",[200,1380,326],{"href":28}," and the interceptor, and that engineers who need a custom simulator put it behind a connector. Framework inside the harness.",[190,1383,1384,1389],{},[200,1385,1388],{"href":1386,"rel":1387},"https://www.thoughtworks.com/insights/articles/operating-system-enterprise-ai",[204],"Thoughtworks"," would say a company that standardises on LangGraph has invested in layer 2 (builder) and still has to build layers 3–4 (user guides/sensors, organisational ownership). A company that buys only a coding harness has a strong inner layer 2–3 and a missing outer layer 4. A company that buys an OS-class product is hoping layer 3–4 shipped. Verify with a refused write, not with a README.",[190,1391,1392,1395,1396,1400],{},[193,1393,1394],{},"Cost of the wrong cut."," Framework-first for operators: six months of platform work, then shadow copilots anyway. Product-first for a unique research loop: you will fight the product and rebuild the graph in Python by week four. ",[200,1397,1399],{"href":1398},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose coding vs enterprise"," plus this page: workspace first, then assemble vs hire.",[190,1402,1403,1406,1407,1409,1410,1412,1413,1416],{},[193,1404,1405],{},"Portability, honestly."," Frameworks make model swap easier ",[270,1408,1257],{}," you used their model interface and did not sprinkle vendor-specific tool formats through application code. Products make operator ratchet easier ",[270,1411,1257],{}," adding a gate is a UI action. Neither gives you portability of ",[270,1414,1415],{},"decisions"," unless the ledger exports. Ask for JSON of the quote and the graph, not a promise of “open.”",[190,1418,1419,1420,1423],{},"Inngest and others have argued that durable execution needs “a harness, not a framework”: retries, state, and recovery as infrastructure. That slogan is directionally right for production. It is incomplete for enterprises. Durable retries of an ",[270,1421,1422],{},"unsigned"," write are a reliable incident. The outer harness adds identity and a stop that retries must not bypass. LangGraph checkpointing is excellent loop infrastructure. It is not a Finance signer.",[190,1425,1426,1427,230],{},"A worked split: the data-science team builds a forecasting graph in LangGraph, evaluates it with their own sensors, exposes it as a tool. RevOps never opens the repo. They brief a workstream, the team calls the forecast tool under read scope, and any CRM write still quotes in the product interceptor. Framework for the specialist. Harness for the company job. Nimbus is the second box; it should consume the first as a connector, not replace the scientists’ graph. ",[200,1428,50],{"href":51},[190,1430,1431],{},"If your platform team’s OKR is “stand up LangChain,” add a second OKR: “unsigned SoR writes are impossible.” The first without the second is a framework programme. The second without any loop is a policy PDF. You need both, in that order of safety.",[190,1433,1434,1435,1437],{},"CrewAI marketing will talk about roles. Roles in a YAML file are not roster identity. If the “legal reviewer” crew member can still call the same Salesforce write tool as the “AE,” you have a framework demo of ",[200,1436,221],{"href":220}," without the contract. Ask to see the tool belt per role, then ask what happens when you remove the write tool from legal and the model asks for it anyway. The harness answer is refuse. The framework-only answer is often “we’ll prompt it.”",[255,1439,807],{"id":806},[809,1441,1443],{"id":1442},"is-langgraph-a-harness","Is LangGraph a harness?",[190,1445,1446,1447,1450],{},"It is a framework for building one. Your graph ",[270,1448,1449],{},"becomes"," a harness when it owns tool dispatch, bounds, and (for production) identity and sensors. Empty graph ≠ harness.",[809,1452,1454],{"id":1453},"is-claude-code-a-framework","Is Claude Code a framework?",[190,1456,1457,1458,230],{},"No. It is a productised inner harness. The Agent SDK is the embeddable form — closer to HaaS in ",[200,1459,1461],{"href":953,"rel":1460},[204],"Osmani’s sense",[809,1463,1465],{"id":1464},"does-mcp-replace-both","Does MCP replace both?",[190,1467,1468],{},"No. Plumbing. Hosts still need a loop and grants.",[809,1470,1472],{"id":1471},"we-already-standardised-on-crewai","We already standardised on CrewAI.",[190,1474,1475],{},"Keep it for the jobs engineers should own. Do not force RevOps to write crews for a renewal write. Put CrewAI behind a scoped tool if the outer harness needs that specialist.",[809,1477,1479],{"id":1478},"how-do-we-evaluate-a-vendor-who-wraps-langchain","How do we evaluate a vendor who wraps LangChain?",[190,1481,1482,1483,1486],{},"Ignore the wrapper. Run ",[200,1484,1485],{"href":251},"how to evaluate an agent harness",". If they cannot refuse a write, you evaluated a demo of a framework.",[809,1488,1490],{"id":1489},"where-does-nimbus-sit","Where does Nimbus sit?",[190,1492,1493,1494,1496,1497,230],{},"Productised outer harness, not a LangChain distribution. ",[200,1495,11],{"href":12},". You should still allow inner harnesses for code. ",[200,1498,1399],{"href":1398},[255,1500,869],{"id":868},[190,1502,1503,876,1505,230],{},[200,1504,861],{"href":358},[200,1506,1508],{"href":1507},"how-to-evaluate-multi-agent-platforms","How to evaluate multi-agent platforms",[255,1510,883],{"id":882},[260,1512,1513,1518,1523,1528,1534,1540,1546,1551,1556,1561,1566,1571,1576,1581],{},[263,1514,1515],{},[200,1516,897],{"href":405,"rel":1517},[204],[263,1519,1520],{},[200,1521,904],{"href":902,"rel":1522},[204],[263,1524,1525],{},[200,1526,891],{"href":202,"rel":1527},[204],[263,1529,1530],{},[200,1531,1533],{"href":1059,"rel":1532},[204],"LangChain, LangGraph overview",[263,1535,1536],{},[200,1537,1539],{"href":1053,"rel":1538},[204],"LangChain Deep Agents",[263,1541,1542],{},[200,1543,1545],{"href":1386,"rel":1544},[204],"Thoughtworks, The operating system for enterprise AI",[263,1547,1548],{},[200,1549,955],{"href":953,"rel":1550},[204],[263,1552,1553],{},[200,1554,923],{"href":411,"rel":1555},[204],[263,1557,1558],{},[200,1559,929],{"href":307,"rel":1560},[204],[263,1562,1563],{},[200,1564,941],{"href":460,"rel":1565},[204],[263,1567,1568],{},[200,1569,983],{"href":383,"rel":1570},[204],[263,1572,1573],{},[200,1574,365],{"href":363,"rel":1575},[204],[263,1577,1578],{},[200,1579,977],{"href":375,"rel":1580},[204],[263,1582,1583],{},[200,1584,989],{"href":296,"rel":1585},[204],{"title":170,"searchDepth":171,"depth":171,"links":1587},[1588,1589,1590,1591,1592,1593,1601,1602],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160},{"id":1209,"depth":171,"text":1210},{"id":1314,"depth":171,"text":1315},{"id":1353,"depth":171,"text":1354},{"id":806,"depth":171,"text":807,"children":1594},[1595,1596,1597,1598,1599,1600],{"id":1442,"depth":1001,"text":1443},{"id":1453,"depth":1001,"text":1454},{"id":1464,"depth":1001,"text":1465},{"id":1471,"depth":1001,"text":1472},{"id":1478,"depth":1001,"text":1479},{"id":1489,"depth":1001,"text":1490},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"An agent framework is a library for assembling a loop. An agent harness is the loop you can actually run — tools, stops, identity, and sensors included. LangChain helps you build one; it is not, by itself, one you can hire.","/blog/agent-harness-vs-agent-framework",{"title":1020,"description":1603},"explainer","blog/agent-harness-vs-agent-framework",[1606,1015,1609,1610],"langchain","frameworks","gXCxGnvUDv02K2_Jxwj878jpKQQXGyrZ7s3d5hNjYZA",{"id":1613,"title":1614,"archived":164,"authors":1615,"badge":1617,"body":1618,"date":1008,"definedTerm":165,"department":165,"description":2264,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":2265,"relatedHeading":165,"seo":2266,"series":1012,"sitemap":130,"status":165,"stem":2267,"subhead":165,"tags":2268,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":2270},"content/blog/eval-loops-for-enterprise-agent-harnesses.md","Eval Loops for Enterprise Agent Harnesses",[1616],{"name":184,"to":135},{"label":186},{"type":167,"value":1619,"toc":2246},[1620,1626,1654,1664,1666,1723,1731,1735,1745,1748,1784,1801,1810,1814,1825,1834,1843,1848,1875,1883,1889,1899,1908,1912,1915,1949,1957,1969,1975,1979,1982,1988,1997,2013,2019,2025,2036,2049,2054,2057,2072,2082,2084,2090,2092,2096,2099,2103,2106,2110,2116,2120,2123,2127,2134,2136,2147,2149,2157,2159],[190,1621,1029,1622,1625],{},[193,1623,1624],{},"eval loop"," for an agent harness is an independent check that the job is actually done — tests, schemas, read-backs, humans — that does not take the model’s word.",[190,1627,1628,1629,1634,1635,1640,1641,1644,1645,1648,1649,1653],{},"Coding harnesses already have a public language for this. ",[200,1630,1633],{"href":1631,"rel":1632},"https://www.swebench.com/",[204],"SWE-bench"," gives an agent a GitHub issue and grades a patch with the repo’s tests. ",[200,1636,1639],{"href":1637,"rel":1638},"https://arxiv.org/abs/2601.11868",[204],"Terminal-Bench"," (Stanford / Laude Institute) gives an agent a machine and grades the ",[270,1642,1643],{},"end state"," of a container, not the transcript. Leaderboards even report ",[193,1646,1647],{},"agent + model"," as a pair, which is the right unit: ",[200,1650,1652],{"href":405,"rel":1651},[204],"Agent = Model + Harness",". Steal that honesty. Do not steal the benchmark as your control for Salesforce.",[190,1655,1656,1657,1660,1661,1663],{},"Enterprise eval is: did the quoted CRM write match the signed payload, and can you replay who signed. A 40% Terminal-Bench score does not tell you whether Opportunity.Amount was authorised. ",[200,1658,1659],{"href":251},"How to evaluate an agent harness"," is the buying sheet. This page is the architecture of the sensor loop ",[200,1662,1199],{"href":358}," keeps tightening.",[255,1665,258],{"id":257},[260,1667,1668,1678,1688,1702,1711,1717],{},[263,1669,1670,1673,1674,1677],{},[193,1671,1672],{},"Oracle / verifier."," The independent test. SWE-bench: ",[422,1675,1676],{},"FAIL_TO_PASS"," tests. Terminal-Bench: pytest-style assertions on container state. Enterprise: SoR read-back and payload hash.",[263,1679,1680,1683,1684,1687],{},[193,1681,1682],{},"Transcript eval."," Grading the chain-of-thought. Useful for debugging. Insufficient as a release gate. Models claim victory; Anthropic’s ",[200,1685,309],{"href":307,"rel":1686},[204]," names premature victory as a failure mode.",[263,1689,1690,463,1693,467,1697,1701],{},[193,1691,1692],{},"Computational vs inferential sensors.",[200,1694,1696],{"href":946,"rel":1695},[204],"Böckeler",[200,1698,1388],{"href":1699,"rel":1700},"https://www.thoughtworks.com/en-us/insights/blog/generative-ai/harness-engineering-agent-feedback-exploring-ai-coding-sensors",[204],". Compiler vs LLM-as-judge. Prefer computational for invariants (schema, identity, hash). Use inferential for taste (narrative quality), never as the only SoR gate.",[263,1703,1704,1707,1708,230],{},[193,1705,1706],{},"LLM-as-judge."," Another stochastic component. Fine as a critic specialist. Not the signer. ",[200,1709,1710],{"href":241},"HITL architecture",[263,1712,1713,1716],{},[193,1714,1715],{},"Offline vs online eval."," Offline: golden jobs, replay. Online: shadow reads, canary writes, production sensors. You need both; most teams only have a demo recording.",[263,1718,1719,1722],{},[193,1720,1721],{},"Harness eval vs model eval."," Changing Claude vs GPT on the same tools is model eval. Changing hooks, grants, or wiki and keeping the model is harness eval. Report them separately or you will buy a new model for a missing schema check.",[190,1724,1725,1726,876,1728,1730],{},"Nimbus’s production sensor for writes is the quote-and-gate plus graph: ",[200,1727,340],{"href":40},[200,1729,23],{"href":24},". That is computational. Wiki playbooks are guides. Do not confuse a fluent Conflux draft with a passed eval.",[255,1732,1734],{"id":1733},"why-coding-benchmarks-are-the-wrong-outer-score","Why coding benchmarks are the wrong outer score",[190,1736,1737,1738,1741,1742,1744],{},"They are the ",[270,1739,1740],{},"right"," inner score. ",[200,1743,287],{"href":286},". Terminal-Bench’s design is even a lesson: grade the environment, not the story. The environment for RevOps is Salesforce, not a Docker VM with a hidden oracle.",[190,1746,1747],{},"Problems when you import SWE-bench into an enterprise RFP:",[260,1749,1750,1756,1762,1768,1778],{},[263,1751,1752,1755],{},[193,1753,1754],{},"Wrong workspace."," Patch quality ≠ payload authorisation.",[263,1757,1758,1761],{},[193,1759,1760],{},"Saturation and leakage."," Public coding benches get gamed; your CRM schema is not a public task.",[263,1763,1764,1767],{},[193,1765,1766],{},"No identity."," Benchmarks do not have a Finance signer.",[263,1769,1770,1773,1774,1777],{},[193,1771,1772],{},"No replay duty."," A leaderboard row is not ",[200,1775,371],{"href":369,"rel":1776},[204]," evidence.",[263,1779,1780,1783],{},[193,1781,1782],{},"Wrong “done.”"," Tests pass on a fixture; production field still wrong.",[190,1785,1786,1789,1790,1793,1794,1796,1797,1800],{},[200,1787,385],{"href":383,"rel":1788},[204]," is about scaling work, not about bash tasks. ",[200,1791,365],{"href":363,"rel":1792},[204]," Measure is: did the control work in ",[270,1795,782],{}," context of use. ",[200,1798,601],{"href":599,"rel":1799},[204]," wants interrupt and record, not a percentile on Terminal-Bench 2.1.",[190,1802,1803,1804,858,1807,230],{},"Use coding benches to pick an inner harness for engineering. Use quote/replay to pick an ",[200,1805,1806],{"href":864},"enterprise harness",[200,1808,1809],{"href":1398},"How to choose",[255,1811,1813],{"id":1812},"what-an-enterprise-eval-loop-actually-runs","What an enterprise eval loop actually runs",[190,1815,1816,1817,1820,1821,1824],{},"Design it like Terminal-Bench in spirit: ",[193,1818,1819],{},"end state of the systems that matter",", plus ",[193,1822,1823],{},"process constraints"," the company cannot waive.",[190,1826,1827,1830,1831,1833],{},[193,1828,1829],{},"Precondition sensors (feed-forward that is checkable)."," Required connectors attached. Roster includes the signer role. Wiki revision pinned. Spend quote accepted. If any fail, the run does not start. That is a harness eval of configuration, not of eloquence. ",[200,1832,685],{"href":220}," declaring required systems belong here.",[190,1835,1836,1839,1840,230],{},[193,1837,1838],{},"Step sensors."," Retrieval logged and in-scope (no confused-deputy dump). Tool errors do not silently retry a write. Routing used compact on extract if that is policy. ",[200,1841,1842],{"href":224},"Connector architecture",[190,1844,1845],{},[193,1846,1847],{},"Release sensors (the outer oracle).",[547,1849,1850,1853,1863,1866,1869],{},[263,1851,1852],{},"Quote is structured: object, fields, values, cardinality, hash.",[263,1854,1855,1856,1859,1860,1862],{},"Named human with the right role signed ",[270,1857,1858],{},"that"," hash (",[200,1861,229],{"href":228},").",[263,1864,1865],{},"Adapter executed only that payload.",[263,1867,1868],{},"SoR read-back equals quote (or a documented, signed delta).",[263,1870,1871,1872,230],{},"Graph (or equivalent ledger) contains brief, team, policy version, signer, payload, result. Export works without the vendor. ",[200,1873,1874],{"href":711},"Lifecycle graph",[190,1876,1877,1880,1881,230],{},[193,1878,1879],{},"Negative tests."," Reject path: SoR unchanged. Detached grant: write impossible. Wrong role: Hard/Critical cannot complete. These are the analogue of tests that must stay red. If your PoV never fails, you did not eval the harness. You evaluated a happy path. ",[200,1882,1279],{"href":1278},[190,1884,1885,1888],{},[193,1886,1887],{},"Inferential sensors (optional, never sole)."," A legal specialist flags language. A critic agent scores a narrative. Useful. If they can waive a Hard gate, you added a second stochastic writer.",[190,1890,1891,463,1894,1898],{},[193,1892,1893],{},"Human as sensor, not as folklore.",[200,1895,1897],{"href":1896},"what-is-human-in-the-loop-ai","HITL"," is a step with identity. A Slack emoji is transport. A six-month zero-reject rate is a finding: either perfect or unread.",[190,1900,1901,1902,1907],{},"Air Canada and the ",[200,1903,1906],{"href":1904,"rel":1905},"https://www.reuters.com/legal/new-york-lawyers-sanctioned-using-fake-chatgpt-cases-legal-brief-2023-06-22/",[204],"ChatGPT brief sanctions"," are eval-loop absences: no independent check before a system of record (policy page, court docket) changed.",[255,1909,1911],{"id":1910},"offline-suites-you-can-actually-keep","Offline suites you can actually keep",[190,1913,1914],{},"You will not publish a public “CRM-bench.” You can keep a private suite:",[260,1916,1917,1923,1929,1935,1943],{},[263,1918,1919,1922],{},[193,1920,1921],{},"Golden jobs."," Anonymised or sandbox SoR. Expected quote. Expected refuse.",[263,1924,1925,1928],{},[193,1926,1927],{},"Replay."," Last month’s signed write: same hash, same graph nodes.",[263,1930,1931,1934],{},[193,1932,1933],{},"Policy diffs."," Change wiki cap; next run must quote the new cap or refuse.",[263,1936,1937,1940,1941,230],{},[193,1938,1939],{},"Model swap."," Same harness, new weights: tools still dispatch; sensors still fire. That isolates model eval. ",[200,1942,1262],{"href":1261},[263,1944,1945,1948],{},[193,1946,1947],{},"Chaos."," Kill the interceptor; writes must not fail open.",[190,1950,1951,1952,1956],{},"Version the suite with the harness. ",[200,1953,1955],{"href":1954},"what-is-an-agentic-workflow","What is an agentic workflow",": the definition that ran is an input. A golden job that still “passes” after you removed the Hard gate is a broken eval, not a better model.",[190,1958,1959,1960,1963,1964,1968],{},"LangSmith, Phoenix, and similar tracing tools help ",[270,1961,1962],{},"observe"," inner and framework loops. They are not the SoR oracle. ",[200,1965,1967],{"href":1966},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",". Tracing without a hash match is a nicer transcript.",[190,1970,1971,1972,1974],{},"Nimbus should be scored on whether you can automate those golden jobs on a sandbox org: attach, refuse, sign, read-back, export. ",[200,1973,31],{"href":32}," are the fixture runner. If we cannot show a red refuse, we fail this architecture too.",[255,1976,1978],{"id":1977},"building-a-private-suite-without-a-public-crm-bench","Building a private suite without a public CRM-bench",[190,1980,1981],{},"You do not need 2,294 GitHub issues. You need a dozen jobs that hurt when they are wrong.",[190,1983,1984,1987],{},[193,1985,1986],{},"Pick three families."," (1) A write that must refuse (wrong role, missing field, detached grant). (2) A write that must match a fixture after sign-off. (3) A read-only job that must not call a write tool at all. Encode each as a workstream template or a scripted PoV. Run weekly. When a wiki cap changes, family (2) must fail until the quote updates — that is harness eval, not flaky CI.",[190,1989,1990,1993,1994,230],{},[193,1991,1992],{},"Grade environment state."," Terminal-Bench does not score the agent’s diary. Copy that. After the run, query the sandbox SoR. Compare to the signed hash. If you only grade the canvas prose, you are back to transcript eval. ",[200,1995,1996],{"href":228},"Write-back",[190,1998,1999,2002,2003,2006,2007,2010,2011,230],{},[193,2000,2001],{},"Keep model and harness scores apart."," Swap GPT vs Claude on the same golden job: if sensors still fire and hashes still match, the harness held. If a new model skips a field and the schema sensor catches it, that is a ",[270,2004,2005],{},"pass"," for the harness and a ",[270,2008,2009],{},"note"," for the model. If the sensor does not catch it, you do not need a larger model. You need a sensor. ",[200,2012,359],{"href":358},[190,2014,2015,2018],{},[193,2016,2017],{},"Report agent + model."," SWE-bench leaderboards already do this. Your internal dashboard should too: “Nimbus + routed compact/frontier” or “LangGraph + GPT + our interceptor.” Hiding the harness is how you buy a new model for a missing hook.",[190,2020,2021,2024],{},[193,2022,2023],{},"Budget the eval itself."," Inferential judges on every step will cost more than the job. Thoughtworks’ advice: deterministic checks on every transaction; probabilistic judges on critical paths. Schema and identity are every-transaction. Narrative quality is not.",[190,2026,2027,2030,2031,2035],{},[193,2028,2029],{},"What you can cite externally."," You can say you run refuse tests and read-backs. You cannot honestly say “we scored 83% on Terminal-Bench therefore Finance is safe.” ",[200,2032,2034],{"href":1637,"rel":2033},[204],"Stanford / Laude’s paper"," is a CLI benchmark. Use it for CLI harnesses.",[190,2037,2038,2039,2044,2045,230],{},"Air Canada needed a sensor on “did we emit a policy commitment.” The court docket needed a sensor on “do these citations exist.” Your suite is that instinct with fixtures. ",[200,2040,2043],{"href":2041,"rel":2042},"https://www.cbc.ca/news/canada/british-columbia/air-canada-chatbot-lawsuit-1.7116416",[204],"CBC","; ",[200,2046,2048],{"href":1904,"rel":2047},[204],"Reuters",[190,2050,2051,2052,230],{},"Online eval is the part teams skip. Offline goldens rot when the wiki moves. Shadow mode — agent quotes, human still writes, compare payloads — is an eval loop that does not need production write permission. Canary — one workstream, one object type, Hard gate, weekly refuse report — is how you learn whether operators rubber-stamp. A six-month zero-reject chart is not a quality medal. It is a sensor that may be dead. ",[200,2053,1897],{"href":1896},[190,2055,2056],{},"Compare this to CI for software. You would not ship because the developer said the tests passed on their laptop. You would not replace CI with an LLM that reads the diff and scores “looks good.” You might add that LLM as a critic. Enterprise write eval is CI for mutations. Nimbus’s gate is the required check; your SoR read-back is the assertion file. If we only store the transcript, we are the laptop. Demand the assertion.",[190,2058,2059,2064,2065,2068,2069,230],{},[200,2060,2063],{"href":2061,"rel":2062},"https://docs.langchain.com/langsmith/observability",[204],"LangSmith"," and similar are the right place to ",[270,2066,2067],{},"debug"," traces for framework and inner loops. Export those traces into your golden runner; do not let the tracing UI become the only evidence for audit. Auditors will ask for the hash and the signer. ",[200,2070,2071],{"href":1966},"How to evaluate AI audit",[190,2073,2074,2075,2077,2078,2081],{},"Do not wait for a consortium bench. Your suite is a competitive advantage if it encodes ",[270,2076,782],{}," caps and objects. Share the ",[270,2079,2080],{},"method"," (refuse, read-back, replay) in the RFP. Keep the fixtures. Vendors who cannot run against your sandbox are not ready for your SoR, however they score on Terminal-Bench.",[255,2083,793],{"id":792},[190,2085,2086,2087,2089],{},"The product’s eval spine is: NTU quote before the run, scoped retrieval, canvas artefacts, write quotes, tiered gates, graph commit. Sensors you should still add: your own SoR read-back in the sandbox, your own golden files (the analogue of pytest). The platform cannot know your “correct Amount” without your oracle. Terminal-Bench ships oracles per task. You must ship oracles per job. That is ",[200,2088,1199],{"href":358},", not a missing model.",[255,2091,807],{"id":806},[809,2093,2095],{"id":2094},"can-we-use-an-llm-as-judge-on-the-quote","Can we use an LLM-as-judge on the quote?",[190,2097,2098],{},"As a critic, yes. As the only signer, no. Computational match of fields is cheap and stable.",[809,2100,2102],{"id":2101},"do-we-wait-for-an-industry-enterprise-swe-bench","Do we wait for an industry “enterprise SWE-bench”?",[190,2104,2105],{},"You would still need private oracles. Start this quarter with sandbox read-backs.",[809,2107,2109],{"id":2108},"our-vendor-only-shares-swe-bench","Our vendor only shares SWE-bench.",[190,2111,2112,2113,230],{},"File as inner evidence. Demand refuse/replay for outer. ",[200,2114,2115],{"href":251},"Evaluate the harness",[809,2117,2119],{"id":2118},"is-tracing-enough-for-iso-42001","Is tracing enough for ISO 42001?",[190,2121,2122],{},"Traces help Measure. You still need Manage: a control that fired. A pretty trace of an unsigned write is a better incident report.",[809,2124,2126],{"id":2125},"how-does-this-relate-to-agent-teams-vs-single-agents","How does this relate to agent teams vs single agents?",[190,2128,2129,2130,230],{},"Teams add hand-off evals (typed artefacts). They do not replace the write oracle. ",[200,2131,2133],{"href":2132},"how-to-evaluate-agent-teams-vs-single-agents","How to evaluate agent teams",[809,2135,853],{"id":852},[190,2137,2138,858,2141,858,2145,230],{},[200,2139,195],{"href":2140},"agent-harness-architecture",[200,2142,2144],{"href":2143},"how-to-evaluate-write-back-governance","How to evaluate write-back governance",[200,2146,861],{"href":358},[255,2148,869],{"id":868},[190,2150,2151,876,2153,230],{},[200,2152,1967],{"href":1966},[200,2154,2156],{"href":2155},"write-back-governance-for-systems-of-record","Write-back governance for systems of record",[255,2158,883],{"id":882},[260,2160,2161,2166,2172,2177,2182,2187,2192,2197,2203,2208,2213,2218,2223,2229,2235,2241],{},[263,2162,2163],{},[200,2164,1633],{"href":1631,"rel":2165},[204],[263,2167,2168],{},[200,2169,2171],{"href":1637,"rel":2170},[204],"Terminal-Bench (arXiv:2601.11868)",[263,2173,2174],{},[200,2175,897],{"href":405,"rel":2176},[204],[263,2178,2179],{},[200,2180,891],{"href":202,"rel":2181},[204],[263,2183,2184],{},[200,2185,929],{"href":307,"rel":2186},[204],[263,2188,2189],{},[200,2190,923],{"href":411,"rel":2191},[204],[263,2193,2194],{},[200,2195,948],{"href":946,"rel":2196},[204],[263,2198,2199],{},[200,2200,2202],{"href":1699,"rel":2201},[204],"Thoughtworks, Harness engineering and agent feedback",[263,2204,2205],{},[200,2206,983],{"href":383,"rel":2207},[204],[263,2209,2210],{},[200,2211,365],{"href":363,"rel":2212},[204],[263,2214,2215],{},[200,2216,966],{"href":369,"rel":2217},[204],[263,2219,2220],{},[200,2221,601],{"href":599,"rel":2222},[204],[263,2224,2225],{},[200,2226,2228],{"href":2041,"rel":2227},[204],"CBC, Air Canada chatbot lawsuit",[263,2230,2231],{},[200,2232,2234],{"href":1904,"rel":2233},[204],"Reuters, ChatGPT legal brief sanctions",[263,2236,2237],{},[200,2238,2240],{"href":2061,"rel":2239},[204],"LangSmith observability",[263,2242,2243],{},[200,2244,989],{"href":296,"rel":2245},[204],{"title":170,"searchDepth":171,"depth":171,"links":2247},[2248,2249,2250,2251,2252,2253,2254,2262,2263],{"id":257,"depth":171,"text":258},{"id":1733,"depth":171,"text":1734},{"id":1812,"depth":171,"text":1813},{"id":1910,"depth":171,"text":1911},{"id":1977,"depth":171,"text":1978},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":2255},[2256,2257,2258,2259,2260,2261],{"id":2094,"depth":1001,"text":2095},{"id":2101,"depth":1001,"text":2102},{"id":2108,"depth":1001,"text":2109},{"id":2118,"depth":1001,"text":2119},{"id":2125,"depth":1001,"text":2126},{"id":852,"depth":1001,"text":853},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Coding agents can be scored on SWE-bench and Terminal-Bench. An enterprise harness is scored on whether the executed write matched the signed payload — independent sensors, not the model’s own claim that it was done.","/blog/eval-loops-for-enterprise-agent-harnesses",{"title":1614,"description":2264},"blog/eval-loops-for-enterprise-agent-harnesses",[1012,1015,2269,340],"evaluation","CM1xXUF5XL1F03Rrcyw9I74LA1IpOJ_4H95nVjEY3r4",{"id":2272,"title":2273,"archived":164,"authors":2274,"badge":2276,"body":2277,"date":1008,"definedTerm":165,"department":165,"description":2886,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":2887,"relatedHeading":165,"seo":2888,"series":1606,"sitemap":130,"status":165,"stem":2889,"subhead":165,"tags":2890,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":2893},"content/blog/inner-vs-outer-agent-harness.md","Inner vs Outer Agent Harness",[2275],{"name":184,"to":135},{"label":1024},{"type":167,"value":2278,"toc":2869},[2279,2293,2312,2330,2345,2347,2418,2438,2440,2443,2445,2459,2476,2479,2500,2510,2514,2529,2538,2547,2556,2567,2577,2592,2601,2614,2618,2621,2631,2641,2650,2654,2664,2673,2682,2688,2693,2704,2707,2709,2713,2716,2720,2727,2731,2742,2746,2752,2756,2762,2764,2775,2777,2783,2785],[190,2280,1029,2281,2284,2285,2288,2289,2292],{},[193,2282,2283],{},"inner agent harness"," is the runtime around a model for a developer and a codebase. An ",[193,2286,2287],{},"outer agent harness"," is the runtime around a model for operators and live business systems. Same equation — ",[200,2290,1652],{"href":405,"rel":2291},[204]," — different workspace, different sensors, different stop.",[190,2294,2295,2298,2299,2302,2303,2308,2309,230],{},[200,2296,1696],{"href":946,"rel":2297},[204]," already uses “outer harness” for the controls ",[270,2300,2301],{},"users"," add around a coding agent (guides, sensors) as distinct from the vendor’s built-in loop. ",[200,2304,2307],{"href":2305,"rel":2306},"https://addyosmani.com/blog/own-the-outer-loop/",[204],"Addy Osmani"," tells engineers to own the outer loop of investigate → implement → verify so accountability does not dissolve into the model. This article borrows those words and draws the cut enterprises actually buy: ",[193,2310,2311],{},"repo versus company",[190,2313,2314,2315,217,2318,467,2320,217,2322,2326,2327,2329],{},"Claude Code, Cursor, and Codex are excellent inner harnesses. They sandboxes, ",[422,2316,2317],{},"apply_patch",[422,2319,646],{},[422,2321,434],{},[200,2323,2325],{"href":460,"rel":2324},[204],"hooks",", and tests. Palantir AIP, Salesforce Agentforce, and OS-class products such as Nimbus are outer harnesses: ",[200,2328,216],{"href":215},", connectors, named signers, a decision record. Confusing them is how Legal is asked to “just use Cursor on the Salesforce repo” and how engineering is asked to “approve CRM writes in a coding agent.”",[190,2331,2332,2335,2336,2340,2341,2344],{},[200,2333,2334],{"href":1398},"How to choose between a coding harness and an enterprise harness"," is the buying version of this page. ",[200,2337,2339],{"href":2338},"how-to-choose-between-a-copilot-and-a-work-os","How to choose between a copilot and a work OS"," is the adjacent cut (personal assistant versus departmental work). Inner/outer is about ",[270,2342,2343],{},"which loop you are hiring",", not whether the UI looks like chat.",[255,2346,258],{"id":257},[260,2348,2349,2355,2361,2371,2383,2392,2405],{},[263,2350,2351,2354],{},[193,2352,2353],{},"Inner loop (classic SE)."," Edit, build, test on a developer’s machine. Fast. Local. The coding-agent inner harness lives here: shell, files, compiler.",[263,2356,2357,2360],{},[193,2358,2359],{},"Outer loop (classic SE)."," PR, CI, review, release. Osmani’s “own the outer loop” is this accountability layer for agentic coding. Still software.",[263,2362,2363,2366,2367,2370],{},[193,2364,2365],{},"Inner harness (this article)."," Vendor + user controls for a ",[193,2368,2369],{},"repository workspace",": Claude Code, Cursor, Codex. Eval: tests, Terminal-Bench, SWE-bench.",[263,2372,2373,2376,2377,2380,2381,230],{},[193,2374,2375],{},"Outer harness (this article)."," Controls for a ",[193,2378,2379],{},"company workspace",": jobs, systems of record, people who may sign. Eval: quoted write, identity, ledger. An ",[200,2382,1337],{"href":864},[263,2384,2385,2388,2389,230],{},[193,2386,2387],{},"Guides vs sensors."," Feed-forward markdown versus feedback from tools. Inner: lint and pytest. Outer: schema of a Salesforce payload and a Hard gate. See ",[200,2390,2391],{"href":358},"what is harness engineering",[263,2393,2394,2397,2398,2401,2402,230],{},[193,2395,2396],{},"CLAUDE.md / AGENTS.md."," Inner guides. ",[200,2399,765],{"href":836,"rel":2400},[204]," is explicit: files are context; hooks are deterministic. A company wiki is the outer analogue of those files — asserted policy, not a repo README. See ",[200,2403,2404],{"href":443},"What is a company wiki for AI agents",[263,2406,2407,2410,2411,2414,2415,2417],{},[193,2408,2409],{},"Write gate."," Inner: hook denies ",[422,2412,2413],{},"rm"," or force-push. Outer: ",[200,2416,1079],{"href":228}," — adapter cannot mutate CRM until a named role signs the quote.",[190,2419,2420,2421,2423,2424,217,2426,2428,2429,2431,2432,2423,2434,2437],{},"Nimbus is built as an outer harness: ",[200,2422,326],{"href":28}," instead of only ",[422,2425,434],{},[200,2427,225],{"href":51}," instead of only a local shell, ",[200,2430,340],{"href":40}," instead of only a pre-commit hook, ",[200,2433,23],{"href":24},[422,2435,2436],{},"git log",". Engineering should still run Claude Code. Those products should not share a write path to NetSuite.",[255,2439,1160],{"id":1159},[190,2441,2442],{},"Demos collapse the cut. Both products answer a question. Both call tools. Both show a transcript. The evaluation is the workspace.",[190,2444,1171],{},[260,2446,2447,2450,2453,2456],{},[263,2448,2449],{},"Security asks whether the coding agent’s MCP server can reach production Salesforce",[263,2451,2452],{},"RevOps wants “an agent” and is shown a SWE-bench slide",[263,2454,2455],{},"Engineering wants Cursor and is told to wait for the enterprise OS",[263,2457,2458],{},"You already have both, and they silently write to the same object",[190,2460,2461,2465,2466,2471,2472,2475],{},[200,2462,2464],{"href":383,"rel":2463},[204],"McKinsey’s 2025 State of AI"," describes agentic systems as an organisational design problem. Inner harnesses scale developer throughput. They do not, by themselves, scale governed operations. ",[200,2467,2470],{"href":2468,"rel":2469},"https://hai.stanford.edu/ai-index/2025-ai-index-report",[204],"Stanford HAI’s 2025 AI Index"," maps how fast coding-agent tooling moved. Speed in the repo is not a substitute for ",[200,2473,601],{"href":599,"rel":2474},[204]," oversight on systems that affect customers and money.",[190,2477,2478],{},"Two failure modes:",[547,2480,2481,2491],{},[263,2482,2483,2486,2487,2490],{},[193,2484,2485],{},"Outer job, inner harness."," A pricing change drafted in Cursor with an MCP Salesforce tool. Tests pass on a fixture. Production Amount changes. ",[422,2488,2489],{},"git blame"," does not name the signer. You used a repo loop on a company record.",[263,2492,2493,2496,2497,2499],{},[193,2494,2495],{},"Inner job, outer harness."," “Rewrite this function” opened as a cross-department ",[200,2498,1083],{"href":32}," with a Critical gate. Engineers will route around it. You used a company loop on a compile.",[190,2501,2502,2505,2506,2509],{},[200,2503,365],{"href":363,"rel":2504},[204]," Map step: know the context of use. Inner and outer are different contexts. ",[200,2507,966],{"href":369,"rel":2508},[204]," wants controls matched to that context. One harness policy for “all AI” is how both jobs get the wrong stop.",[255,2511,2513],{"id":2512},"what-each-harness-actually-owns","What each harness actually owns",[190,2515,2516,2518,2519,2522,2523,2525,2526,2528],{},[193,2517,282],{}," Inner: a checkout, often sandboxed. Anthropic’s ",[200,2520,309],{"href":307,"rel":2521},[204]," keeps progress in git and files because the workspace ",[270,2524,272],{}," the filesystem. Outer: a job folder with people, budget, and attached systems — a ",[200,2527,1083],{"href":215},". Files may appear as artefacts. They are not the system of record.",[190,2530,2531,2534,2535,2537],{},[193,2532,2533],{},"Identity."," Inner: the developer’s machine credentials, a repo token, maybe a sandbox role. Outer: org roster, workstream membership, named approver. The model is not the principal. ",[200,2536,879],{"href":224}," is the outer identity plane.",[190,2539,2540,2543,2544,2546],{},[193,2541,2542],{},"Tools."," Inner: shell, editor, tests, browser, maybe MCP to docs. Outer: CRM, ERP, warehouse, ticket systems, mail — default read, write as a separate plane. ",[200,2545,758],{"href":757}," can sit under both. The grant must not.",[190,2548,2549,2552,2553,2555],{},[193,2550,2551],{},"Guides."," Inner: ",[422,2554,434],{},", skills, directory-local rules. Outer: company wiki, playbooks versioned with the run. Mixing them is useful (engineering conventions in the repo; discount policy in the wiki). Collapsing them is how a style guide becomes “legal approval.”",[190,2557,2558,2561,2562,2566],{},[193,2559,2560],{},"Sensors."," Inner: typechecker, unit tests, CI, architecture tests. Böckeler and ",[200,2563,2565],{"href":1699,"rel":2564},[204],"Thoughtworks on sensors",". Outer: payload schema, blast-radius cardinality, maker-checker, exportable ledger. A passing pytest does not mean Opportunity.Stage was authorised.",[190,2568,2569,2572,2573,2576],{},[193,2570,2571],{},"Stop."," Inner: tests red, hook exit 2, max steps, human in the IDE. Outer: wait-for-named-signer, missing connector, budget, reject. ",[200,2574,2575],{"href":1896},"Human-in-the-loop"," in a coding agent is “the developer kept going.” HITL in an outer harness is a first-class step with identity.",[190,2578,2579,2552,2581,217,2584,2587,2588,2591],{},[193,2580,1268],{},[200,2582,1633],{"href":1631,"rel":2583},[204],[200,2585,1639],{"href":1637,"rel":2586},[204],", your suite. Outer: replay the signer; compare quote to SoR; see ",[200,2589,2590],{"href":503},"eval loops",". Leaderboard scores are not a SOX control.",[190,2593,2594,2597,2598,2600],{},[193,2595,2596],{},"Memory."," Inner: files, commits, session transcripts, memory files the next coding session loads. Outer: wiki + ",[200,2599,23],{"href":711}," so next quarter’s operator can ask why a field changed. Chat logs of a coding session are not institutional memory for RevOps.",[190,2602,2603,2604,2606,2607,2610,2611,2613],{},"Nimbus’s ",[200,2605,221],{"href":20}," sit on the outer side: mandates, required connectors, approval triggers. You can still ",[270,2608,2609],{},"use"," an inner harness as a bounded tool behind a connector (for example a coding agent that only opens a draft PR). Do not let that inner harness become the orchestrator of record for a CRM write. ",[200,2612,276],{"href":236}," says the same thing with specialists: hands are not roles.",[255,2615,2617],{"id":2616},"how-they-should-sit-together","How they should sit together",[190,2619,2620],{},"Most companies need both. That is not a hedge. It is how software and operations already split.",[190,2622,2623,2626,2627,2630],{},[193,2624,2625],{},"Pattern that works."," Engineers use Cursor or Claude Code on application repos. CI remains the merge sensor. Separately, RevOps and Finance run outer-harness jobs on Salesforce and NetSuite. If a coding agent must touch a live business system, it proposes an artefact; the outer harness quotes and gates the write. Two writers to the same object without a single quote is the failure ",[200,2628,2629],{"href":236},"multi-agent architecture"," already names.",[190,2632,2633,2636,2637,230],{},[193,2634,2635],{},"Pattern that fails."," One MCP mesh with production tokens, used from the IDE and from the chatbot and from the OS. Confused deputy. ",[200,2638,2640],{"href":2639},"mcp-for-enterprise-integrations","MCP for enterprise integrations",[190,2642,2643,2646,2647,230],{},[193,2644,2645],{},"Thoughtworks’ four layers"," — model, builder harness, user harness, organisational harness — map cleanly: Claude Code is builder + user on the inner side; the organisational layer is the outer operating model. Nimbus is one productisation of that outer layer, not the only one. AIP is a programme-shaped outer harness. Agentforce is CRM-anchored. Score scope and time-to-value separately. See ",[200,2648,2649],{"href":1155},"self-service vs forward-deployed",[255,2651,2653],{"id":2652},"a-week-that-uses-both","A week that uses both",[190,2655,2656,2657,2659,2660,2663],{},"Monday an engineer uses Cursor to fix a pricing calculator in the billing service. ",[422,2658,434],{}," says no raw SQL in the request path. A hook blocks ",[422,2661,2662],{},"git push --force",". CI runs the unit suite. The PR is the artefact. CODEOWNERS signs the merge. That is a complete inner story. SWE-bench is relevant only as a vendor quality signal for the coding tool, not as a control.",[190,2665,2666,2667,2669,2670,2672],{},"Tuesday RevOps needs the list price on twenty renewals updated after Legal changed the cap in the playbook. The artefact is Salesforce. The signer is a named RevOps lead. The sensor is: quoted fields, hash, read-back. If Tuesday’s job is opened as a Cursor session with an MCP Salesforce server using a shared integration user, you have imported Monday’s workspace into Tuesday’s system of record. ",[422,2668,2436],{}," will not name the RevOps lead. ",[200,2671,1996],{"href":228}," did not fire because the inner harness does not have that interceptor.",[190,2674,2675,2676,2678,2679,2681],{},"Wednesday someone proposes “one agent for everything.” The honest architecture is: Monday’s harness stays. Tuesday’s job runs on an outer harness — in Nimbus, a ",[200,2677,1083],{"href":32}," with the CRM connector, the wiki revision that contains the new cap, a Hard gate. If the calculator ",[270,2680,422],{}," must change as well, the outer job can spawn a bounded inner step that opens a draft PR. Two artefacts, two sensors, one company rule: unsigned SoR writes are impossible from either loop.",[190,2683,2684,2685,230],{},"Thursday Security reviews MCP. The question is not “is MCP approved.” It is “which workspace may this server mutate.” Inner: sandbox and repo. Outer: workstream grant. Same protocol, different identity box. ",[200,2686,2687],{"href":2639},"MCP for enterprise",[190,2689,2690,2691,230],{},"Friday you look at evals. Engineering posts a Terminal-Bench plot for the coding vendor. Finance asks who signed Amount. Those are not competing dashboards. They are different oracles. ",[200,2692,504],{"href":503},[190,2694,2695,2698,2699,2703],{},[200,2696,1388],{"href":1386,"rel":2697},[204]," would call Monday layers 2–3 on a builder harness, Tuesday a delegation question on layer 4, and “one agent” a way to skip layer 4. ",[200,2700,2702],{"href":2305,"rel":2701},[204],"Osmani"," would say engineering still owns verify-and-merge on Monday. Neither author is selling Nimbus. Both are describing why the cut exists.",[190,2705,2706],{},"If you only fund inner harnesses, Tuesday happens in paste and Slack. If you only fund outer harnesses, Monday happens in unsanctioned Cursor anyway. Fund both. Bind writes.",[255,2708,807],{"id":806},[809,2710,2712],{"id":2711},"is-cursor-an-enterprise-harness-if-we-sso-it","Is Cursor an enterprise harness if we SSO it?",[190,2714,2715],{},"SSO is admin control. It does not quote a NetSuite journal or bind a Finance signer. Cursor can be an inner harness in an enterprise. That is not the same as an outer harness.",[809,2717,2719],{"id":2718},"can-claude-code-hooks-replace-write-back-governance","Can Claude Code hooks replace write-back governance?",[190,2721,2722,2723,2726],{},"They can replace ",[270,2724,2725],{},"some"," inner invariants (dangerous bash). They do not give you a payload in the language of Salesforce, a roster-aware approver, or an exportable operations ledger. Different workspace.",[809,2728,2730],{"id":2729},"should-we-ban-coding-agents-until-the-os-is-live","Should we ban coding agents until the OS is live?",[190,2732,2733,2734,2737,2738,230],{},"Usually no. Ban unsigned writes to systems of record from ",[270,2735,2736],{},"any"," agent, inner or outer. Let inner harnesses keep compiling. ",[200,2739,2741],{"href":2740},"how-to-solve-unapproved-crm-writes-from-ai","How to solve unapproved CRM writes from AI",[809,2743,2745],{"id":2744},"where-does-a-copilot-fit","Where does a copilot fit?",[190,2747,2748,2749,230],{},"A copilot is often not a full inner harness — no repo loop, no tests. Personal throughput. Keep it for mail. Do not give it the CRM write token. ",[200,2750,2751],{"href":2338},"Copilot vs work OS",[809,2753,2755],{"id":2754},"is-nimbus-trying-to-replace-claude-code","Is Nimbus trying to replace Claude Code?",[190,2757,2758,2759,2761],{},"No. Different workspace. Nimbus is the company loop; Claude Code is the repo loop. ",[200,2760,11],{"href":12}," is the product map. This page is the architectural cut.",[809,2763,853],{"id":852},[190,2765,2766,858,2768,2771,2772,2774],{},[200,2767,865],{"href":864},[200,2769,2770],{"href":593},"Agent harness vs agent framework"," if you are assembling rather than hiring. ",[200,2773,1106],{"href":246}," for the base noun.",[255,2776,869],{"id":868},[190,2778,2779,876,2781,230],{},[200,2780,861],{"href":358},[200,2782,195],{"href":2140},[255,2784,883],{"id":882},[260,2786,2787,2792,2797,2803,2808,2813,2818,2823,2828,2833,2838,2843,2849,2854,2859,2864],{},[263,2788,2789],{},[200,2790,897],{"href":405,"rel":2791},[204],[263,2793,2794],{},[200,2795,948],{"href":946,"rel":2796},[204],[263,2798,2799],{},[200,2800,2802],{"href":2305,"rel":2801},[204],"Addy Osmani, Own the outer loop",[263,2804,2805],{},[200,2806,955],{"href":953,"rel":2807},[204],[263,2809,2810],{},[200,2811,1545],{"href":1386,"rel":2812},[204],[263,2814,2815],{},[200,2816,929],{"href":307,"rel":2817},[204],[263,2819,2820],{},[200,2821,935],{"href":836,"rel":2822},[204],[263,2824,2825],{},[200,2826,941],{"href":460,"rel":2827},[204],[263,2829,2830],{},[200,2831,1633],{"href":1631,"rel":2832},[204],[263,2834,2835],{},[200,2836,2171],{"href":1637,"rel":2837},[204],[263,2839,2840],{},[200,2841,983],{"href":383,"rel":2842},[204],[263,2844,2845],{},[200,2846,2848],{"href":2468,"rel":2847},[204],"Stanford HAI, 2025 AI Index",[263,2850,2851],{},[200,2852,365],{"href":363,"rel":2853},[204],[263,2855,2856],{},[200,2857,966],{"href":369,"rel":2858},[204],[263,2860,2861],{},[200,2862,601],{"href":599,"rel":2863},[204],[263,2865,2866],{},[200,2867,989],{"href":296,"rel":2868},[204],{"title":170,"searchDepth":171,"depth":171,"links":2870},[2871,2872,2873,2874,2875,2876,2884,2885],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160},{"id":2512,"depth":171,"text":2513},{"id":2616,"depth":171,"text":2617},{"id":2652,"depth":171,"text":2653},{"id":806,"depth":171,"text":807,"children":2877},[2878,2879,2880,2881,2882,2883],{"id":2711,"depth":1001,"text":2712},{"id":2718,"depth":1001,"text":2719},{"id":2729,"depth":1001,"text":2730},{"id":2744,"depth":1001,"text":2745},{"id":2754,"depth":1001,"text":2755},{"id":852,"depth":1001,"text":853},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"An inner agent harness runs a developer and a repository — CLAUDE.md, hooks, tests. An outer harness runs the company — wiki, connectors, write gates, and a ledger. Most enterprises need both.","/blog/inner-vs-outer-agent-harness",{"title":2273,"description":2886},"blog/inner-vs-outer-agent-harness",[1606,1015,2891,2892],"coding-agents","enterprise-ai","_il0EOS7LGYcsPqRPjgGBR-hd3QC3P9SSGIQjCv3Y0M",{"id":2895,"title":2896,"archived":164,"authors":2897,"badge":2899,"body":2900,"date":3101,"definedTerm":165,"department":165,"description":3102,"extension":173,"eyebrow":165,"faqHeader":3103,"faqs":3106,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":3116,"relatedHeading":165,"seo":3117,"series":1606,"sitemap":130,"status":165,"stem":3118,"subhead":165,"tags":3119,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":3122},"content/blog/multiplayer-ai-and-multi-agent-ai.md","Multiplayer AI vs multi-agent AI: what is the difference?",[2898],{"name":184,"to":135},{"label":1024},{"type":167,"value":2901,"toc":3094},[2902,2905,2908,2913,2917,2920,2923,2926,2929,2937,2941,2944,2947,2950,2953,2963,2969,2973,2976,2979,2990,2993,3001,3004,3008,3011,3014,3031,3041,3050,3053,3057,3060,3077,3084,3091],[190,2903,2904],{},"Multiplayer AI is people and AI on the same job at the same time. Multi-agent AI is more than one model handing work to another. They are not the same product, and they fail in different places. You can have both — several models staffing steps inside one shared room — but buying a swarm is not the same as buying a room.",[190,2906,2907],{},"You should care if a demo shows agents passing tickets to each other and you still cannot name who would refuse a write to a live system. This is a useful distinction, not a verdict on agent platforms. Plenty of teams will keep specialists for retrieval or checks. The question is whether people still share the job.",[190,2909,2910,2912],{},[200,2911,494],{"href":493}," is the cast-of-models definition. This page keeps that word apart from multiplayer: people and AI on the same job at the same time.",[255,2914,2916],{"id":2915},"what-is-the-difference-between-multiplayer-ai-and-multi-agent-ai","What is the difference between multiplayer AI and multi-agent AI?",[190,2918,2919],{},"Multiplayer answers: who is in the room, what they can see, and who can halt a change. The unit is the job. Finance and sales can open the same brief while the model drafts.",[190,2921,2922],{},"Multi-agent answers: how work is split between models. One specialist retrieves. Another drafts. A third “reviews.” The unit is the graph — the sequence of model calls.",[190,2924,2925],{},"A simple check: if you remove every extra model and two departments still cannot share the files and the stop, you never had multiplayer. If you remove the second human and the run still completes in private, you had a personal tool with extra model calls.",[190,2927,2928],{},"Write-back is when AI changes a live system. In a multiplayer setup, one job holds one payload — the exact change — and a named signer. In a multi-agent setup, several writers can exist unless you bind them to that same stop. Fail-closed means if nobody approves, nothing happens. That rule belongs to a person on the roster, not to the orchestrator.",[190,2930,2931,2932,2936],{},"McKinsey’s ",[200,2933,2935],{"href":383,"rel":2934},[204],"State of AI"," (2025) found that most organisations using AI are still piloting. A common pilot is either one copilot or a small agent demo. Neither automatically creates a shared job.",[255,2938,2940],{"id":2939},"why-does-that-distinction-matter","Why does that distinction matter?",[190,2942,2943],{},"It matters when something goes out wrong and you need a name.",[190,2945,2946],{},"In multiplayer AI, a named person owns the finish line: the model drafts, and a human on the roster signs or rejects. If the artefact is wrong, you can say who was on the job, including which AI role, and who was allowed to stop it.",[190,2948,2949],{},"In multi-agent AI, accountability is easy to lose. Each specialist did “its step.” The human who started the run may not have seen the intermediate draft. A log can show that agent B called agent C at 14:03. It does not show that finance agreed.",[190,2951,2952],{},"Orchestration decides sequence. Accountability is a person with a duty who can refuse at the moment a live system is about to change. A node labelled “human review” is not a name until you can say whose name, on this job, for which class of write.",[190,2954,2955,2959,2960,2962],{},[200,2956,2958],{"href":2957},"rbac-for-enterprise-ai","RBAC for enterprise AI"," is that list. ",[200,2961,474],{"href":228}," is the companion for the write itself.",[190,2964,2965,2966,2968],{},"The harness — the tools, stops, and checks around the model — is how a cast of specialists stays bounded. ",[200,2967,359],{"href":358}," is the guide to that environment.",[255,2970,2972],{"id":2971},"when-do-you-need-several-people-versus-several-models","When do you need several people versus several models?",[190,2974,2975],{},"You need several people when more than one owner must stand on the result, or when a handover will happen, or when a customer-facing sentence can leave.",[190,2977,2978],{},"You need several models when the hand-off already exists between human roles and you want a narrower tool for each step. Useful examples:",[260,2980,2981,2984,2987],{},[263,2982,2983],{},"A research pass that must not share an identity with the agent drafting customer email.",[263,2985,2986],{},"A finance check that should not be able to send mail, even by accident.",[263,2988,2989],{},"A long retrieval over many files that a person will then judge on the job.",[190,2991,2992],{},"Separation of duties is the useful idea. The specialist that recommends a CRM update is not the principal that executes it. Multiplayer AI still puts a human on the execute step.",[190,2994,2995,2996,3000],{},"You do not need a swarm to summarise your own notes. That is a ",[200,2997,2999],{"href":2998},"collaborative-ai-and-personal-assistants","personal assistant",". You do not need a second department on a private brainstorm. You do need both people and a stop when the output can change CRM, a journal, or a message a customer will keep.",[190,3002,3003],{},"A disagreement is a good test. Sales’ specialist wants to send. Legal’s specialist wants to hold. If the orchestrator averages them, or picks the last speaker, you do not have a stop. You have a race. Multiplayer AI makes the human with the duty the one who decides.",[255,3005,3007],{"id":3006},"how-do-you-talk-about-this-with-a-vendor","How do you talk about this with a vendor?",[190,3009,3010],{},"Ask to see the room and the cast as two demos, not one slide.",[190,3012,3013],{},"Useful questions:",[260,3015,3016,3019,3022,3025,3028],{},[263,3017,3018],{},"Can a second department join live, see the same brief, and reject a proposal?",[263,3020,3021],{},"If we remove the person who started the run, can someone else still refuse a write?",[263,3023,3024],{},"When two specialists disagree, who decides — a person with a name, or the graph?",[263,3026,3027],{},"Can we open the intermediate draft tomorrow, including a stored no?",[263,3029,3030],{},"Is the write identity a named human role, or a shared service credential?",[190,3032,3033,3037,3038,3040],{},[200,3034,3036],{"href":3035},"what-auditors-are-asking-for","What auditors are asking for"," is the evidence cut. A common first rule is: do not give the swarm a production write token so the demo looks complete. ",[200,3039,474],{"href":228}," is that checklist.",[190,3042,3043,3044,3049],{},"Stanford HAI’s ",[200,3045,3048],{"href":3046,"rel":3047},"https://hai.stanford.edu/ai-index",[204],"AI Index"," (2025) tracks adoption, investment, and incident reporting. Incident stories are easier to learn from when you can name the job and the signer, not only the model family.",[190,3051,3052],{},"If the vendor can only show a happy path of agents completing a ticket, ask for a specialist disagreement and a human rejection. That is a fair request. You may still buy the swarm for staffing. You will know whether you also bought a workplace.",[255,3054,3056],{"id":3055},"how-do-you-start-without-buying-a-new-stack","How do you start without buying a new stack?",[190,3058,3059],{},"Bind what you already have to one job.",[547,3061,3062,3065,3068,3071,3074],{},[263,3063,3064],{},"Pick a recurring job that already has two owners (a weekly exception, a clause check, a forecast update).",[263,3066,3067],{},"Put the brief and two files in one place those people can both open.",[263,3069,3070],{},"If you already run specialists, let them draft into that place. Keep the intermediate draft visible.",[263,3072,3073],{},"Name who can sign a write. Keep the connection read-only until that name exists.",[263,3075,3076],{},"After two cycles, ask: did we fail because we needed another model, or because the second person could not see the file?",[190,3078,2603,3079,876,3081,3083],{},[200,3080,216],{"href":32},[200,3082,340],{"href":40}," are one attempt at that shape. You can start with a shared folder, a ticket, and a written stop if that is what you have.",[190,3085,3086,3090],{},[200,3087,3089],{"href":3088},"nimbus-vs-paperclip","Nimbus vs Paperclip"," is a vendor-shaped version of the same cut: governing what agents do inside one platform is not the same as two departments finishing a signed forecast in your CRM.",[190,3092,3093],{},"Keep the words apart because they help you buy the right next thing. Multiplayer is the room. Multi-agent is the cast. Adding to the cast is a staffing decision. The room is what owns the result.",{"title":170,"searchDepth":171,"depth":171,"links":3095},[3096,3097,3098,3099,3100],{"id":2915,"depth":171,"text":2916},{"id":2939,"depth":171,"text":2940},{"id":2971,"depth":171,"text":2972},{"id":3006,"depth":171,"text":3007},{"id":3055,"depth":171,"text":3056},"2026-08-17","Multiplayer AI is people and AI on one job. Multi-agent AI is models coordinating. A guide to the distinction, when you need each, and how to talk about it with a vendor.",{"eyebrow":3104,"title":3105},"Short answers","People in the room, or models in a loop?",[3107,3110,3113],{"question":3108,"answer":3109},"Can we have both multiplayer AI and multi-agent AI?","Yes. Several models can staff steps on one shared job. The distinction is whether people share the job, not how many models you run.",{"question":3111,"answer":3112},"Does more agents mean more accountability?","Not by itself. Accountability is a named person who can refuse a change. Extra models without that name make the trail harder to read.",{"question":3114,"answer":3115},"Is a human-in-the-loop node enough?","Only if you can say whose name, on this job, for which class of write. A node labelled “review” is not a roster until it is a person.","/blog/multiplayer-ai-and-multi-agent-ai",{"title":2896,"description":3102},"blog/multiplayer-ai-and-multi-agent-ai",[1606,3120,3121],"multiplayer AI","multi-agent AI","-n6tINQ6Df9I6LG9Or7a_b3lFTjNqMd-AQcySlZwiZE",{"id":3124,"title":3125,"archived":164,"authors":3126,"badge":3128,"body":3129,"date":3101,"definedTerm":165,"department":165,"description":3582,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":3583,"relatedHeading":165,"seo":3584,"series":1606,"sitemap":130,"status":165,"stem":3585,"subhead":165,"tags":3586,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":3589},"content/blog/what-is-a-company-wiki-for-ai-agents.md","What is a Company Wiki for AI Agents",[3127],{"name":184,"to":135},{"label":1024},{"type":167,"value":3130,"toc":3557},[3131,3138,3141,3148,3151,3153,3156,3188,3191,3239,3242,3244,3252,3254,3287,3291,3306,3315,3321,3327,3337,3341,3348,3355,3362,3366,3369,3376,3379,3391,3393,3399,3406,3409,3421,3423,3427,3434,3438,3441,3445,3448,3452,3455,3459,3469,3473,3476,3480,3483,3487,3490,3494,3501,3505,3511,3515,3521,3525,3528,3530,3539,3541],[190,3132,3133,3134,3137],{},"A company wiki for AI agents is the ",[193,3135,3136],{},"official playbook the AI must follow",": owned, versioned, and scoped — not a pile of old Drive files that search might find.",[190,3139,3140],{},"Human wikis (Confluence, Notion, SharePoint) were built for people: pages, comments, “someone should update this.” Agent wikis have a harder job. Models will obey the loudest chunk in the prompt unless you separate kinds of text on purpose.",[190,3142,3143,3144,3147],{},"If the discount floor lives in a slide, a Slack rumour, and last year’s deck, an assistant asked to draft an exception will pick whichever document ",[270,3145,3146],{},"sounds"," closest to the question. That is not policy. That is folklore with a search box.",[190,3149,3150],{},"The distinction is easy to miss because both surfaces look like “knowledge.” One is a library. The other is a constitution. An agent that can retrieve every file still does not know which file is currently in force unless the runtime loads asserted policy on purpose.",[255,3152,258],{"id":257},[190,3154,3155],{},"Keep three kinds of text apart:",[260,3157,3158,3164,3178],{},[263,3159,3160,3163],{},[193,3161,3162],{},"Asserted."," What the company currently wants. Owned. Dated. Scoped. This is the wiki. At work, this is the pricing floor, the refund rule, the journal-posting checklist, the approved customer language. If legal updated it on Tuesday, the agent must cite Tuesday’s version on Wednesday — not the semantically similar PDF from 2023.",[263,3165,3166,3169,3170,3173,3174,3177],{},[193,3167,3168],{},"Retrieved."," What exists in systems. Possibly stale or contradictory. That is ",[200,3171,3172],{"href":438},"enterprise RAG",": look up authorised files, then answer. Lookup is not the same as “this is policy.” At work, retrieval is last quarter’s board pack, a ticket thread, a contract PDF. Those documents may be true as ",[270,3175,3176],{},"records",". They are not automatically the rule you want the agent to follow next.",[263,3179,3180,3183,3184,3187],{},[193,3181,3182],{},"Decided."," What we already approved in a run, stored on the ",[200,3185,3186],{"href":711},"lifecycle graph",". A signed exception should not silently overwrite the playbook for everyone else. At work, this is “this renewal was allowed 18% because of a named exception.” That fact belongs on the decision chain. It does not become the new global discount floor unless a human promotes it into the wiki.",[190,3189,3190],{},"Other terms you will hear in vendor decks and internal Slack, and how they actually show up:",[260,3192,3193,3199,3209,3215,3221,3227,3233],{},[263,3194,3195,3198],{},[193,3196,3197],{},"Vault."," A scoped partition of knowledge (finance vs people ops) with role-based access. At work, finance’s close checklist should not ride along in a recruiting workstream “just in case the model finds it useful.”",[263,3200,3201,3204,3205,3208],{},[193,3202,3203],{},"Citation."," The answer names the page and version — ",[422,3206,3207],{},"pricing v4.2"," — not “the wiki.” At work, an auditor or a new manager should be able to open the same page the agent used, not reconstruct a vibe.",[263,3210,3211,3214],{},[193,3212,3213],{},"Conflict rule."," If Drive contradicts the wiki, the wiki wins unless a human promotes a change. At work, this is the only way a retrieval-heavy assistant stops treating the loudest PDF as law.",[263,3216,3217,3220],{},[193,3218,3219],{},"Authority marker."," Labels such as policy, draft, archive, and local exception. Drafts must not load as binding context.",[263,3222,3223,3226],{},[193,3224,3225],{},"Owner."," A named role, not “the AI team.” The discount floor is owned by revenue operations or finance, not by whoever last edited a Notion page.",[263,3228,3229,3232],{},[193,3230,3231],{},"Review cadence."," A date when the page is re-checked. Silence becomes folklore.",[263,3234,3235,3238],{},[193,3236,3237],{},"Scope."," Which jobs may load this page. People-ops rules are not in the go-to-market context by default.",[190,3240,3241],{},"Most “knowledge bases,” custom GPTs, and giant system prompts fail here because they are either too global (one constitution for every department) or too private (each user pastes rules into a personal assistant). Neither is owned. Neither is maintained.",[255,3243,1160],{"id":1159},[190,3245,3246,3247,3251],{},"When official policy is unusable, employees ask consumer models to invent policy. See ",[200,3248,3250],{"href":3249},"what-is-shadow-ai","What is shadow AI",". The unofficial tool will synthesise a refund rule or a customer commitment from whatever was pasted. The company still owns the result.",[190,3253,1171],{},[260,3255,3256,3262,3272,3278],{},[263,3257,3258,3261],{},[193,3259,3260],{},"Numbers in playbooks disagree"," with numbers in CRM, and nobody can say which is official.",[263,3263,3264,3267,3268,230],{},[193,3265,3266],{},"People leave."," Tacit knowledge — the hallway version of the rule — leaves with them. See ",[200,3269,3271],{"href":3270},"what-is-institutional-memory-in-enterprise-ai","What is institutional memory in enterprise AI",[263,3273,3274,3277],{},[193,3275,3276],{},"Legal or finance must cite a version",", not a vibe.",[263,3279,3280,3283,3284,230],{},[193,3281,3282],{},"Agents can propose writes."," A model that can change CRM without a binding playbook is improvising in production. See ",[200,3285,3286],{"href":228},"What is write-back governance",[809,3288,3290],{"id":3289},"what-changes-by-role","What changes by role",[190,3292,3293,3296,3297,3300,3301,3305],{},[193,3294,3295],{},"Finance."," The wiki is where recognition rules, posting checklists, and materiality thresholds live as asserted text. Retrieval of last year’s close pack is not a substitute. If an agent drafts a journal from a Slack thread that contradicts the close checklist, finance needs the conflict rule to fire ",[270,3298,3299],{},"before"," a named signer is asked to approve. Token spend also changes: re-deriving the same policy from a pile of PDFs every run is how ",[200,3302,3304],{"href":3303},"what-is-ai-token-economics","token economics"," inflate without improving the artefact.",[190,3307,3308,3311,3312,3314],{},[193,3309,3310],{},"Legal."," Approved language, retention classes, and “do not say” lists belong in asserted pages with owners. A retrieved contract is evidence of what was signed with ",[270,3313,1858],{}," counterparty. It is not the company’s current standard terms. Legal also cares that citations name a version. “According to our documents” is not a defence if those documents include drafts.",[190,3316,3317,3320],{},[193,3318,3319],{},"Operations."," Runbooks, escalation thresholds, and supplier exception rules need to be loadable as the current procedure, not as the closest matching incident write-up. Ops already knows that a stale SOP is worse than no SOP, because people follow it. Agents do the same, faster.",[190,3322,3323,3326],{},[193,3324,3325],{},"Go-to-market."," Discount floors, win/loss taxonomies, and approved competitive language are the pages that stop an assistant inventing a concession. GTM also feels the scope problem first: a “help me close this” chat that can see every playbook in the company will mix people-ops rules, finance forecasts, and last year’s campaign into one fluent paragraph.",[190,3328,3329,3332,3333,230],{},[193,3330,3331],{},"Security."," Vaults and least-privilege loading are access control. Indexing every SharePoint site into a single “brain” is a new store of sensitive data. Security’s question is not “does the model know enough?” It is “which pages is this job allowed to load, and can we prove it?” GDPR-style purpose limitation still applies when the reader is an AI. See ",[200,3334,3336],{"href":3335},"what-is-ai-governance","What is AI governance",[809,3338,3340],{"id":3339},"what-people-get-wrong","What people get wrong",[190,3342,3343,3344,3347],{},"The common failure is treating ",[193,3345,3346],{},"search as policy",". Teams export Confluence into a vector index, label it “the brain,” and congratulate themselves for grounding. Search will surface the outdated note because it is semantically close to the question. Grounding on the wrong document is still grounding. It is just grounding on folklore.",[190,3349,3350,3351,3354],{},"The second failure is the ",[193,3352,3353],{},"personal constitution",": each power user pastes rules into a custom GPT. Those rules are not org-owned, not scoped per job, and not cited as a version in an audit. When two users paste different discount floors, the company has two unofficial policies.",[190,3356,3357,3358,3361],{},"The third failure is the ",[193,3359,3360],{},"mega-prompt",". One global instruction block tries to encode every department. It is never current. It cannot be scoped. It cannot be reviewed by the owner of a single domain. It also burns tokens on every call.",[809,3363,3365],{"id":3364},"what-good-looks-like-versus-what-fails","What good looks like versus what fails",[190,3367,3368],{},"A good agent wiki has authority markers (policy vs draft vs archive), scope (people-ops rules are not in the go-to-market context by default), versions, named owners, a review cadence, and tables for numbers. Numbers belong in tables because prose rounds them. Agents will quote the table if you give them one.",[190,3370,3371,3372,3375],{},"A good wiki is also ",[193,3373,3374],{},"written for two audiences",": the human who must own the page, and the agent that must cite it. Humans need headings and owners. Agents need unambiguous numbers and conflict rules.",[190,3377,3378],{},"A bad agent wiki is an export of Confluence into a search index. The intranet wiki remains useful for humans. It is still not binding on agents unless the runtime loads a controlled subset. Connection is not the same as “everything in Confluence is policy.”",[190,3380,3381,3382,3384,3385,3387,3388,3390],{},"Adjacent ideas: retrieval without assertion is ",[200,3383,3172],{"href":438},". Decisions without a playbook are a ",[200,3386,3186],{"href":711}," with nothing to cite. A job that loads the wrong vault is a ",[200,3389,1083],{"href":215}," with the wrong attachments.",[255,3392,793],{"id":792},[190,3394,3395,3396,3398],{},"The ",[193,3397,27],{}," is the asserted policy layer every agent team must treat as binding. Workstreams subscribe to wiki sections so scope is enforced at runtime. Pages connect to runs and approvals on the Lifecycle Graph.",[190,3400,3401,3402,3405],{},"The wiki is not a second search engine. Connectors remain the path to live systems, and they default to read-only. Retrieval of Drive or CRM is still retrieval. The wiki is what those reads are interpreted ",[270,3403,3404],{},"against",". When a write is proposed, the named signer should see the playbook version the draft claims to follow.",[190,3407,3408],{},"Perception can ask what the current playbook says, and which run last cited it.",[190,3410,3411,3412,3415,3416,3418,3419,230],{},"See ",[200,3413,3414],{"href":28},"Wiki",". For the job that loads a subset of pages, see ",[200,3417,31],{"href":32},". For the chain that records which version was used, see ",[200,3420,23],{"href":24},[255,3422,807],{"id":806},[809,3424,3426],{"id":3425},"isnt-this-just-confluence","Isn’t this just Confluence?",[190,3428,3429,3430,3433],{},"Confluence is a human wiki. An agent wiki is a ",[270,3431,3432],{},"binding"," subset: owned, versioned, scoped, and loaded on purpose. You can connect Confluence into that layer. Connection is not the same as “everything in Confluence is policy.”",[809,3435,3437],{"id":3436},"cant-we-just-search-drive","Can’t we just search Drive?",[190,3439,3440],{},"Search is retrieval. Retrieval finds what exists. It does not decide what the company currently wants. If Drive contains three discount floors, search will return the closest one, not the official one.",[809,3442,3444],{"id":3443},"what-if-the-wiki-is-wrong","What if the wiki is wrong?",[190,3446,3447],{},"Then a human updates it, with a version and an owner. Do not let a one-off exception silently become the new global rule. Promote the change; do not hope the next retrieval will “learn.”",[809,3449,3451],{"id":3450},"how-is-this-different-from-a-custom-gpts-instructions","How is this different from a custom GPT’s instructions?",[190,3453,3454],{},"Instructions in a personal GPT are not org-owned, not scoped per job, and not cited as a version in an audit. Two users can ship two unofficial policies without anyone noticing until a customer is told the wrong thing.",[809,3456,3458],{"id":3457},"do-we-need-a-wiki-if-we-already-have-rag","Do we need a wiki if we already have RAG?",[190,3460,3461,3462,3465,3466,230],{},"Yes, if agents will act. RAG reduces invention on ",[270,3463,3464],{},"existing"," files. It does not mark which file is in force. Without assertion, retrieval-augmented generation is retrieval-augmented folklore. See ",[200,3467,3468],{"href":438},"What is enterprise RAG",[809,3470,3472],{"id":3471},"who-should-own-wiki-pages","Who should own wiki pages?",[190,3474,3475],{},"The same function that owns the analogue rule. Pricing belongs to revenue operations or finance. Employment language belongs to people ops and legal. “The AI team” is a coordinator, not a policy owner.",[809,3477,3479],{"id":3478},"how-often-should-pages-be-reviewed","How often should pages be reviewed?",[190,3481,3482],{},"On a cadence that matches how often the rule changes, plus a hard date so silence is visible. A discount floor that never expires is how last year’s promotion becomes this year’s default.",[809,3484,3486],{"id":3485},"what-should-we-put-in-tables-versus-prose","What should we put in tables versus prose?",[190,3488,3489],{},"Numbers, thresholds, codes, and “never / always” lists belong in tables. Narrative belongs in prose. Agents quote tables more reliably than they extract a number buried in a paragraph.",[809,3491,3493],{"id":3492},"can-one-wiki-serve-the-whole-company","Can one wiki serve the whole company?",[190,3495,3496,3497,3500],{},"One ",[270,3498,3499],{},"product",", many vaults. A single unscoped corpus recreates the god workspace. Finance close pages and recruiting pages should not share a default context.",[809,3502,3504],{"id":3503},"how-do-exceptions-work-without-rewriting-the-playbook","How do exceptions work without rewriting the playbook?",[190,3506,3507,3508,230],{},"Record the exception on the decision chain — who signed, which page version, which record — and leave the playbook intact unless a human promotes a change. See ",[200,3509,3510],{"href":711},"What is a lifecycle graph",[809,3512,3514],{"id":3513},"will-a-better-model-make-the-wiki-unnecessary","Will a better model make the wiki unnecessary?",[190,3516,3517,3518,230],{},"No. Stronger models are better at sounding like policy. That makes an unowned corpus more dangerous, not less. Model routing can send interpretation to a stronger model; it cannot invent an owner. See ",[200,3519,3520],{"href":1261},"What is model routing",[809,3522,3524],{"id":3523},"how-does-this-relate-to-access-control","How does this relate to access control?",[190,3526,3527],{},"Loading a page is still processing. A recruiting workstream should not load compensation policy “because it might help.” Vaults and workstream subscriptions keep that promise in software rather than in a PDF.",[255,3529,869],{"id":868},[190,3531,3532,217,3534,3536,3537,230],{},[200,3533,3468],{"href":438},[200,3535,3271],{"href":3270},", and ",[200,3538,3336],{"href":3335},[255,3540,883],{"id":882},[260,3542,3543,3550],{},[263,3544,3545],{},[200,3546,3549],{"href":3547,"rel":3548},"https://ico.org.uk/for-organisations/uk-gdpr-guidance-and-resources/artificial-intelligence/",[204],"ICO, AI and data protection",[263,3551,3552],{},[200,3553,3556],{"href":3554,"rel":3555},"https://eur-lex.europa.eu/eli/reg/2016/679/oj",[204],"GDPR",{"title":170,"searchDepth":171,"depth":171,"links":3558},[3559,3560,3565,3566,3580,3581],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":3561},[3562,3563,3564],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":3364,"depth":1001,"text":3365},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":3567},[3568,3569,3570,3571,3572,3573,3574,3575,3576,3577,3578,3579],{"id":3425,"depth":1001,"text":3426},{"id":3436,"depth":1001,"text":3437},{"id":3443,"depth":1001,"text":3444},{"id":3450,"depth":1001,"text":3451},{"id":3457,"depth":1001,"text":3458},{"id":3471,"depth":1001,"text":3472},{"id":3478,"depth":1001,"text":3479},{"id":3485,"depth":1001,"text":3486},{"id":3492,"depth":1001,"text":3493},{"id":3503,"depth":1001,"text":3504},{"id":3513,"depth":1001,"text":3514},{"id":3523,"depth":1001,"text":3524},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"A company wiki for AI agents is the official playbook the AI must follow — versioned, owned, and scoped — not a pile of old Drive files the search might find.","/blog/what-is-a-company-wiki-for-ai-agents",{"title":3125,"description":3582},"blog/what-is-a-company-wiki-for-ai-agents",[1606,326,3587,3588],"agents","knowledge","uIdVqeIYg1mF11bA-8NHtcgSoEw6IxtDS1EpjAzO_pk",{"id":3591,"title":3592,"archived":164,"authors":3593,"badge":3595,"body":3596,"date":3101,"definedTerm":165,"department":165,"description":4040,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":4041,"relatedHeading":165,"seo":4042,"series":1606,"sitemap":130,"status":165,"stem":4043,"subhead":165,"tags":4044,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":4048},"content/blog/what-is-a-lifecycle-graph.md","What is a Lifecycle Graph",[3594],{"name":184,"to":135},{"label":1024},{"type":167,"value":3597,"toc":4015},[3598,3605,3608,3611,3614,3616,3671,3688,3690,3693,3696,3699,3725,3740,3742,3750,3755,3760,3765,3770,3774,3777,3794,3797,3800,3808,3810,3817,3828,3835,3844,3847,3849,3854,3860,3867,3877,3879,3883,3890,3894,3900,3904,3907,3911,3914,3918,3924,3928,3931,3935,3938,3942,3947,3951,3954,3958,3961,3965,3970,3974,3981,3983,3993,3995],[190,3599,3600,3601,3604],{},"When people ask what a lifecycle graph is, they are usually asking about ",[193,3602,3603],{},"causality",": why did this number, field, or decision change?",[190,3606,3607],{},"Causality is the difference between “two things happened around the same time” and “this caused that.” If pipeline coverage went up in the same month AI usage went up, that is a coincidence until you can show the actual steps: what was asked, what was used, who approved it, and what the live system did.",[190,3609,3610],{},"A lifecycle graph is the company’s record of those steps. It is not a chat history. Chat history shows that someone talked to a model. A graph shows the chain from the question to the outcome, so the next person — or an auditor — can follow it.",[190,3612,3613],{},"This is an operations problem that existed before generative AI. ERP journals already needed authorisation trails. CRM already had field history. What changed is that a new kind of actor can now propose, and sometimes execute, those changes in fluent language. If the “why” lives only in a personal chat, the company has a causality gap the moment that person leaves, the vendor rotates logs, or the model version rolls.",[255,3615,258],{"id":257},[260,3617,3618,3624,3630,3642,3648,3654,3660,3666],{},[263,3619,3620,3623],{},[193,3621,3622],{},"Causality."," Being able to say what caused what, with evidence. At work, this is “this next-step field changed because this brief ran, cited this playbook version, and this person signed this payload.”",[263,3625,3626,3629],{},[193,3627,3628],{},"Correlation."," Two things moving together. Not the same as cause. At work, AI usage and pipeline moving in the same quarter is a slide, not an explanation.",[263,3631,3632,3635,3636,3641],{},[193,3633,3634],{},"Provenance."," The trail of who, what, and when behind a piece of data. ",[200,3637,3640],{"href":3638,"rel":3639},"https://www.w3.org/TR/prov-overview/",[204],"W3C PROV"," is the open standard for that idea: entities, activities, and agents, linked so you can reconstruct derivation. A lifecycle graph is that instinct applied to AI-mediated work, not a claim that you have implemented the full W3C stack.",[263,3643,3644,3647],{},[193,3645,3646],{},"System of record."," The official live system that holds the fact — Salesforce for an opportunity, NetSuite for a journal. The graph should point at that record, not become a second copy of it. At work, pointing is how you avoid a second CRM that nobody can delete.",[263,3649,3650,3653],{},[193,3651,3652],{},"Audit trail."," A log that something happened. Useful, but thin if it cannot join the question, the policy, the signer, and the change. Provider API logs are an audit trail of calls. They are not a story of the job.",[263,3655,3656,3659],{},[193,3657,3658],{},"Lineage."," Which sources fed which proposal. At work, “which wiki version and which CRM records were in scope when this quote was generated?”",[263,3661,3662,3665],{},[193,3663,3664],{},"Retention."," How long a class of node is kept. A journal that feeds the books may need years. A draft may need weeks.",[263,3667,3668,3670],{},[193,3669,3237],{}," Which job’s chain you are allowed to see. At work, a go-to-market question should not surface People Ops briefs.",[190,3672,3673,3674,3677,3678,3680,3681,858,3684,3687],{},"Keep this graph apart from two neighbours. A ",[193,3675,3676],{},"business knowledge graph"," models customers, products, and sites. ",[200,3679,439],{"href":438}," retrieves documents that ",[270,3682,3683],{},"exist",[200,3685,3686],{"href":3270},"Institutional memory"," is the broader goal — what the company still knows after people leave. The lifecycle graph is the decision-memory layer of that goal: the chain of AI-mediated work.",[255,3689,1160],{"id":1159},[190,3691,3692],{},"AI makes it easy to change company systems without leaving a story. Someone asks a model to tidy CRM notes. A field moves. Next quarter, finance or legal asks why. The person who asked has left. The “why” lived in a personal chat. The CRM only shows the new value.",[190,3694,3695],{},"That is a causality problem. You cannot manage what you cannot reconstruct.",[190,3697,3698],{},"It affects you if you:",[260,3700,3701,3707,3713,3719],{},[263,3702,3703,3706],{},[193,3704,3705],{},"Sign off on numbers."," Forecasts, journals, and board packs inherit whatever AI changed last month.",[263,3708,3709,3712],{},[193,3710,3711],{},"Inherit someone else’s work."," You need the exception, not a rumour that “we always do 18% for strategic accounts.”",[263,3714,3715,3718],{},[193,3716,3717],{},"Answer auditors or regulators."," They will not accept “the chatbot did it.”",[263,3720,3721,3724],{},[193,3722,3723],{},"Switch vendors or models."," Provider logs are the vendor’s artefact. They are not your company memory.",[190,3726,3727,3728,3731,3732,3736,3737],{},"This is not the same as proving that a discount ",[270,3729,3730],{},"caused"," a won deal. That is a statistics question. See ",[200,3733,3735],{"href":3734},"what-is-causal-ai-for-operations","What is causal AI for operations",". A lifecycle graph answers a more basic one: ",[193,3738,3739],{},"what did we actually do, and who caused it?",[809,3741,3290],{"id":3289},[190,3743,3744,3746,3747,230],{},[193,3745,3295],{}," Close packs and forecasts inherit field history. If an AI-proposed journal posted, finance needs the brief, the playbook version, the named signer, and the ERP response — not a Slack screenshot. Spend also belongs on the chain: a run that stopped because a cap was hit is a causal fact, not a missing invoice. See ",[200,3748,3749],{"href":3303},"What is AI token economics",[190,3751,3752,3754],{},[193,3753,3310],{}," Exception language, customer commitments, and “who saw what” are discovery questions. A graph that points at the payload the signer saw is evidence. A chat export from a personal account is not. Legal also cares about retention and deletion: infinite chat fails a privacy review; typed retention with export and legal hold is how records programmes already work.",[190,3756,3757,3759],{},[193,3758,3319],{}," Handoffs fail when the next shift cannot see why a run paused. Human wait is a node, not an interruption. Incident reviews need the same chain: which connector was read-only, which write was refused, which wiki page caused the flag.",[190,3761,3762,3764],{},[193,3763,3325],{}," Pipeline hygiene and renewal exceptions are where “the bot updated it” becomes a forecast problem. GTM needs to see the quoted fields, not a summary that says “updated pricing.” They also need scope: one team’s competitive notes should not leak into another region’s chain.",[190,3766,3767,3769],{},[193,3768,3331],{}," The graph is a sensitive store. It should not hold full transcripts with secrets by default, other teams’ out-of-scope work, or the model’s private scratch reasoning. Access control on the graph is as important as access control on the CRM. A query surface that ignores vaults recreates the god workspace.",[809,3771,3773],{"id":3772},"what-belongs-on-the-chain","What belongs on the chain",[190,3775,3776],{},"Keep the links that let a non-engineer reconstruct a change:",[260,3778,3779,3782,3785,3788,3791],{},[263,3780,3781],{},"the job and the question",[263,3783,3784],{},"the sources (which playbook version, which records, which files)",[263,3786,3787],{},"the people (the model is not an answer for “who”)",[263,3789,3790],{},"the proposed change, in the language of the live system — fields and values, not “updated pricing”",[263,3792,3793],{},"the decision, the timestamp, and whether the live system accepted it",[190,3795,3796],{},"Point at the CRM record and the policy page. Do not copy the whole company into the graph. Copies become a second official system, and a deletion problem.",[190,3798,3799],{},"Do not keep, by default: the model’s private scratch reasoning, full transcripts with secrets, or other teams’ work that was never in scope.",[190,3801,3802,3803,3807],{},"Retention should follow the type of record. A journal that feeds the books may need years. A draft may need weeks. “Keep everything forever because AI” fails a privacy review. ",[200,3804,3806],{"href":3547,"rel":3805},[204],"UK ICO guidance on AI and data protection"," still wants purpose and minimisation when the “user” is an AI.",[809,3809,3340],{"id":3339},[190,3811,3812,3813,3816],{},"The first mistake is ",[193,3814,3815],{},"treating chat history as the record",". Chat is a user interface. It is not a join of brief, policy, signer, and system response.",[190,3818,3819,3820,3823,3824,3827],{},"The second is ",[193,3821,3822],{},"retrofitting",". Copying six months of ChatGPT and Slack into a warehouse is archaeology. You still need something that ",[270,3825,3826],{},"emits"," events at the moment of the brief, the quote, and the approval.",[190,3829,3830,3831,3834],{},"The third is ",[193,3832,3833],{},"a second CRM",". Duplicating every opportunity into the graph “for completeness” creates conflicting official numbers and an erasure nightmare.",[190,3836,3837,3838,3841,3842,230],{},"The fourth is ",[193,3839,3840],{},"confusing this with causal science",". A fluent model paragraph that says “because” is not identification. Neither is a dashboard of two rising lines. See ",[200,3843,3735],{"href":3734},[190,3845,3846],{},"Good looks like reconstructable interventions with pointers, named people, and typed retention. Failure looks like a vendor log, a personal thread, or an infinite lake of tokens.",[255,3848,793],{"id":792},[190,3850,2603,3851,3853],{},[193,3852,23],{}," is that chain as a product: briefs, playbook citations, connector reads, spend, approvals, and write results are linked as work proceeds.",[190,3855,3856,3859],{},[193,3857,3858],{},"Perception"," is how you ask it in ordinary language — “why did this opportunity change last month?” — instead of reconstructing Slack.",[190,3861,3862,3863,3866],{},"The graph records ",[193,3864,3865],{},"work",", not every token the company ever sent to a model. Scope follows the job, so a go-to-market question should not surface People Ops briefs. Connectors default to read-only; a read that did not write is itself a node worth knowing. Fail-closed writes mean a missing named signer is a recorded refusal, not a silent mutation.",[190,3868,3869,3870,876,3872,3874,3875,230],{},"Product: ",[200,3871,23],{"href":24},[200,3873,3858],{"href":36},". The job that produces the chain is a ",[200,3876,1083],{"href":215},[255,3878,807],{"id":806},[809,3880,3882],{"id":3881},"is-this-just-a-knowledge-graph-of-the-business","Is this just a knowledge graph of the business?",[190,3884,3885,3886,3889],{},"No. A business knowledge graph models customers, products, and sites. A lifecycle graph models ",[193,3887,3888],{},"AI-mediated work"," — what was asked, who signed, what changed. They can link (the write points at an opportunity). They are not the same thing.",[809,3891,3893],{"id":3892},"cant-the-warehouse-be-the-record","Can’t the warehouse be the record?",[190,3895,3896,3897,3899],{},"You can copy events into a warehouse for reporting. You still need something that ",[270,3898,3826],{}," those events at the moment of the brief, the quote, and the approval. Retrofitting six months of ChatGPT and Slack is archaeology, not operations.",[809,3901,3903],{"id":3902},"how-is-this-different-from-the-model-providers-logs","How is this different from the model provider’s logs?",[190,3905,3906],{},"Provider logs show API calls. They do not know your job, your playbook version, your approver, or whether the write was rejected.",[809,3908,3910],{"id":3909},"how-long-should-we-keep-it","How long should we keep it?",[190,3912,3913],{},"Treat it like other control evidence. Align retention with the type of record, legal hold, and storage limits. The product must support export, deletion, and access control — not infinite chat.",[809,3915,3917],{"id":3916},"how-does-this-relate-to-a-person-having-to-approve","How does this relate to a person having to approve?",[190,3919,3920,3921,230],{},"A human gate only counts if you can later show who signed and what they saw. Without a graph, that gate is a popup that forgets. See ",[200,3922,3923],{"href":1896},"What is human-in-the-loop AI",[809,3925,3927],{"id":3926},"does-the-graph-replace-crm-field-history","Does the graph replace CRM field history?",[190,3929,3930],{},"No. Field history says the value changed. The graph says which job, which policy version, and which named signer caused the proposal. Keep both. Point; do not duplicate.",[809,3932,3934],{"id":3933},"what-if-the-models-explanation-disagrees-with-the-graph","What if the model’s explanation disagrees with the graph?",[190,3936,3937],{},"Trust the structure. Fluent “because” text is often written after the fact. The chain of brief, sources, quote, and signature is the operational cause.",[809,3939,3941],{"id":3940},"can-we-store-every-prompt-and-completion","Can we store every prompt and completion?",[190,3943,3944,3945,230],{},"You can. You usually should not. Completeness is reconstructability, not hoarding. Secrets in transcripts become a new breach class. See ",[200,3946,3271],{"href":3270},[809,3948,3950],{"id":3949},"how-do-permissions-work-on-the-graph","How do permissions work on the graph?",[190,3952,3953],{},"The same least-privilege instinct as the job. If you could not see the People Ops workstream, you should not query its chain in ordinary language either.",[809,3955,3957],{"id":3956},"is-a-screenshot-of-the-approval-enough","Is a screenshot of the approval enough?",[190,3959,3960],{},"For a one-off incident, maybe. For a control, no. Screenshots do not join, do not retain by type, and do not survive the laptop.",[809,3962,3964],{"id":3963},"where-does-spend-sit-on-the-chain","Where does spend sit on the chain?",[190,3966,3967,3968,230],{},"Quotes, caps, and stop-on-budget are causal events. “The run did not write because the ceiling was hit” is an answer finance can use. See ",[200,3969,3749],{"href":3303},[809,3971,3973],{"id":3972},"how-is-this-different-from-mlops-experiment-tracking","How is this different from MLOps experiment tracking?",[190,3975,3976,3977,3980],{},"MLOps tracks model training and deployment. A lifecycle graph tracks operational work that ",[270,3978,3979],{},"uses"," models. They stack. They do not substitute.",[255,3982,869],{"id":868},[190,3984,3985,3986,3874,3988,3990,3991,230],{},"If the goal is what the company still knows after people leave, read ",[200,3987,3271],{"href":3270},[200,3989,1083],{"href":215},". For the science versus operations cut, ",[200,3992,3735],{"href":3734},[255,3994,883],{"id":882},[260,3996,3997,4003,4010],{},[263,3998,3999],{},[200,4000,4002],{"href":3638,"rel":4001},[204],"W3C PROV overview",[263,4004,4005],{},[200,4006,4009],{"href":4007,"rel":4008},"https://www.w3.org/TR/prov-dm/",[204],"W3C PROV data model",[263,4011,4012],{},[200,4013,3549],{"href":3547,"rel":4014},[204],{"title":170,"searchDepth":171,"depth":171,"links":4016},[4017,4018,4023,4024,4038,4039],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":4019},[4020,4021,4022],{"id":3289,"depth":1001,"text":3290},{"id":3772,"depth":1001,"text":3773},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":4025},[4026,4027,4028,4029,4030,4031,4032,4033,4034,4035,4036,4037],{"id":3881,"depth":1001,"text":3882},{"id":3892,"depth":1001,"text":3893},{"id":3902,"depth":1001,"text":3903},{"id":3909,"depth":1001,"text":3910},{"id":3916,"depth":1001,"text":3917},{"id":3926,"depth":1001,"text":3927},{"id":3933,"depth":1001,"text":3934},{"id":3940,"depth":1001,"text":3941},{"id":3949,"depth":1001,"text":3950},{"id":3956,"depth":1001,"text":3957},{"id":3963,"depth":1001,"text":3964},{"id":3972,"depth":1001,"text":3973},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"A lifecycle graph is how a company answers “why did this happen?” after AI is involved — the chain of cause and effect, not a chat log.","/blog/what-is-a-lifecycle-graph",{"title":3592,"description":4040},"blog/what-is-a-lifecycle-graph",[1606,4045,4046,4047],"lifecycle-graph","institutional-memory","audit","PBBZGOGmt4iOW9tVMKpXENF9DNLNe4GLbRJYW6-_uBA",{"id":4050,"title":4051,"archived":164,"authors":4052,"badge":4054,"body":4055,"date":3101,"definedTerm":165,"department":165,"description":4590,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":4591,"relatedHeading":165,"seo":4592,"series":1606,"sitemap":130,"status":165,"stem":4593,"subhead":165,"tags":4594,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":4596},"content/blog/what-is-ai-governance.md","What is AI Governance",[4053],{"name":184,"to":135},{"label":1024},{"type":167,"value":4056,"toc":4566},[4057,4064,4067,4070,4086,4089,4102,4104,4192,4201,4203,4211,4213,4243,4246,4274,4281,4283,4288,4298,4303,4308,4317,4319,4325,4331,4337,4345,4351,4357,4374,4376,4379,4386,4399,4408,4410,4414,4422,4426,4429,4433,4440,4444,4447,4451,4454,4458,4465,4469,4472,4476,4479,4483,4486,4490,4493,4497,4502,4506,4509,4511,4520,4522],[190,4058,4059,4060,4063],{},"AI governance is the set of rules, ",[193,4061,4062],{},"enforced in the software people actually use",", that decide who may use which AI, on which company data, and whether that AI is allowed to change a live business system — plus a record of what happened afterwards.",[190,4065,4066],{},"A training video is not that. An acceptable-use PDF is not that. An admin toggle the model can ignore is not that. If an unapproved change can still succeed, you have guidance, not governance.",[190,4068,4069],{},"People use the phrase for three different things, and they get mixed up:",[547,4071,4072,4080,4083],{},[263,4073,4074,4075,230],{},"A public commitment — for example the ",[200,4076,4079],{"href":4077,"rel":4078},"https://oecd.ai/en/ai-principles",[204],"OECD AI Principles",[263,4081,4082],{},"A company committee with a risk register.",[263,4084,4085],{},"The runtime that actually stops a change to CRM, ERP, or a customer message.",[190,4087,4088],{},"All three are real. Only the third one would have blocked an unlogged field change that later showed up in a forecast.",[190,4090,4091,4096,4097,4101],{},[200,4092,4095],{"href":4093,"rel":4094},"https://www.gartner.com/en/articles/ai-governance-trism",[204],"Gartner’s TRiSM"," language is about that third layer: trust, risk, and security around the systems that run — not a quarterly slide about principles. The ",[200,4098,4100],{"href":363,"rel":4099},[204],"NIST AI Risk Management Framework"," says the same thing in public-sector language: Govern, Map, Measure, Manage. Mapping systems and measuring incidents still fail if the product people click can write to Salesforce without a named signer.",[255,4103,258],{"id":257},[260,4105,4106,4112,4125,4134,4140,4146,4159,4167,4179],{},[263,4107,4108,4111],{},[193,4109,4110],{},"Live business system."," CRM, ERP, HR, billing — the tools that hold official numbers and customer records. At work, this is where a fluent sentence becomes a fact other teams will inherit.",[263,4113,4114,4117,4118,4121,4122,4124],{},[193,4115,4116],{},"Write / write-back."," The AI is allowed to ",[270,4119,4120],{},"change"," that system, not only draft a suggestion. See ",[200,4123,3286],{"href":228},". At work, a next-step note and an Amount field are not the same risk class.",[263,4126,4127,4130,4131,4133],{},[193,4128,4129],{},"Human-in-the-loop."," A person must approve before the job can finish. See ",[200,4132,3923],{"href":1896},". At work, the gate shows the payload in the language of the live system, not a wall of prompt text.",[263,4135,4136,4139],{},[193,4137,4138],{},"Named signer."," The identity that authorised the change. At work, “someone in the channel clicked yes” is not a signer.",[263,4141,4142,4145],{},[193,4143,4144],{},"Fail-closed."," Missing approval means nothing happens. Fail-open means the change goes through unless someone happens to stop it.",[263,4147,4148,4151,4152,4154,4155,4158],{},[193,4149,4150],{},"DPIA."," A data-protection impact assessment — thinking through purpose, risk, and personal data ",[270,4153,3299],{}," you turn a tool loose. ",[200,4156,3806],{"href":3547,"rel":4157},[204]," still wants a lawful basis and purpose when the “user” is an AI.",[263,4160,4161,4164,4165,230],{},[193,4162,4163],{},"Shadow AI."," Personal ChatGPT for work because the official path is missing. See ",[200,4166,3250],{"href":3249},[263,4168,4169,4172,4173,4178],{},[193,4170,4171],{},"Inventory."," A list of where AI actually runs. The ",[200,4174,4177],{"href":4175,"rel":4176},"https://www.justice.gov/media/1373026/dl",[204],"US plan described in OMB M-24-10"," puts a named owner and an inventory first, not a PDF.",[263,4180,4181,4184,4185,4188,4189,4191],{},[193,4182,4183],{},"Least privilege."," Only the data and tools required for ",[270,4186,4187],{},"this"," job. A ",[200,4190,1083],{"href":215}," is how that instinct becomes a company object.",[190,4193,4194,4195,4197,4198,4200],{},"Model safety is adjacent and different. Safety is about what a model will say in the abstract. Enterprise governance is about what ",[270,4196,782],{}," people and tools may do with ",[270,4199,782],{}," systems and data. You can have a carefully aligned model and still have ungoverned CRM writes.",[255,4202,1160],{"id":1159},[190,4204,4205,4206,4210],{},"Without working rules, AI becomes a side effect. A field moves. A journal posts. A customer is told a policy the company does not hold. Nobody can say who allowed it. In February 2024, a British Columbia tribunal held Air Canada responsible for a chatbot that invented a bereavement-fare policy. ",[200,4207,4209],{"href":2041,"rel":4208},[204],"CBC reported"," that the airline’s argument — the chatbot is a separate legal entity — failed. A customer-facing commitment without a working gate is still the company’s commitment.",[190,4212,3698],{},[260,4214,4215,4221,4227,4237],{},[263,4216,4217,4220],{},[193,4218,4219],{},"Own a number."," Forecasts and close packs inherit whatever changed.",[263,4222,4223,4226],{},[193,4224,4225],{},"Own a customer relationship."," Model output that becomes a commitment is still the company’s commitment.",[263,4228,4229,4232,4233,4236],{},[193,4230,4231],{},"Own risk or legal."," Privacy law does not pause for a chatbot. ",[200,4234,3556],{"href":3554,"rel":4235},[204]," still applies to purpose, minimisation, and erasure.",[263,4238,4239,4242],{},[193,4240,4241],{},"Are asked “who is in charge of AI here?”"," An inventory and a named owner beat a principles slide.",[190,4244,4245],{},"Good governance in practice is four working rules:",[260,4247,4248,4254,4260,4266],{},[263,4249,4250,4253],{},[193,4251,4252],{},"People and rights."," Humans and AI tools are both actors. Roles decide what they may start, see, and sign.",[263,4255,4256,4259],{},[193,4257,4258],{},"Data at question time."," Purpose and minimisation still apply when an AI is the one looking.",[263,4261,4262,4265],{},[193,4263,4264],{},"Action rights."," Read-only is a control. Unrestricted tools are an incident waiting for a bad prompt.",[263,4267,4268,4271,4272,230],{},[193,4269,4270],{},"Evidence and spend."," Chat scrollback is not a management system. Uncapped spend is a budget failure and often a security failure. See ",[200,4273,3749],{"href":3303},[190,4275,4276,4277,4280],{},"Blocking consumer ChatGPT at the office network, while people use personal phones, is not governance. It is a ",[200,4278,4279],{"href":3249},"shadow AI"," problem with extra steps.",[809,4282,3290],{"id":3289},[190,4284,4285,4287],{},[193,4286,3295],{}," Governance is whether an AI-proposed journal can post, against which checklist, with which signer, and whether the spend of the run was capped. “Unlimited AI” is not a control. Surprise inference bills are a governance failure that looks like a cloud invoice.",[190,4289,4290,4292,4293,4297],{},[193,4291,3310],{}," Lawful basis, purpose limitation, customer-facing language, and reconstructable authorisation. Legal also has to separate the OECD-style public commitment from the runtime. A principles page does not implement Article-style oversight. For higher-risk systems, ",[200,4294,4296],{"href":599,"rel":4295},[204],"EU AI law"," Article 14 talks about effective oversight: people must be able to interpret outputs and interrupt the system. A footer that says “generated by AI” is not that.",[190,4299,4300,4302],{},[193,4301,3319],{}," Isolation of jobs, connector scope, and a place to put a paused run. Ops already runs change control. Governance is change control that includes a model as a proposer.",[190,4304,4305,4307],{},[193,4306,3325],{}," The difference between a draft email and a sent commitment; between a suggested next step and a changed Amount. GTM feels friction first. The honest metric is time-to-approved-write, not time-to-first-answer.",[190,4309,4310,4312,4313,4316],{},[193,4311,3331],{}," Identity of the connected user, read versus write, prompt injection as a path to a tool call, and the new store created by logs and indexes. The ",[200,4314,977],{"href":375,"rel":4315},[204]," treats retrieval and tool use as a security surface, not only a quality issue. Network DLP helps with paste-out. It does not quote a CRM change.",[809,4318,3340],{"id":3339},[190,4320,4321,4324],{},[193,4322,4323],{},"Governance as a committee."," Useful for risk registers. Useless if the product can still write.",[190,4326,4327,4330],{},[193,4328,4329],{},"Governance as model safety."," Refusals on public-web questions do not bind Salesforce.",[190,4332,4333,4336],{},[193,4334,4335],{},"Governance as a secure web gateway."," Necessary for some paste-out paths. Insufficient for writes, approvals, and causal history.",[190,4338,4339,4342,4343,230],{},[193,4340,4341],{},"Governance as blocking."," Blocks without a sanctioned path train people onto phones. See ",[200,4344,3250],{"href":3249},[190,4346,4347,4350],{},[193,4348,4349],{},"Theatre."," A checkbox, a prompt that says “ask first,” or an admin toggle the model can ignore.",[190,4352,4353,4354,4356],{},"Good looks like: connectors default to read-only; writes are quoted; a named signer cannot be waived by the model; evidence lives on a ",[200,4355,3186],{"href":711},"; spend has a ceiling; scope follows the job. Failure looks like a PDF, a blocked URL, and a personal API key in a wiki.",[190,4358,4359,4360,4362,4363,4365,4366,4368,4369,4373],{},"Adjacent concepts: ",[200,4361,1079],{"href":228}," is the write subset. ",[200,4364,2575],{"href":1896}," is the gate. ",[200,4367,31],{"href":215}," are the isolation unit. An ",[200,4370,4372],{"href":4371},"what-is-an-enterprise-ai-operating-system","enterprise AI operating system"," is the product shape that makes those rules the default path.",[255,4375,793],{"id":792},[190,4377,4378],{},"Nimbus treats governance as how work is released, not as a sidecar policy engine.",[190,4380,4381,4382,4385],{},"Connectors — secure links to live systems — default to ",[193,4383,4384],{},"read-only",". When a change is proposed, the product shows the intended action and waits. A named person must sign. The model cannot waive the gate. Missing approval is fail-closed: nothing happens.",[190,4387,4388,4389,4391,4392,4394,4395,4398],{},"Scope is the ",[200,4390,1083],{"href":215},": one job, with the playbooks, systems, teams, and budget that belong to that job. Evidence is the ",[200,4393,23],{"href":711},". The ",[200,4396,4397],{"href":443},"company wiki"," is the asserted policy the run must cite. Model routing does not bypass the gate.",[190,4400,3411,4401,4403,4404,230],{},[200,4402,39],{"href":40},". For scoring vendors: ",[200,4405,4407],{"href":4406},"how-to-evaluate-ai-governance-platforms","How to evaluate AI governance platforms",[255,4409,807],{"id":806},[809,4411,4413],{"id":4412},"is-ai-governance-the-same-as-making-the-model-safe","Is AI governance the same as making the model “safe”?",[190,4415,4416,4417,4197,4419,4421],{},"No. Model safety is about what the model will say in the abstract. Enterprise governance is about what ",[270,4418,782],{},[270,4420,782],{}," systems and data.",[809,4423,4425],{"id":4424},"can-we-rely-on-the-secure-web-gateway","Can we rely on the secure web gateway?",[190,4427,4428],{},"Network controls help with paste-out. They do not quote a CRM change, bind an approver, or store a causal history. Use both.",[809,4430,4432],{"id":4431},"must-a-person-always-approve","Must a person always approve?",[190,4434,4435,4436,4439],{},"For many operational writes, yes. For read-only analysis, maybe not. The mistake is calling a system “human-approved” because a human ",[270,4437,4438],{},"could"," look, while changes proceed on model initiative.",[809,4441,4443],{"id":4442},"do-the-oecd-ai-principles-require-a-specific-product","Do the OECD AI Principles require a specific product?",[190,4445,4446],{},"No. They are a public commitment. A product can make evidence cheaper to produce. The commitment does not implement a gate.",[809,4448,4450],{"id":4449},"is-a-dpia-enough-to-go-live","Is a DPIA enough to go live?",[190,4452,4453],{},"It is necessary thinking, not a runtime. You still need identity, scope, fail-closed writes, and a record. The DPIA should describe those controls, not replace them.",[809,4455,4457],{"id":4456},"does-blocking-chatgpt-count-as-governance","Does blocking ChatGPT count as governance?",[190,4459,4460,4461,4464],{},"It is a network control. Without a sanctioned path that can see the right files, people use personal phones. Blocking can tighten ",[270,4462,4463],{},"after"," substitution exists.",[809,4466,4468],{"id":4467},"how-is-this-different-from-it-change-management","How is this different from IT change management?",[190,4470,4471],{},"It is the same instinct — who may change production, with what evidence — applied to a proposer that speaks English. Existing CAB processes rarely see model-initiated payloads unless the product emits them.",[809,4473,4475],{"id":4474},"who-should-be-the-named-owner-of-ai","Who should be the named owner of AI?",[190,4477,4478],{},"Someone who can inventory systems and stop a write path, not a volunteer “champion” with no authority over CRM. Federal-style guidance starts with inventory and ownership for a reason.",[809,4480,4482],{"id":4481},"can-we-govern-only-customer-facing-chatbots-and-ignore-internal-copilots","Can we govern only customer-facing chatbots and ignore internal copilots?",[190,4484,4485],{},"Internal tools still process personal data and still write to live systems. Air Canada was customer-facing. Ungoverned CRM hygiene is an internal path to the same class of invented fact.",[809,4487,4489],{"id":4488},"do-we-need-the-eu-ai-act-if-we-are-not-a-high-risk-provider","Do we need the EU AI Act if we are not a high-risk provider?",[190,4491,4492],{},"You may still have GDPR duties, sector rules, and customer contracts. Oversight and records are useful even when a specific Act title does not apply. Do not claim “Act compliant” because you have a button.",[809,4494,4496],{"id":4495},"where-does-spend-fit","Where does spend fit?",[190,4498,4499,4500,230],{},"Uncapped inference is a control failure. Quotes, ceilings, and attribution by job are governance of a scarce, abusable resource. See ",[200,4501,3749],{"href":3303},[809,4503,4505],{"id":4504},"is-an-acceptable-use-policy-still-worth-writing","Is an acceptable-use policy still worth writing?",[190,4507,4508],{},"Yes, as communication. No, as enforcement. Write the PDF. Then put the same rules in the product people actually use.",[255,4510,869],{"id":868},[190,4512,4513,217,4515,3536,4517,230],{},[200,4514,3286],{"href":228},[200,4516,3250],{"href":3249},[200,4518,4519],{"href":4371},"What is an enterprise AI operating system",[255,4521,883],{"id":882},[260,4523,4524,4530,4535,4540,4545,4550,4555,4561],{},[263,4525,4526],{},[200,4527,4529],{"href":4093,"rel":4528},[204],"Gartner, AI governance and TRiSM",[263,4531,4532],{},[200,4533,4079],{"href":4077,"rel":4534},[204],[263,4536,4537],{},[200,4538,3549],{"href":3547,"rel":4539},[204],[263,4541,4542],{},[200,4543,4100],{"href":363,"rel":4544},[204],[263,4546,4547],{},[200,4548,2228],{"href":2041,"rel":4549},[204],[263,4551,4552],{},[200,4553,3556],{"href":3554,"rel":4554},[204],[263,4556,4557],{},[200,4558,4560],{"href":599,"rel":4559},[204],"EU AI Act (Regulation 2024/1689)",[263,4562,4563],{},[200,4564,977],{"href":375,"rel":4565},[204],{"title":170,"searchDepth":171,"depth":171,"links":4567},[4568,4569,4573,4574,4588,4589],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":4570},[4571,4572],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":4575},[4576,4577,4578,4579,4580,4581,4582,4583,4584,4585,4586,4587],{"id":4412,"depth":1001,"text":4413},{"id":4424,"depth":1001,"text":4425},{"id":4431,"depth":1001,"text":4432},{"id":4442,"depth":1001,"text":4443},{"id":4449,"depth":1001,"text":4450},{"id":4456,"depth":1001,"text":4457},{"id":4467,"depth":1001,"text":4468},{"id":4474,"depth":1001,"text":4475},{"id":4481,"depth":1001,"text":4482},{"id":4488,"depth":1001,"text":4489},{"id":4495,"depth":1001,"text":4496},{"id":4504,"depth":1001,"text":4505},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"AI governance is the working rules for who may use which AI, on which data, and whether it may change a live business system — plus a record of what happened.","/blog/what-is-ai-governance",{"title":4051,"description":4590},"blog/what-is-ai-governance",[1606,340,4595,4047],"compliance","Y4LiEvlSghMUjAWsRBiJC1NWRjNmnPNAP2Naf31kScI",{"id":4598,"title":4599,"archived":164,"authors":4600,"badge":4602,"body":4603,"date":3101,"definedTerm":165,"department":165,"description":5038,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":5039,"relatedHeading":165,"seo":5040,"series":1606,"sitemap":130,"status":165,"stem":5041,"subhead":165,"tags":5042,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":5046},"content/blog/what-is-ai-token-economics.md","What is AI Token Economics",[4601],{"name":184,"to":135},{"label":1024},{"type":167,"value":4604,"toc":5014},[4605,4622,4628,4631,4638,4640,4722,4733,4735,4737,4757,4760,4783,4786,4788,4797,4802,4807,4812,4820,4822,4828,4838,4844,4852,4862,4870,4872,4875,4881,4890,4892,4896,4899,4903,4906,4910,4913,4917,4922,4926,4929,4933,4936,4940,4945,4949,4952,4956,4959,4963,4968,4972,4977,4981,4984,4986,4993,4995],[190,4606,4607,4608,4611,4612,876,4617,4621],{},"A ",[193,4609,4610],{},"token"," is a chunk of text the model reads or writes. You pay per chunk. Different models cost different amounts. Input, output, and sometimes tools all meter differently. ",[200,4613,4616],{"href":4614,"rel":4615},"https://openai.com/api/pricing/",[204],"OpenAI",[200,4618,765],{"href":4619,"rel":4620},"https://www.anthropic.com/pricing",[204]," publish those ladders. Finance still cannot run the business on “12 million tokens of vendor A’s flagship.”",[190,4623,4624,4627],{},[193,4625,4626],{},"AI token economics"," is treating that usage like a real budget: measuring, allocating, controlling, and attributing spend so operators can quote before a run, cap during it, and attribute after it — instead of a slide that says “unlimited AI.”",[190,4629,4630],{},"Without it, organisations either freeze (no production AI) or send every small task to the most expensive model until the bill becomes a board slide.",[190,4632,4633,4634,4637],{},"The unit problem is the same one cloud had in its first decade: a metered resource sold with a headcount story. Seat licences predict people. Inference predicts work. When those two are collapsed into “unlimited,” the next chunk ",[270,4635,4636],{},"feels"," free, so people pick the flagship every time. The ladder did not disappear. It hid.",[255,4639,258],{"id":257},[260,4641,4642,4648,4654,4663,4672,4678,4684,4693,4699,4708,4714],{},[263,4643,4644,4647],{},[193,4645,4646],{},"Token."," A piece of text the model processes. Not a business unit. At work, a long wiki dump and a short field extract are wildly different token counts for the same “question.”",[263,4649,4650,4653],{},[193,4651,4652],{},"Seat licence."," Predictable cost per person. Often marketed as “unlimited.” The underlying work is still metered.",[263,4655,4656,4659,4660,4662],{},[193,4657,4658],{},"Pass-through API bill."," Each team has keys. Simple. Invites key sprawl and ",[200,4661,4279],{"href":3249}," on personal keys. At work, the invoice lands in engineering while go-to-market did the looping.",[263,4664,4665,4668,4669,4671],{},[193,4666,4667],{},"Quote."," A number ",[270,4670,3299],{}," they run. At work, this is what makes a brief a decision rather than a surprise.",[263,4673,4674,4677],{},[193,4675,4676],{},"Cap / ceiling."," A hard stop. The loop cannot spend past it. At work, weekend agent loops die here instead of in next month’s cloud bill.",[263,4679,4680,4683],{},[193,4681,4682],{},"Pool."," Organisation-level allowance. At work, one department should not be able to burn the company pool on a vanity run.",[263,4685,4686,4689,4690,4692],{},[193,4687,4688],{},"Attribution."," Chargeback by job, not “the AI bill.” At work, finance can ask which ",[200,4691,1083],{"href":215}," consumed the units.",[263,4694,4695,4698],{},[193,4696,4697],{},"NTU (Nimbus Token Unit)."," Nimbus’s normalised work credit for completed AI activity — analysis, tools, runs, writes — sitting above raw provider tokens. Everyday questions can be included; heavier work consumes pool credits. Finance gets one tape measure across vendors and steps.",[263,4700,4701,4704,4705,4707],{},[193,4702,4703],{},"Model routing."," Cheaper model for simple steps, stronger only when needed. See ",[200,4706,3520],{"href":1261},". At work, classify-this-ticket should not pay flagship rates.",[263,4709,4710,4713],{},[193,4711,4712],{},"Context window."," How much text the model can see at once. Dumping the whole Drive into context is an economic choice, not a quality strategy.",[263,4715,4716,4719,4720,230],{},[193,4717,4718],{},"Stop condition."," Budget hit, empty result, human cancel. Agent loops can dominate the bill without improving the artefact. See ",[200,4721,1955],{"href":1954},[190,4723,4724,4725,4728,4729,4732],{},"The point is ",[193,4726,4727],{},"value per unit",", not minimum units regardless of outcome. Caching, wiki citations, and memory should make the ",[270,4730,4731],{},"same"," outcome cheaper over time. If unit cost of an approved update never falls, you are re-deriving folklore every run.",[255,4734,1160],{"id":1159},[190,4736,3698],{},[260,4738,4739,4745,4751],{},[263,4740,4741,4744],{},[193,4742,4743],{},"Own the budget."," Surprise invoices arrive after agents looped all weekend.",[263,4746,4747,4750],{},[193,4748,4749],{},"Run the work."," You should see a number before you commit, not a lecture after.",[263,4752,4753,4756],{},[193,4754,4755],{},"Are tempted to shame people for using AI."," Shame drives personal keys. Cap the official path so it is safe to use.",[190,4758,4759],{},"Practical rhythm:",[260,4761,4762,4771,4777],{},[263,4763,4764,4767,4768,4770],{},[193,4765,4766],{},"Name the run."," Unnamed chats cannot be attributed. That is what a ",[200,4769,1083],{"href":215}," is for.",[263,4772,4773,4776],{},[193,4774,4775],{},"Separate exploration from production."," Sandboxes can have tighter caps and cheaper default routes.",[263,4778,4779,4782],{},[193,4780,4781],{},"Review unit cost of outcomes"," — approved updates per unit — not tokens in the abstract.",[190,4784,4785],{},"Anti-pattern: a single corporate API key in a wiki, no per-job cap, monthly surprise. That is an unmetered utility.",[809,4787,3290],{"id":3289},[190,4789,4790,4792,4793,4796],{},[193,4791,3295],{}," You need a quote, a ceiling, and a chargeback dimension that matches how the business already thinks — by job, department, or cost centre — not by vendor token type. Multi-vendor ladders are incomparable until you normalise. NTU is that normalisation in Nimbus. Finance should also see ",[270,4794,4795],{},"stops",": a cap that fired is a successful control, not a failed project.",[190,4798,4799,4801],{},[193,4800,3310],{}," Spend logs are not only money. They are a map of which data classes went to which provider. Uncapped personal keys are a processing-agreement gap. Legal will also ask whether you can stop a run, not only whether you can pay for it.",[190,4803,4804,4806],{},[193,4805,3319],{}," Caps are operational stops, like a queue limit. Ops needs to know whether a paused run is waiting on a person or waiting on budget. Mixing those two in one “it failed” status is how you get the wrong pager.",[190,4808,4809,4811],{},[193,4810,3325],{}," GTM feels the quality-versus-cost trade first. A compact model that extracts fields is usually enough. A flagship model that argues a clause may be worth it. Without routing and quotes, GTM either hoards “the best model” or gets blamed for the bill. Neither produces better pipeline hygiene.",[190,4813,4814,4816,4817,4819],{},[193,4815,3331],{}," API keys are credentials. Personal keys in browser plugins are ",[200,4818,4279],{"href":3249},". A pooled official path with per-workstream ceilings reduces key sprawl. Spend spikes can also be an anomaly signal — a loop that never stops is sometimes a bug, sometimes a prompt-injection success.",[809,4821,3340],{"id":3339},[190,4823,4824,4827],{},[193,4825,4826],{},"“Unlimited” as a strategy."," Seats hide the ladder. They do not delete it. Heavy agentic work will still surface as a true-up, a throttle, or a degraded model.",[190,4829,4830,4833,4834,4837],{},[193,4831,4832],{},"Punishing usage."," Chargeback without a sanctioned path recreates personal keys. Celebrate lower units ",[270,4835,4836],{},"per artefact"," as playbooks and memory compound.",[190,4839,4840,4843],{},[193,4841,4842],{},"Tokens as the KPI."," Tokens measure consumption. Outcomes measure value. A cheap run that produces a rejected write is still waste. A dearer run that produces one approved journal may be fine.",[190,4845,4846,4849,4850,230],{},[193,4847,4848],{},"One model for everything."," That is a routing failure dressed as quality culture. See ",[200,4851,3520],{"href":1261},[190,4853,4854,463,4857,4861],{},[193,4855,4856],{},"No stop on loops.",[200,4858,4860],{"href":411,"rel":4859},[204],"Anthropic’s note on building effective agents"," treats workflows with stop conditions as the grown-up shape. Economics is one of those stops.",[190,4863,4864,4865,876,4867,4869],{},"Good looks like: named jobs, quotes before commit, hard ceilings, routing policy, attribution, and falling unit cost as the ",[200,4866,326],{"href":443},[200,4868,3186],{"href":711}," reduce re-derivation. Failure looks like a shared key, a flagship default, and a board slide titled “AI spend.”",[255,4871,793],{"id":792},[190,4873,4874],{},"Workstreams show quotes and ceilings before runs. Orgs draw from a pooled NTU allowance. Routing is a policy, not a dropdown labelled “best.” Memory and wiki reduce re-derivation, which is how unit cost of an outcome should fall over time.",[190,4876,4877,4878,4880],{},"Everyday questions can sit inside the allowance; heavier analysis, tools, and writes consume pool credits. The ",[200,4879,23],{"href":711}," can record spend as part of the chain, so “the run stopped because the ceiling was hit” is a causal fact.",[190,4882,3411,4883,876,4885,4887,4888,230],{},[200,4884,44],{"href":45},[200,4886,3520],{"href":1261},". Product context: ",[200,4889,31],{"href":32},[255,4891,807],{"id":806},[809,4893,4895],{"id":4894},"why-cant-we-just-pay-seats-and-call-it-unlimited","Why can’t we just pay seats and call it unlimited?",[190,4897,4898],{},"Seats predict headcount. Production AI spend is inference, tools, and writes. “Unlimited” hides the ladder; it does not delete it.",[809,4900,4902],{"id":4901},"what-should-finance-actually-see","What should finance actually see?",[190,4904,4905],{},"A quote before commit, a cap during the run, and attribution by job afterwards — in one unit they can compare across vendors and steps.",[809,4907,4909],{"id":4908},"wont-cheaper-models-get-worse-answers","Won’t cheaper models get worse answers?",[190,4911,4912],{},"For extract and classify, often no. For hard judgment, often yes. That is a routing policy, not a religion. Measure reject rates on the job, not vibes.",[809,4914,4916],{"id":4915},"do-we-punish-teams-for-using-ai","Do we punish teams for using AI?",[190,4918,4919,4920,4837],{},"No. Punishing usage revives shadow AI. Celebrate lower units ",[270,4921,4836],{},[809,4923,4925],{"id":4924},"what-is-an-ntu-in-plain-language","What is an NTU in plain language?",[190,4927,4928],{},"A normalised work credit above raw provider tokens, so a finance partner is not asked to compare “vendor A input tokens” with “vendor B output tokens” plus tool calls. In Nimbus, completed activity — analysis, tools, runs, writes — is what consumes the unit.",[809,4930,4932],{"id":4931},"should-every-chat-be-billed-to-a-cost-centre","Should every chat be billed to a cost centre?",[190,4934,4935],{},"Named production jobs, yes. Tiny sanctioned copilots for personal drafting can live on a lighter path. The failure is mixing them so neither can be capped.",[809,4937,4939],{"id":4938},"how-do-agent-loops-blow-the-budget","How do agent loops blow the budget?",[190,4941,4942,4943,230],{},"They call tools, re-read context, and retry without a finish line. Without a ceiling and a stop condition, “being thorough” is an unbounded loop. See ",[200,4944,1955],{"href":1954},[809,4946,4948],{"id":4947},"is-caching-the-same-as-token-economics","Is caching the same as token economics?",[190,4950,4951],{},"Caching is a tactic. Economics is the management system: quote, cap, attribute, route. Caching without attribution still leaves you unable to explain the bill.",[809,4953,4955],{"id":4954},"do-we-need-a-data-warehouse-to-do-this","Do we need a data warehouse to do this?",[190,4957,4958],{},"You need events at run time. A warehouse can hold copies for reporting. It cannot quote a run that has not emitted a number yet.",[809,4960,4962],{"id":4961},"how-does-this-relate-to-write-back","How does this relate to write-back?",[190,4964,4965,4966,230],{},"Writes are usually a small number of tokens and a large operational risk. Do not use spend as a substitute for a named signer. Do use spend as a stop so a looping agent cannot keep proposing writes all weekend. See ",[200,4967,3286],{"href":228},[809,4969,4971],{"id":4970},"can-we-lock-one-vendor-to-simplify-pricing","Can we lock one vendor to simplify pricing?",[190,4973,4974,4975,230],{},"You can. You will pay for it in price, outages, and lock-in. A normalised unit plus routing is how finance keeps a second tape measure. See ",[200,4976,3520],{"href":1261},[809,4978,4980],{"id":4979},"why-not-just-set-a-monthly-company-cap","Why not just set a monthly company cap?",[190,4982,4983],{},"A company cap without per-job attribution is a shared kitchen. The loudest workflow starves the others, and nobody can say which job did it.",[255,4985,869],{"id":868},[190,4987,4988,876,4990,230],{},[200,4989,3520],{"href":1261},[200,4991,4992],{"href":215},"What is an AI workstream",[255,4994,883],{"id":882},[260,4996,4997,5003,5009],{},[263,4998,4999],{},[200,5000,5002],{"href":4614,"rel":5001},[204],"OpenAI API pricing",[263,5004,5005],{},[200,5006,5008],{"href":4619,"rel":5007},[204],"Anthropic pricing",[263,5010,5011],{},[200,5012,923],{"href":411,"rel":5013},[204],{"title":170,"searchDepth":171,"depth":171,"links":5015},[5016,5017,5021,5022,5036,5037],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":5018},[5019,5020],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":5023},[5024,5025,5026,5027,5028,5029,5030,5031,5032,5033,5034,5035],{"id":4894,"depth":1001,"text":4895},{"id":4901,"depth":1001,"text":4902},{"id":4908,"depth":1001,"text":4909},{"id":4915,"depth":1001,"text":4916},{"id":4924,"depth":1001,"text":4925},{"id":4931,"depth":1001,"text":4932},{"id":4938,"depth":1001,"text":4939},{"id":4947,"depth":1001,"text":4948},{"id":4954,"depth":1001,"text":4955},{"id":4961,"depth":1001,"text":4962},{"id":4970,"depth":1001,"text":4971},{"id":4979,"depth":1001,"text":4980},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"AI token economics is treating AI usage like a real budget: you pay per chunk of text the model reads and writes, so finance can quote, cap, and attribute spend instead of hoping for “unlimited AI.”","/blog/what-is-ai-token-economics",{"title":4599,"description":5038},"blog/what-is-ai-token-economics",[1606,5043,5044,5045],"token-economics","ntu","model-routing","jR9bMUoNyruLuHUSydPVyMVDkAiHjgDs6DUl90nOWZc",{"id":5048,"title":5049,"archived":164,"authors":5050,"badge":5052,"body":5053,"date":1008,"definedTerm":165,"department":165,"description":5644,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":5645,"relatedHeading":165,"seo":5646,"series":1606,"sitemap":130,"status":165,"stem":5647,"subhead":165,"tags":5648,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":5650},"content/blog/what-is-an-agent-harness.md","What is an Agent Harness",[5051],{"name":184,"to":135},{"label":1024},{"type":167,"value":5054,"toc":5626},[5055,5068,5076,5087,5100,5102,5193,5207,5209,5216,5218,5235,5247,5262,5269,5273,5276,5291,5302,5315,5327,5336,5349,5362,5366,5369,5372,5378,5384,5401,5419,5423,5434,5443,5452,5459,5461,5465,5468,5472,5475,5479,5485,5489,5492,5496,5502,5506,5511,5513,5524,5526,5534,5536],[190,5056,1029,5057,5059,5060,5064,5065,5067],{},[193,5058,1036],{}," is the software around a large language model that turns next-token prediction into work: tools, memory, a loop, permissions, and a stop. ",[200,5061,5063],{"href":405,"rel":5062},[204],"LangChain’s 2026 documentation"," writes the equation in plain type: ",[193,5066,1652],{},". The model reasons. The harness is everything else.",[190,5069,5070,5071,5075],{},"That sentence is not marketing. An unaided model is stateless. It produces text. It cannot keep a file, call Salesforce, fail a linter, or refuse a write. The ",[200,5072,5074],{"href":915,"rel":5073},[204],"Wikipedia entry on agent harnesses"," records the same split, and notes that the UK’s AI Security Institute already described an AI agent as the model plus scaffolding in 2023. The industry spent two years arguing about which model was smartest. In 2026 it started arguing about which environment the model was sitting in.",[190,5077,5078,5081,5082,5086],{},[200,5079,211],{"href":209,"rel":5080},[204]," uses a body-and-brain analogy: the model is the brain; the harness is the body and the workspace. ",[200,5083,5085],{"href":202,"rel":5084},[204],"LangChain’s anatomy post"," is more mechanical. A harness is every piece of code, configuration, and execution logic that is not the model itself. A raw model is not an agent. It becomes one when a harness gives it state, tool execution, feedback loops, and constraints that do not depend on the model’s mood.",[190,5088,5089,5090,5092,5093,5096,5097,5099],{},"This article is the definition. ",[200,5091,861],{"href":358}," is the practice of tightening that environment when the agent fails. ",[200,5094,5095],{"href":286},"Inner vs outer agent harness"," is the cut between a repo and a company. An ",[200,5098,1337],{"href":864}," is the outer case: operators, signers, a ledger.",[255,5101,258],{"id":257},[260,5103,5104,5110,5116,5122,5131,5144,5160,5173,5182],{},[263,5105,5106,5109],{},[193,5107,5108],{},"Harness / scaffolding."," Same object, two eras. Scaffolding is the 2023–2024 research word. Harness is the 2026 product word. Both mean the runtime around the weights.",[263,5111,5112,5115],{},[193,5113,5114],{},"Agent."," The composed system. Not the model. Not the chat UI. If you can swap the model and the job still runs, you were looking at the harness.",[263,5117,5118,5121],{},[193,5119,5120],{},"Loop."," Plan, act, observe, repeat — the ReAct-shaped cycle popularised in 2022 and now owned by the harness, not by the prompt. The harness dispatches the tool, returns the result, and decides whether to continue.",[263,5123,5124,5126,5127,5130],{},[193,5125,2571],{}," Budget, max steps, empty retrieval, tool error, human cancel, wait-for-named-signer. “The model says it is done” is a suggestion. ",[200,5128,4860],{"href":411,"rel":5129},[204]," is honest about this: encoding the job and deciding what “done” means is the boring part that actually matters.",[263,5132,5133,5136,5137,5141,5142,230],{},[193,5134,5135],{},"Tools / skills / MCP."," Hands. The ",[200,5138,5140],{"href":296,"rel":5139},[204],"Model Context Protocol"," is a common plug so hosts can call the same servers. Plumbing. A plug is not a permission model. See ",[200,5143,1141],{"href":757},[263,5145,5146,5149,5150,5153,5154,5156,5157,5159],{},[193,5147,5148],{},"Hooks / middleware."," Deterministic intercepts on the loop. ",[200,5151,1120],{"href":460,"rel":5152},[204]," run shell or HTTP at ",[422,5155,466],{}," and can block with exit code 2. LangChain middleware is the same instinct in a library. A line in ",[422,5158,646],{}," is advice. A hook is a gate.",[263,5161,5162,463,5165,5169,5170,5172],{},[193,5163,5164],{},"Guides and sensors.",[200,5166,5168],{"href":946,"rel":5167},[204],"Birgitta Böckeler’s framing on martinfowler.com",": feed-forward context (conventions, architecture, ",[422,5171,434],{},") versus feedback (linters, tests, reviewers). A harness that only prompts is half a harness.",[263,5174,5175,5178,5179,230],{},[193,5176,5177],{},"Inner harness."," Coding agents: Claude Code, Cursor, Codex. Workspace is a repository. Tests are the eval. See ",[200,5180,5181],{"href":286},"inner vs outer",[263,5183,5184,5187,5188,217,5190,5192],{},[193,5185,5186],{},"Outer / enterprise harness."," Operators. Connectors, ",[200,5189,216],{"href":215},[200,5191,229],{"href":228},", a ledger. Workspace is the company. A passing unit test does not prove a CRM write was authorised.",[190,5194,5195,5196,217,5198,217,5200,217,5202,217,5204,5206],{},"Nimbus is one outer harness: ",[200,5197,326],{"href":28},[200,5199,221],{"href":20},[200,5201,216],{"href":32},[200,5203,340],{"href":40},[200,5205,23],{"href":24},". Claude Code is a strong inner harness. Calling either “an agent” without naming the harness is how RFPs buy a model and inherit someone else’s loop.",[255,5208,1160],{"id":1159},[190,5210,5211,5215],{},[200,5212,5214],{"href":383,"rel":5213},[204],"McKinsey’s 2025 State of AI survey"," is the scale gap in one chart: most organisations use AI in at least one function; far fewer have begun to scale. Copilots produce usage. Harnesses produce jobs that finish under a stop. If your programme is “we rolled out ChatGPT Enterprise,” you have licensed a model surface. You have not yet chosen a harness for the work that writes back.",[190,5217,1171],{},[260,5219,5220,5223,5226,5229,5232],{},[263,5221,5222],{},"the job is multi-step and tool-using, not a single completion",[263,5224,5225],{},"a live system can change (CRM, ERP, repo, ticket queue)",[263,5227,5228],{},"someone will ask, six months later, why a field or a file changed",[263,5230,5231],{},"you need to swap models without rewriting every tool",[263,5233,5234],{},"you already noticed that a better model still skips the linter, invents a policy, or pastes into Salesforce",[190,5236,5237,5238,5241,5242,5246],{},"In February 2024 a British Columbia tribunal held Air Canada responsible for a chatbot that invented a bereavement-fare policy. ",[200,5239,4209],{"href":2041,"rel":5240},[204]," that the airline’s argument — the chatbot is a separate legal entity — failed. That failure is a missing harness, not a missing model: no quote, no signer, no stop before a commitment left the building. In 2023 a New York court ",[200,5243,5245],{"href":1904,"rel":5244},[204],"sanctioned lawyers"," who filed ChatGPT-invented cases. Ungated generation reached a system of record. CRM writes are the operational twin with money attached.",[190,5248,5249,5253,5254,5257,5258,5261],{},[200,5250,5252],{"href":363,"rel":5251},[204],"NIST’s AI RMF"," organises Govern, Map, Measure, Manage. ",[200,5255,966],{"href":369,"rel":5256},[204]," is an AI ",[270,5259,5260],{},"management system"," standard. Neither is implemented by a system prompt that says “be careful.” They are implemented by a runtime that can refuse a tool call.",[190,5263,5264,5268],{},[200,5265,5267],{"href":953,"rel":5266},[204],"Addy Osmani’s 2026 write-up"," states the engineering claim operators keep rediscovering: a decent model with a great harness beats a great model with a bad harness. When the agent does something dumb, the default instinct is to blame the weights. Harness engineering treats most of those failures as configuration. That is the rest of this cluster.",[255,5270,5272],{"id":5271},"what-a-harness-actually-contains","What a harness actually contains",[190,5274,5275],{},"LangChain’s anatomy and Databricks’s list converge on the same parts. You can inspect each one before you buy a product or assemble a library.",[190,5277,5278,5281,5282,5286,5287,5290],{},[193,5279,5280],{},"The loop."," The harness owns plan → act → observe. It executes the tool. It feeds the result back. It enforces max steps and a cost budget so a stuck agent cannot run forever. ",[200,5283,5285],{"href":307,"rel":5284},[204],"Anthropic’s long-running harness note"," shows why this is not a prompt: tasks that outlast one context window need an initializer, incremental sessions, git commits, and a progress file the ",[270,5288,5289],{},"next"," session can read. The model does not remember. The environment does.",[190,5292,5293,5296,5297,5301],{},[193,5294,5295],{},"Tools and execution."," Search, shell, apply_patch, browser, CRM, ERP. The model proposes a call. The harness runs it in a sandbox or against an adapter, handles timeouts, and returns structured results. A generic HTTP tool with a production token is not a harness. It is a confused deputy. ",[200,5298,5300],{"href":375,"rel":5299},[204],"OWASP’s Top 10 for LLM applications"," still applies: excessive agency and unbounded tool use are design failures, not model quirks.",[190,5303,5304,5307,5308,5310,5311,876,5313,230],{},[193,5305,5306],{},"Context and memory."," Working memory is the current window. Session state is progress for this job. Durable memory is files, ",[422,5309,434],{},", a wiki, or a graph — something that survives compaction. Anthropic’s initializer/coding-agent split is a memory design: feature lists and commits as cross-session state. A company that stores “what we approved” only in Slack search does not have durable memory for operations. See ",[200,5312,3510],{"href":711},[200,5314,2404],{"href":443},[190,5316,5317,5320,5321,5323,5324,5326],{},[193,5318,5319],{},"Permissions and hooks."," Who may call which tool, with which identity, on which object. Claude Code’s ",[422,5322,466],{}," hook can deny Bash regardless of what the model intended. That is the inner version of ",[200,5325,1079],{"href":228},": the write API is unreachable until a named role signs a quoted payload. A prompt that says “ask Legal first” is not this layer. The model can forget. The user can paste anyway.",[190,5328,5329,5332,5333,230],{},[193,5330,5331],{},"Feedback."," Compilers, tests, linters, schema validators, human review. Böckeler’s sensors. Without them the loop is open: the model reports success and the harness believes it. Terminal-Bench and SWE-bench exist because coding harnesses can grade against an environment. Enterprise writes need an equivalent: did the signed payload match what executed. See ",[200,5334,5335],{"href":503},"eval loops for enterprise agent harnesses",[190,5337,5338,5341,5342,5345,5346,5348],{},[193,5339,5340],{},"Orchestration."," Subagents, hand-offs, model routing. Optional until the job already splits in the organisation. ",[200,5343,5344],{"href":493},"Multi-agent AI"," is the pattern. ",[200,5347,484],{"href":220}," is the hiring object. A harness that spawns specialists without a stop is a faster way to share a production login.",[190,5350,1029,5351,5354,5355,5357,5358,5361],{},[200,5352,5353],{"href":1954},"agentic workflow"," is a designed sequence with business stops. The harness is the runtime that can actually run that sequence. Mixing those two words is how demos skip isolation. A ",[200,5356,1083],{"href":215}," is the company object that hosts the job: brief, connectors, people, budget. In Nimbus the workstream is that folder; the harness is wiki + teams + connectors + gates + graph sitting around whichever model ",[200,5359,5360],{"href":45},"routing"," picks for the step.",[255,5363,5365],{"id":5364},"what-is-not-a-harness","What is not a harness",[190,5367,5368],{},"A chat window with plugins. The human is still the message bus, the permission system, and the audit log.",[190,5370,5371],{},"A system prompt. Advice inside the window. Useful. Not a stop.",[190,5373,5374,5375,5377],{},"A policy PDF. ",[200,5376,3336],{"href":3335}," is a management claim. A harness is whether an unapproved write is impossible.",[190,5379,5380,5381,5383],{},"MCP on its own. A standard plug. See ",[200,5382,2640],{"href":2639},". If the server can PATCH Salesforce from natural language, you built a bypass.",[190,5385,5386,5387,5393,5394,5397,5398,230],{},"A framework on its own. ",[200,5388,5390,5391],{"href":902,"rel":5389},[204],"LangChain’s ",[422,5392,407],{}," is a way to ",[270,5395,5396],{},"assemble"," a harness. CrewAI, LangGraph, and Pydantic AI are in the same neighbourhood. You still have to choose tools, stops, and identity. See ",[200,5399,5400],{"href":593},"agent harness vs agent framework",[190,5402,5403,5404,876,5409,5414,5415,876,5417,230],{},"A copilot seat. ",[200,5405,5408],{"href":5406,"rel":5407},"https://openai.com/business/chatgpt-enterprise/",[204],"ChatGPT Enterprise",[200,5410,5413],{"href":5411,"rel":5412},"https://www.microsoft.com/en-us/microsoft-365/copilot",[204],"Microsoft 365 Copilot"," are excellent personal surfaces. They are not, by default, a company loop with fail-closed writes. See ",[200,5416,2339],{"href":2338},[200,5418,2334],{"href":1398},[255,5420,5422],{"id":5421},"how-this-shows-up-in-products","How this shows up in products",[190,5424,5425,5428,5429,467,5431,5433],{},[193,5426,5427],{},"Coding harnesses."," Claude Code, Cursor, Codex, open shells like OpenHands. Workspace is a checkout. ",[422,5430,646],{},[422,5432,434],{}," are guides. Hooks, tests, and CI are sensors. Eval is SWE-bench or Terminal-Bench, or your own suite. These are the right shape for software.",[190,5435,5436,5439,5440,5442],{},[193,5437,5438],{},"Library harnesses."," LangChain ",[422,5441,407],{},", Deep Agents, LangGraph graphs. You compose the loop in code. You own production identity. Good when the job is yours to engineer. A liability when operators are expected to “just add Salesforce.”",[190,5444,5445,5448,5449,5451],{},[193,5446,5447],{},"Enterprise / outer harnesses."," Palantir AIP, Salesforce Agentforce, and self-service OS-class products such as Nimbus. The workspace is a job with connectors and people, not a git root. The interesting stop is a named signer on a quoted write, not a green test. ",[200,5450,1659],{"href":251}," is the buying sheet.",[190,5453,5454,5455,5458],{},"Nimbus’s mapping is deliberate and not unique as a ",[270,5456,5457],{},"category",": Perception orients, Conflux collaborates, agent teams run, governance quotes, the graph records. You can score that mapping against the parts above. You should score AIP and Agentforce the same way. Category names do not substitute for a failed write.",[255,5460,807],{"id":806},[809,5462,5464],{"id":5463},"is-the-model-the-agent","Is the model the agent?",[190,5466,5467],{},"No. The agent is model plus harness. Shopping for a model is shopping for a chip. Shopping for a harness is shopping for how work finishes.",[809,5469,5471],{"id":5470},"do-i-need-a-harness-for-a-single-prompt","Do I need a harness for a single prompt?",[190,5473,5474],{},"No. A completion does not need a loop. Multi-step tool use does. Long-running work that outlasts one window does. Writes to live systems do.",[809,5476,5478],{"id":5477},"is-rag-a-harness","Is RAG a harness?",[190,5480,5481,5482,5484],{},"Retrieval is a tool and a memory pattern inside a step. ",[200,5483,439],{"href":438}," does not dispatch tools, enforce a signer, or persist a decision. Useful. Incomplete.",[809,5486,5488],{"id":5487},"can-i-just-use-mcp-as-my-harness","Can I just use MCP as my harness?",[190,5490,5491],{},"You can use MCP as the plug. You still need identity, scope, quoting, and a stop. The spec does not require those.",[809,5493,5495],{"id":5494},"will-a-better-model-shrink-the-harness","Will a better model shrink the harness?",[190,5497,5498,5501],{},[200,5499,2702],{"href":953,"rel":5500},[204]," and Anthropic’s long-running work both say the ceiling moves. Tasks that were unreachable come into play and bring new failure modes. Stronger models still do not know your signer, your budget, or your CRM field map.",[809,5503,5505],{"id":5504},"how-is-this-different-from-an-enterprise-ai-os","How is this different from an enterprise AI OS?",[190,5507,1029,5508,5510],{},[200,5509,4372],{"href":4371}," is the company-shaped product: wiki, workstreams, teams, gates, ledger. A harness is the runtime idea underneath — including coding harnesses that are not an OS. Nimbus is an OS-class outer harness. Claude Code is not an OS. Both are harnesses.",[809,5512,853],{"id":852},[190,5514,5515,5517,5518,5520,5521,5523],{},[200,5516,861],{"href":358}," for the practice. ",[200,5519,195],{"href":2140}," for the parts in one diagram. ",[200,5522,1659],{"href":251}," before a vendor demo.",[255,5525,869],{"id":868},[190,5527,5528,217,5530,3536,5532,230],{},[200,5529,1955],{"href":1954},[200,5531,494],{"href":493},[200,5533,3286],{"href":228},[255,5535,883],{"id":882},[260,5537,5538,5544,5549,5554,5559,5564,5569,5574,5579,5584,5589,5594,5599,5604,5610,5615,5621],{},[263,5539,5540],{},[200,5541,5543],{"href":405,"rel":5542},[204],"LangChain, Agents (Agent = Model + Harness)",[263,5545,5546],{},[200,5547,891],{"href":202,"rel":5548},[204],[263,5550,5551],{},[200,5552,904],{"href":902,"rel":5553},[204],[263,5555,5556],{},[200,5557,917],{"href":915,"rel":5558},[204],[263,5560,5561],{},[200,5562,910],{"href":209,"rel":5563},[204],[263,5565,5566],{},[200,5567,923],{"href":411,"rel":5568},[204],[263,5570,5571],{},[200,5572,929],{"href":307,"rel":5573},[204],[263,5575,5576],{},[200,5577,948],{"href":946,"rel":5578},[204],[263,5580,5581],{},[200,5582,955],{"href":953,"rel":5583},[204],[263,5585,5586],{},[200,5587,983],{"href":383,"rel":5588},[204],[263,5590,5591],{},[200,5592,4100],{"href":363,"rel":5593},[204],[263,5595,5596],{},[200,5597,966],{"href":369,"rel":5598},[204],[263,5600,5601],{},[200,5602,2228],{"href":2041,"rel":5603},[204],[263,5605,5606],{},[200,5607,5609],{"href":1904,"rel":5608},[204],"Reuters, New York lawyers sanctioned over ChatGPT citations",[263,5611,5612],{},[200,5613,977],{"href":375,"rel":5614},[204],[263,5616,5617],{},[200,5618,5620],{"href":296,"rel":5619},[204],"Model Context Protocol specification (2025-11-25)",[263,5622,5623],{},[200,5624,941],{"href":460,"rel":5625},[204],{"title":170,"searchDepth":171,"depth":171,"links":5627},[5628,5629,5630,5631,5632,5633,5642,5643],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160},{"id":5271,"depth":171,"text":5272},{"id":5364,"depth":171,"text":5365},{"id":5421,"depth":171,"text":5422},{"id":806,"depth":171,"text":807,"children":5634},[5635,5636,5637,5638,5639,5640,5641],{"id":5463,"depth":1001,"text":5464},{"id":5470,"depth":1001,"text":5471},{"id":5477,"depth":1001,"text":5478},{"id":5487,"depth":1001,"text":5488},{"id":5494,"depth":1001,"text":5495},{"id":5504,"depth":1001,"text":5505},{"id":852,"depth":1001,"text":853},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"An agent harness is everything around a model that lets it do work: tools, memory, permissions, loops, and stops — Agent = Model + Harness, not a chat window with plugins.","/blog/what-is-an-agent-harness",{"title":5049,"description":5644},"blog/what-is-an-agent-harness",[1606,1015,3587,5649],"harness-engineering","QJwyjpDmAFo0HEREXF3rzNUYmrP7mfEYXyPR7kZLbgw",{"id":5652,"title":5653,"archived":164,"authors":5654,"badge":5656,"body":5657,"date":3101,"definedTerm":165,"department":165,"description":6059,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":6060,"relatedHeading":165,"seo":6061,"series":1606,"sitemap":130,"status":165,"stem":6062,"subhead":165,"tags":6063,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":6065},"content/blog/what-is-an-agentic-workflow.md","What is an Agentic Workflow",[5655],{"name":184,"to":135},{"label":1024},{"type":167,"value":5658,"toc":6035},[5659,5662,5671,5677,5680,5682,5685,5707,5710,5752,5761,5763,5766,5773,5776,5810,5813,5815,5820,5830,5835,5840,5853,5855,5861,5867,5873,5879,5887,5890,5892,5899,5905,5913,5915,5919,5922,5926,5929,5933,5936,5940,5943,5947,5952,5956,5959,5963,5966,5970,5976,5980,5983,5987,5992,5996,6001,6005,6008,6010,6016,6018],[190,5660,5661],{},"“We have an agent” often means a chat that never knows when to stop. Someone types a goal. The model keeps calling tools until the budget dies, or until a human closes the tab. There is no finish line. There is a conversation that looked busy.",[190,5663,4607,5664,5667,5668,5670],{},[193,5665,5666],{},"workflow"," has steps and a stop. An ",[193,5669,5353],{}," is a sequence of steps an AI can run toward a goal, with rules for when to stop — including a person who must approve before a live system changes.",[190,5672,5673,5676],{},[200,5674,4860],{"href":411,"rel":5675},[204]," makes the same cut: workflows with tools and stop conditions, not endless chat. The note is worth reading because it is honest about the boring parts — encoding the job, bounding the tools, and deciding what “done” means — rather than treating fluency as a process.",[190,5678,5679],{},"Older automation without models is brittle but auditable. Models without a workflow are flexible but unaccountable. An agentic workflow is the attempt to get both: language where the input is messy, and a finish line where the company needs one.",[255,5681,258],{"id":257},[190,5683,5684],{},"Vendors collapse three different layers into the word “agentic”:",[260,5686,5687,5693,5699],{},[263,5688,5689,5692],{},[193,5690,5691],{},"Agentic capability."," The model can use tools, plan, and reflect. At work, this is “it can search Drive and draft a note.” It is not yet a job.",[263,5694,5695,5698],{},[193,5696,5697],{},"Agentic workflow."," A designed sequence of those capabilities, with business stop conditions. This article is about this layer. At work, this is “extract, compare to the playbook, quote the CRM fields, wait for the named signer, write or refuse.”",[263,5700,5701,5704,5705,230],{},[193,5702,5703],{},"Agent platform."," Identity, connectors, tests, and governance around many workflows. At work, this is closer to an ",[200,5706,4372],{"href":4371},[190,5708,5709],{},"Other terms:",[260,5711,5712,5718,5723,5731,5739,5745],{},[263,5713,5714,5717],{},[193,5715,5716],{},"Tool."," An action the AI can take: search files, query CRM, post a message. At work, a tool is a hand. Hands are not roles, and they are not stop conditions.",[263,5719,5720,5722],{},[193,5721,4718],{}," Budget hit, waiting on approval, error, empty result, human cancel. “The model says it is done” is a weak stop by itself.",[263,5724,5725,5728,5729,230],{},[193,5726,5727],{},"Write-back."," The AI is allowed to change a live system, not just draft. See ",[200,5730,3286],{"href":228},[263,5732,5733,5736,5737,230],{},[193,5734,5735],{},"Human wait."," A step in the sequence, not an interruption. See ",[200,5738,3923],{"href":1896},[263,5740,5741,5744],{},[193,5742,5743],{},"Version."," Which workflow definition ran. When policy changes, retrieval changes. Operators need to know which version ran last Tuesday.",[263,5746,5747,5749,5750,230],{},[193,5748,1137],{}," A common plug so AI apps can use the same tools. Plumbing. It does not define your stops. See ",[200,5751,1141],{"href":757},[190,5753,4607,5754,5756,5757,5760],{},[200,5755,1083],{"href":215}," is the company object that ",[270,5758,5759],{},"hosts"," the workflow: brief, connectors, people, budget, finish line. The workflow is the sequence. The workstream is the job folder. Mixing those two words is how demos skip isolation.",[255,5762,1160],{"id":1159},[190,5764,5765],{},"Capability demos look like workflows. They are not. A fluent plan is not a paused run waiting on approval, a failed run that did not retry a write, or a replay of which step ran.",[190,5767,5768,5769,5772],{},"It affects you if the job is ",[193,5770,5771],{},"multi-step, tool-using, and repeated"," — the opposite of one-off chat. Close checklists, renewal playbooks, and incident runbooks already have steps. Encode those. If the job is not written down, you will encode folklore and then fight the folklore.",[190,5774,5775],{},"Practical rules:",[260,5777,5778,5784,5790,5796,5802],{},[263,5779,5780,5783],{},[193,5781,5782],{},"Read-heavy workflows"," can be long. They should still finish in an artefact with sources.",[263,5785,5786,5789],{},[193,5787,5788],{},"Write-heavy workflows"," should be short after the quote: one payload, one gate, one execution, one record. Do not hide ten writes in a “cleanup agent.”",[263,5791,5792,5795],{},[193,5793,5794],{},"Human wait is a step",", not an interruption.",[263,5797,5798,5801],{},[193,5799,5800],{},"Version the workflow."," Policy and retrieval drift. Last Tuesday’s run needs a definition you can still open.",[263,5803,5804,5807,5808,230],{},[193,5805,5806],{},"Budget is a stop."," See ",[200,5809,3749],{"href":3303},[190,5811,5812],{},"A mega-agent with “figure it out” as the spec is not a workflow. It is a hope.",[809,5814,3290],{"id":3289},[190,5816,5817,5819],{},[193,5818,3295],{}," Close and forecast jobs already have checklists. An agentic workflow that posts a journal without a stop at the named signer is not “agentic.” It is unattended posting. Finance also needs spend stops so a retry loop cannot become the month’s inference bill.",[190,5821,5822,5824,5825,5829],{},[193,5823,3310],{}," Customer-facing steps and anything that asserts a term need a gate before send. Air Canada’s chatbot invented a bereavement fare and the company was held to it — ",[200,5826,5828],{"href":2041,"rel":5827},[204],"CBC’s report"," is the cautionary case for “the workflow ended at the message.” Legal also cares that the workflow version is reconstructable.",[190,5831,5832,5834],{},[193,5833,3319],{}," This is the native language: runbooks, queues, retries, and “do not proceed.” Ops should refuse workflows that cannot pause cleanly, cannot show which step failed, and cannot distinguish “waiting on a person” from “waiting on a tool error.”",[190,5836,5837,5839],{},[193,5838,3325],{}," Renewal and hygiene jobs are repeated and tool-using. GTM should demand a short write path after the quote, not a weekend “cleanup” that touches hundreds of records behind one click. Time-to-approved-write is the metric, not time-to-first-plan.",[190,5841,5842,5844,5845,5848,5849,5852],{},[193,5843,3331],{}," Tool belts are attack surface. Prompt injection that tricks a model into ",[270,5846,5847],{},"requesting"," a write should still die at a fail-closed gate. Importing every MCP helper into one workflow is how a demo becomes one actor with every production login. The ",[200,5850,977],{"href":375,"rel":5851},[204]," treats tool use as a security topic for this reason.",[809,5854,3340],{"id":3339},[190,5856,5857,5860],{},[193,5858,5859],{},"Chat as workflow."," A conversation that looks busy has no durable instance, no version, and no gate.",[190,5862,5863,5866],{},[193,5864,5865],{},"A checklist in a prompt."," A start. Without tools, a durable job, and a stop, it is still a prompt.",[190,5868,5869,5872],{},[193,5870,5871],{},"Replacing a stable bot."," If the job is a scheduled export, older automation is the right tool. Agentic workflows help on messy documents. They are not a prestige upgrade for a cron job.",[190,5874,5875,5878],{},[193,5876,5877],{},"Fully autonomous production."," Only for actions you would already automate without a model, plus logging. If you would not let a scheduled job do it, do not let an agent do it unattended.",[190,5880,5881,5884,5885,230],{},[193,5882,5883],{},"Multi-agent as a requirement."," A single tool-using agent can execute a workflow. Multiple agents help when duties already split. See ",[200,5886,494],{"href":493},[190,5888,5889],{},"Good looks like: named steps, bounded tools, explicit stops (including human wait and budget), versioned definitions, read-only by default, fail-closed writes. Failure looks like a flagship model with every connector and a spec that says “be helpful.”",[255,5891,793],{"id":792},[190,5893,5894,5895,230],{},"Nimbus’s delivery unit for operators is the ",[193,5896,5897],{},[200,5898,1083],{"href":215},[190,5900,5901,5902,5904],{},"The mapping in everyday terms: the brief is the goal; ",[200,5903,221],{"href":493}," run the steps; connectors are the tools (default read-only); wiki is the playbook the steps must respect; governance is the wait/write stop; the Lifecycle Graph is the executed run. Model routing chooses the brain per step; it does not choose the stop.",[190,5906,3411,5907,217,5909,3536,5911,230],{},[200,5908,31],{"href":32},[200,5910,685],{"href":20},[200,5912,39],{"href":40},[255,5914,807],{"id":806},[809,5916,5918],{"id":5917},"is-a-checklist-in-a-prompt-an-agentic-workflow","Is a checklist in a prompt an agentic workflow?",[190,5920,5921],{},"It is a start. If there are no tools, no durable instance, and no gate, it is a prompt.",[809,5923,5925],{"id":5924},"how-is-this-different-from-older-robotic-automation","How is this different from older robotic automation?",[190,5927,5928],{},"Older automation executes deterministic steps. Agentic workflows add language and planning. That helps on messy documents. It also means you need tests and human gates. Do not replace a stable bot with an agent if the job is still a scheduled export.",[809,5930,5932],{"id":5931},"do-agentic-workflows-require-multiple-agents","Do agentic workflows require multiple agents?",[190,5934,5935],{},"No. A single tool-using agent can execute a workflow. Multiple agents help when duties already split in the organisation.",[809,5937,5939],{"id":5938},"can-a-workflow-be-fully-autonomous-in-production","Can a workflow be fully autonomous in production?",[190,5941,5942],{},"Only for actions you would already automate without a model, plus logging.",[809,5944,5946],{"id":5945},"where-do-tool-connection-standards-fit","Where do tool-connection standards fit?",[190,5948,5949,5950,230],{},"A common plug so AI apps can use the same tools is plumbing. It does not define your stops or approvals. See ",[200,5951,1141],{"href":757},[809,5953,5955],{"id":5954},"what-is-a-good-stop-condition-besides-the-model-is-done","What is a good stop condition besides “the model is done”?",[190,5957,5958],{},"Budget ceiling, empty retrieval, tool error, human cancel, and wait-for-named-signer. “Done” from the model is a suggestion. Encode the others.",[809,5960,5962],{"id":5961},"how-long-should-a-write-heavy-workflow-be","How long should a write-heavy workflow be?",[190,5964,5965],{},"Short after the quote. One payload, one gate, one execution, one record. Length belongs in the read and compare steps, not in a bundle of hidden mutations.",[809,5967,5969],{"id":5968},"how-do-we-version-a-workflow-when-the-wiki-changes","How do we version a workflow when the wiki changes?",[190,5971,5972,5973,5975],{},"Treat the playbook version as an input to the run. The ",[200,5974,3186],{"href":711}," should cite which wiki version the steps respected. Changing policy without recording which definition ran is how Tuesday becomes unexplained.",[809,5977,5979],{"id":5978},"is-agentic-the-same-as-autonomous","Is “agentic” the same as “autonomous”?",[190,5981,5982],{},"No. Agentic means the model can plan and use tools. Autonomy is a policy about whether a person must still sign. Most production writes should not be autonomous.",[809,5984,5986],{"id":5985},"can-we-import-every-available-tool-and-let-the-model-choose","Can we import every available tool and let the model choose?",[190,5988,5989,5990,230],{},"That is a confused workflow. Least privilege applies to tools as much as to data. See ",[200,5991,4992],{"href":215},[809,5993,5995],{"id":5994},"how-does-this-relate-to-human-in-the-loop","How does this relate to human-in-the-loop?",[190,5997,5998,5999,230],{},"Human wait is a first-class step. If the person is only “on the loop” with a kill switch, you have a different design. See ",[200,6000,3923],{"href":1896},[809,6002,6004],{"id":6003},"will-a-better-model-remove-the-need-for-a-workflow","Will a better model remove the need for a workflow?",[190,6006,6007],{},"Stronger models plan more fluently. They still do not know your finish line, your signer, or your budget. Fluency without stops is a more expensive loop.",[255,6009,869],{"id":868},[190,6011,6012,876,6014,230],{},[200,6013,4519],{"href":4371},[200,6015,494],{"href":493},[255,6017,883],{"id":882},[260,6019,6020,6025,6030],{},[263,6021,6022],{},[200,6023,923],{"href":411,"rel":6024},[204],[263,6026,6027],{},[200,6028,2228],{"href":2041,"rel":6029},[204],[263,6031,6032],{},[200,6033,977],{"href":375,"rel":6034},[204],{"title":170,"searchDepth":171,"depth":171,"links":6036},[6037,6038,6042,6043,6057,6058],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":6039},[6040,6041],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":6044},[6045,6046,6047,6048,6049,6050,6051,6052,6053,6054,6055,6056],{"id":5917,"depth":1001,"text":5918},{"id":5924,"depth":1001,"text":5925},{"id":5931,"depth":1001,"text":5932},{"id":5938,"depth":1001,"text":5939},{"id":5945,"depth":1001,"text":5946},{"id":5954,"depth":1001,"text":5955},{"id":5961,"depth":1001,"text":5962},{"id":5968,"depth":1001,"text":5969},{"id":5978,"depth":1001,"text":5979},{"id":5985,"depth":1001,"text":5986},{"id":5994,"depth":1001,"text":5995},{"id":6003,"depth":1001,"text":6004},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"An agentic workflow is a sequence of steps an AI can run toward a goal, with rules for when to stop — including a person who must approve before a live system changes.","/blog/what-is-an-agentic-workflow",{"title":5653,"description":6059},"blog/what-is-an-agentic-workflow",[1606,6064,3587,216],"agentic-workflow","KI2LgTR-lutBOE9CeaOi2IYIQJBJfcrcEobQ8g8Lml8",{"id":6067,"title":6068,"archived":164,"authors":6069,"badge":6071,"body":6072,"date":3101,"definedTerm":165,"department":165,"description":6491,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":6492,"relatedHeading":165,"seo":6493,"series":1606,"sitemap":130,"status":165,"stem":6494,"subhead":165,"tags":6495,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":6496},"content/blog/what-is-an-ai-workstream.md","What is an AI Workstream",[6070],{"name":184,"to":135},{"label":1024},{"type":167,"value":6073,"toc":6467},[6074,6081,6084,6087,6096,6099,6101,6172,6177,6179,6185,6188,6207,6210,6235,6238,6240,6245,6253,6258,6263,6271,6273,6279,6284,6290,6296,6302,6308,6311,6317,6319,6326,6329,6335,6346,6348,6352,6355,6359,6362,6366,6369,6373,6379,6383,6388,6392,6395,6399,6404,6408,6411,6415,6420,6424,6427,6431,6434,6438,6443,6445,6451,6453],[190,6075,6076,6077,6080],{},"An AI workstream is a ",[193,6078,6079],{},"shared workspace for one job",": a brief, the tools allowed, the people and AI on it, a budget, and a finish line.",[190,6082,6083],{},"A Slack channel with a bot is not a job. It is a room. Anyone can paste anything. The bot never knows when the work is done. Next quarter, nobody can say which systems were in play or who was allowed to change them.",[190,6085,6086],{},"If you cannot name the systems in scope and the approval policy on writes, you do not have a workstream. You have a conversation.",[190,6088,6089,6090,6095],{},"Software teams already learned this. Work lives in issues and tickets, not in unbounded chat. ",[200,6091,6094],{"href":6092,"rel":6093},"https://www.atlassian.com/agile/project-management/epics-stories-themes",[204],"Atlassian’s epics and stories"," are named packages with a boundary. AI operations are still catching up. The missing object is often the work package: a place where the job actually lives.",[190,6097,6098],{},"The analogy is not decoration. Tickets have a requester, a scope, an owner, and a closed state. Copilots have a thread. Threads do not archive cleanly, do not attach least-privilege connectors, and do not carry a named signer. When AI started touching live systems, the thread stopped being a sufficient container.",[255,6100,258],{"id":257},[260,6102,6103,6109,6115,6129,6135,6143,6148,6156,6164],{},[263,6104,6105,6108],{},[193,6106,6107],{},"Brief."," What this job is for, and what “done” means. At work, “Q3 regional discount hygiene” is a brief. “My stuff” is not.",[263,6110,6111,6114],{},[193,6112,6113],{},"Connector."," A secure link to a live system (CRM, ERP, Drive). Attach what this job needs — not every system “just in case.” Default is read-only.",[263,6116,6117,4184,6120,6122,6123,6128],{},[193,6118,6119],{},"Scope / least privilege.",[270,6121,4187],{}," job. ",[200,6124,6127],{"href":6125,"rel":6126},"https://www.hhs.gov/hipaa/for-professionals/privacy/guidance/minimum-necessary-requirement/index.html",[204],"HIPAA’s minimum necessary"," is the same instinct: do not attach every system to every task. GDPR purpose limitation is the privacy-law cousin.",[263,6130,6131,6134],{},[193,6132,6133],{},"God workspace."," One org-wide chat that can see every folder and every CRM object because setup was easier. At work, this is how recruiting sees finance forecasts.",[263,6136,6137,6140,6141,230],{},[193,6138,6139],{},"Agent team."," The AI specialists assigned to the job. The workstream is the stage; the team is the cast. See ",[200,6142,494],{"href":493},[263,6144,6145,6147],{},[193,6146,4138],{}," Who must approve a write. At work, this is a role that already owns that class of change.",[263,6149,6150,6153,6154,230],{},[193,6151,6152],{},"NTU / budget."," The spend ceiling for the job. See ",[200,6155,3749],{"href":3303},[263,6157,6158,6161,6162,230],{},[193,6159,6160],{},"Wiki section."," The asserted playbooks this job may load. See ",[200,6163,2404],{"href":443},[263,6165,6166,6169,6170,230],{},[193,6167,6168],{},"Lifecycle Graph."," The chain this job emits as it runs. See ",[200,6171,3510],{"href":711},[190,6173,1029,6174,6176],{},[200,6175,5353],{"href":1954}," is the sequence of steps. The workstream is the durable instance those steps run inside. A workflow definition without a workstream is a script on someone’s laptop. A workstream without a workflow is a folder with no process.",[255,6178,1160],{"id":1159},[190,6180,6181,6182,6184],{},"Without a boundary, two departments sharing an AI tool will either over-share (the recruiting job can see finance forecasts) or under-share (people export spreadsheets to personal ChatGPT). The workstream is the compromise: enough context to do ",[270,6183,4187],{}," job, not the whole company.",[190,6186,6187],{},"It affects you if work:",[260,6189,6190,6193,6196,6199,6202],{},[263,6191,6192],{},"touches more than one system",[263,6194,6195],{},"involves more than one role",[263,6197,6198],{},"can change a live record",[263,6200,6201],{},"needs a budget you can attribute",[263,6203,6204,6205],{},"must still be explainable after people leave — see ",[200,6206,3510],{"href":711},[190,6208,6209],{},"Open workstreams the way you would open a ticket:",[260,6211,6212,6219,6226,6229,6232],{},[263,6213,6214,6215,6218],{},"One workstream per ",[193,6216,6217],{},"outcome",", not per person. “Q3 regional discount hygiene” can have several humans. “My stuff” cannot be governed or archived.",[263,6220,6221,6222,6225],{},"Attach the ",[193,6223,6224],{},"minimum"," connectors.",[263,6227,6228],{},"Set the write policy on day one, even if you start read-only.",[263,6230,6231],{},"Reuse templates, not last month’s chat thread.",[263,6233,6234],{},"Close or archive when the job ends. A sprint that never ends is not a sprint.",[190,6236,6237],{},"A standing “Ask AI” workstream with org-wide connectors recreates the copilot, including the blast radius.",[809,6239,3290],{"id":3289},[190,6241,6242,6244],{},[193,6243,3295],{}," Chargeback becomes possible because the job is named. Close workstreams can attach ERP read-only, load the close checklist from the wiki, and keep GTM out of the ledger. A company-wide AI pool with no workstream attribution is a shared kitchen.",[190,6246,6247,6249,6250,6252],{},[193,6248,3310],{}," Scope is a processing purpose. A workstream for a renewal can include legal and go-to-market on ",[270,6251,4187],{}," goal without merging their entire universes. Legal also gets a closed state: when the job ends, retention follows the type of record instead of an immortal channel.",[190,6254,6255,6257],{},[193,6256,3319],{}," This is the ticket analogue they already wanted. Ops should refuse god workspaces, insist on a finish line, and treat human wait as a status, not a side conversation in Slack.",[190,6259,6260,6262],{},[193,6261,3325],{}," Cross-functional launches finally have a place that is not a merged Slack. GTM still should not get finance’s ERP “for context.” Templates beat copying last quarter’s thread, which silently copies last quarter’s over-attached connectors.",[190,6264,6265,6267,6268,6270],{},[193,6266,3331],{}," Least privilege is now a product object, not a memo. Connectors default to read-only. Adding a write path is a deliberate change to ",[270,6269,4187],{}," job, not a tenant-wide toggle. A workstream that never closes is a standing access grant.",[809,6272,3340],{"id":3339},[190,6274,6275,6278],{},[193,6276,6277],{},"One workstream per person."," You cannot archive “my stuff.” You cannot attribute it. You cannot apply least privilege.",[190,6280,6281,6283],{},[193,6282,6133],{}," Setup is easier. Blast radius is the company.",[190,6285,6286,6289],{},[193,6287,6288],{},"ChatGPT Project as the unit."," Some files, some instructions. Typically no connector-level least privilege, quoted writes, spend caps, or lasting record. Fine for personal research. Not an operations unit.",[190,6291,6292,6295],{},[193,6293,6294],{},"Too small."," If setup exceeds the job, use a lighter sanctioned copilot path. Do not open a workstream to rewrite one sentence.",[190,6297,6298,6301],{},[193,6299,6300],{},"Too large."," If you cannot explain the purpose in one sentence, or you keep attaching “one more connector,” split.",[190,6303,6304,6307],{},[193,6305,6306],{},"Never closing."," Standing rooms recreate Slack, including the archaeology problem.",[190,6309,6310],{},"Good looks like: one outcome, minimum connectors, write policy on day one, wiki sections subscribed, budget capped, named signer, archive when done. Failure looks like an org-wide copilot with every OAuth grant and a channel that outlives the campaign.",[190,6312,3395,6313,6316],{},[200,6314,6315],{"href":4371},"enterprise AI OS"," metaphor is isolation plus I/O plus state. The workstream is the isolation unit. Without it, connectors, wiki, and agent teams have nowhere to attach that an auditor could name.",[255,6318,793],{"id":792},[190,6320,6321,6322,6325],{},"In Nimbus, workstreams are how ",[200,6323,6324],{"href":1954},"agentic workflows"," become company objects rather than a file only one engineer can run.",[190,6327,6328],{},"Each workstream carries a brief, wiki sections (approved playbooks), connector attachments (read-only by default), agent team assignment, spend budget, release policy on writes, and nodes on the Lifecycle Graph.",[190,6330,6331,6332,6334],{},"Cross-department work is multiple teams on one workstream, not a merged Slack. Operators open this themselves; the point of an ",[200,6333,4372],{"href":4371}," is that the job folder is a product, not a forward-deployed spreadsheet.",[190,6336,3411,6337,6339,6340,217,6342,217,6344,230],{},[200,6338,31],{"href":32},". Related product: ",[200,6341,685],{"href":20},[200,6343,3414],{"href":28},[200,6345,39],{"href":40},[255,6347,807],{"id":806},[809,6349,6351],{"id":6350},"is-a-chatgpt-project-a-workstream","Is a ChatGPT “Project” a workstream?",[190,6353,6354],{},"It is a weak analogue: some files, some custom instructions. It typically lacks connector-level least privilege, quoted writes, spend caps, and a lasting record. Useful for personal research. Not an operations unit.",[809,6356,6358],{"id":6357},"how-small-is-too-small","How small is too small?",[190,6360,6361],{},"If the setup cost exceeds the job, use a lighter sanctioned copilot path. Do not create a workstream to rewrite one sentence.",[809,6363,6365],{"id":6364},"how-large-is-too-large","How large is too large?",[190,6367,6368],{},"If you cannot explain the purpose in one sentence, or you keep attaching “one more connector,” split.",[809,6370,6372],{"id":6371},"can-one-workstream-serve-multiple-departments","Can one workstream serve multiple departments?",[190,6374,6375,6376,6378],{},"Yes — go-to-market and legal on a renewal, for example. They share ",[270,6377,4187],{}," goal’s scope, not each other’s entire universe.",[809,6380,6382],{"id":6381},"how-do-we-budget-them","How do we budget them?",[190,6384,6385,6386,230],{},"Caps per workstream, plus an organisation pool. Chargeback by workstream beats “the AI bill.” See ",[200,6387,3749],{"href":3303},[809,6389,6391],{"id":6390},"is-a-slack-channel-with-a-bot-enough-if-we-add-a-approve-command","Is a Slack channel with a bot enough if we add a /approve command?",[190,6393,6394],{},"No. A command is not connector least privilege, a quoted payload, a durable chain, or an archive policy. It is still a room.",[809,6396,6398],{"id":6397},"who-is-allowed-to-open-a-workstream","Who is allowed to open a workstream?",[190,6400,6401,6402,230],{},"Whoever is allowed to open that class of job in analogue life — with the same instinct as who may open a ticket or a change request. An “AI team” bottleneck recreates the waitlist that causes ",[200,6403,4279],{"href":3249},[809,6405,6407],{"id":6406},"what-happens-when-the-job-ends","What happens when the job ends?",[190,6409,6410],{},"Close or archive. Revoke standing connector usefulness. Keep the reconstructable chain according to retention, not the entire chat.",[809,6412,6414],{"id":6413},"do-we-need-a-workstream-for-read-only-analysis","Do we need a workstream for read-only analysis?",[190,6416,6417,6418,230],{},"When the analysis crosses systems, roles, or must be replayed later, yes. When it is personal drafting with no live-system scope, a sanctioned copilot may be enough. See ",[200,6419,2339],{"href":2338},[809,6421,6423],{"id":6422},"how-do-wiki-and-connectors-differ-inside-a-workstream","How do wiki and connectors differ inside a workstream?",[190,6425,6426],{},"Wiki is asserted policy the job must follow. Connectors are live systems the job may read (and, if enabled, write). Mixing them into one “knowledge” pile is how Drive folklore overwrites the playbook.",[809,6428,6430],{"id":6429},"can-we-keep-one-standing-workstream-for-ask-anything","Can we keep one standing workstream for “ask anything”?",[190,6432,6433],{},"You can. You will recreate the copilot, including over-share. Standing Q&A belongs on a tightly scoped, read-only path if it exists at all.",[809,6435,6437],{"id":6436},"how-does-this-relate-to-agent-teams","How does this relate to agent teams?",[190,6439,6440,6441,230],{},"The workstream is the job. The agent team is the cast assigned to it. Changing the cast does not change the brief, the connectors, or the signer. See ",[200,6442,494],{"href":493},[255,6444,869],{"id":868},[190,6446,6447,876,6449,230],{},[200,6448,4519],{"href":4371},[200,6450,1955],{"href":1954},[255,6452,883],{"id":882},[260,6454,6455,6461],{},[263,6456,6457],{},[200,6458,6460],{"href":6092,"rel":6459},[204],"Atlassian, epics, stories, and themes",[263,6462,6463],{},[200,6464,6466],{"href":6125,"rel":6465},[204],"HHS, HIPAA minimum necessary requirement",{"title":170,"searchDepth":171,"depth":171,"links":6468},[6469,6470,6474,6475,6489,6490],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":6471},[6472,6473],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":6476},[6477,6478,6479,6480,6481,6482,6483,6484,6485,6486,6487,6488],{"id":6350,"depth":1001,"text":6351},{"id":6357,"depth":1001,"text":6358},{"id":6364,"depth":1001,"text":6365},{"id":6371,"depth":1001,"text":6372},{"id":6381,"depth":1001,"text":6382},{"id":6390,"depth":1001,"text":6391},{"id":6397,"depth":1001,"text":6398},{"id":6406,"depth":1001,"text":6407},{"id":6413,"depth":1001,"text":6414},{"id":6422,"depth":1001,"text":6423},{"id":6429,"depth":1001,"text":6430},{"id":6436,"depth":1001,"text":6437},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"An AI workstream is a shared workspace for one job: a brief, the tools allowed, the people and AI on it, a budget, and a finish line — not a Slack channel with a bot.","/blog/what-is-an-ai-workstream",{"title":6068,"description":6491},"blog/what-is-an-ai-workstream",[1606,216,2892,340],"AP3dNrCungMeJEPiKuPsm6GFWpucehVPCdH_-42HMxg",{"id":6498,"title":6499,"archived":164,"authors":6500,"badge":6502,"body":6503,"date":1008,"definedTerm":165,"department":165,"description":7076,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":7077,"relatedHeading":165,"seo":7078,"series":1606,"sitemap":130,"status":165,"stem":7079,"subhead":165,"tags":7080,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":7081},"content/blog/what-is-an-enterprise-agent-harness.md","What is an Enterprise Agent Harness",[6501],{"name":184,"to":135},{"label":1024},{"type":167,"value":6504,"toc":7057},[6505,6510,6521,6532,6534,6593,6609,6611,6627,6629,6646,6667,6680,6684,6691,6706,6720,6733,6745,6753,6761,6770,6776,6780,6785,6790,6799,6802,6813,6817,6828,6835,6838,6841,6851,6855,6858,6864,6875,6878,6880,6896,6906,6908,6912,6915,6919,6922,6926,6929,6933,6940,6944,6950,6952,6960,6962,6970,6972],[190,6506,1029,6507,6509],{},[193,6508,1337],{}," is the outer runtime that lets a model work on company jobs: policy it actually loads, connectors with least privilege, a loop that can stop for a named signer, and a record you can query after the people change.",[190,6511,6512,6513,6516,6517,6520],{},"It is still ",[200,6514,1652],{"href":405,"rel":6515},[204],". The workspace is not a git root. The sensor is not only pytest. The stop is not only max steps. ",[200,6518,1388],{"href":1386,"rel":6519},[204]," calls the missing piece an organisational harness: identity, ownership, economics, and learning around whatever builder harnesses (Claude Code, Cursor, LangChain graphs) teams already bought. An enterprise agent harness is that layer made operable — whether you assemble it or hire it.",[190,6522,6523,6525,6526,6528,6529,6531],{},[200,6524,287],{"href":286}," is the cut. This page is the outer object in full. An ",[200,6527,4372],{"href":4371}," is the product category that usually ships it: wiki, ",[200,6530,216],{"href":215},", teams, gates, ledger. You can have OS-class products (Nimbus, Palantir AIP, Salesforce Agentforce) and still fail the harness test if writes are a boolean on an API key. You can assemble an enterprise harness in LangGraph and pass the test. The noun is the runtime properties, not the logo.",[255,6533,258],{"id":257},[260,6535,6536,6544,6550,6556,6564,6569,6577,6585],{},[263,6537,6538,6541,6542,230],{},[193,6539,6540],{},"Outer harness."," Company workspace. See ",[200,6543,5181],{"href":286},[263,6545,6546,6549],{},[193,6547,6548],{},"Organizational harness."," Thoughtworks’ fourth layer after model, builder harness, and user harness. Governance architecture, not another markdown file.",[263,6551,6552,6555],{},[193,6553,6554],{},"Workstream."," Isolation domain: roster, connectors, budget, finish line. The job folder. Not a chat title.",[263,6557,6558,6561,6562,230],{},[193,6559,6560],{},"Write quoting."," The human sees the change in the language of the live system before sign-off. ",[200,6563,474],{"href":228},[263,6565,6566,6568],{},[193,6567,4144],{}," Missing approval, detached grant, or down interceptor means nothing mutates. Fail-open is a faster incident.",[263,6570,6571,6574,6575,230],{},[193,6572,6573],{},"Ledger / Lifecycle Graph."," AI operations events: brief, agents, policy version, signer, payload. Distinct from the warehouse’s business events. See ",[200,6576,3510],{"href":711},[263,6578,6579,6582,6583,230],{},[193,6580,6581],{},"SWE-bench / Terminal-Bench."," Inner evals. Useful for engineering vendors. Not a SOX control. ",[200,6584,504],{"href":503},[263,6586,6587,6590,6591,230],{},[193,6588,6589],{},"Forward-deployed programme."," Vendor engineers for months. AIP at scale. Capability can be real. Time-to-value is staffing. ",[200,6592,1156],{"href":1155},[190,6594,6595,6596,217,6598,217,6600,217,6602,217,6604,217,6606,6608],{},"Nimbus is one self-service enterprise harness: ",[200,6597,326],{"href":28},[200,6599,216],{"href":32},[200,6601,221],{"href":20},[200,6603,340],{"href":40},[200,6605,778],{"href":24},[200,6607,5360],{"href":45},". Score it as an example of the shape, next to AIP and Agentforce, not as the definition of the category.",[255,6610,1160],{"id":1159},[190,6612,6613,6616,6617,6619,6620,6623,6624,6626],{},[200,6614,2464],{"href":383,"rel":6615},[204]," keeps separating ",[270,6618,2609],{}," from ",[270,6621,6622],{},"scale",". Copilots and coding harnesses can produce the first. Enterprise harnesses are how writes to systems of record become the second without becoming ",[200,6625,4279],{"href":3249}," in the CRM.",[190,6628,1171],{},[260,6630,6631,6634,6637,6640,6643],{},[263,6632,6633],{},"RevOps, Legal, and Finance must share a job, not a Slack channel of screenshots",[263,6635,6636],{},"Salesforce or NetSuite can change because a model proposed it",[263,6638,6639],{},"last quarter’s pricing chat is unrecoverable",[263,6641,6642],{},"security cannot list the AI actors that may write",[263,6644,6645],{},"the vendor demo is a SWE-bench plot and a “we have MCP”",[190,6647,6648,6649,6652,6653,6656,6657,6662,6663,6666],{},"In 2024 Air Canada was held to a chatbot’s invented policy (",[200,6650,2043],{"href":2041,"rel":6651},[204],"). That is an outer-harness failure: a commitment left the building without a quote or a signer. ",[200,6654,3556],{"href":3554,"rel":6655},[204]," constrains personal data in payloads. ",[200,6658,6661],{"href":6659,"rel":6660},"https://www.sec.gov/about/laws.shtml",[204],"Sarbanes–Oxley"," constrains who may change revenue truth. ",[200,6664,601],{"href":599,"rel":6665},[204]," Article 14 wants people who can interpret, interrupt, and leave a record. A coding-agent hook that formats Python does not satisfy those.",[190,6668,6669,876,6672,6675,6676,6679],{},[200,6670,365],{"href":363,"rel":6671},[204],[200,6673,966],{"href":369,"rel":6674},[204]," assume operational controls, not a slide titled governance. ",[200,6677,4079],{"href":4077,"rel":6678},[204]," are a board checklist. They do not implement a gate. The harness does.",[255,6681,6683],{"id":6682},"what-enterprise-adds-to-a-harness","What “enterprise” adds to a harness",[190,6685,6686,6687,6690],{},"Start from ",[200,6688,6689],{"href":246},"what a harness contains"," — loop, tools, memory, permissions, feedback, orchestration — and raise the bar.",[190,6692,6693,6696,6697,6699,6700,6702,6703,6705],{},[193,6694,6695],{},"Policy that loads."," Inner harnesses inject ",[422,6698,434],{},". Enterprise harnesses inject asserted company policy for ",[270,6701,4187],{}," job, versioned. A Drive dump is not policy. A ",[200,6704,326],{"href":443}," that agents cite, with the revision on the run, is. If Legal’s discount cap lives only in a PDF nobody attached, the model will invent a number. That is not hallucination as a personality. That is a missing guide.",[190,6707,6708,6711,6712,6714,6715,6717,6718,230],{},[193,6709,6710],{},"Connectors as grants, not a toolbox."," Default read. Write is a separate plane. Least privilege is a workstream property. ",[200,6713,1842],{"href":224},". MCP may be the plug; it must inherit the grant. ",[200,6716,2640],{"href":2639},". A Finance team assigned to a GTM-only stream still must not reach ERP “because it is Finance.” ",[200,6719,484],{"href":220},[190,6721,6722,6725,6726,6728,6729,6732],{},[193,6723,6724],{},"A hiring object for operators."," Not a folder of personal GPTs. A mandate, required systems, approval triggers — ",[200,6727,221],{"href":493}," as a roster. ",[200,6730,6731],{"href":2132},"How to evaluate agent teams vs single agents",". Nimbus ships functional teams on that roster; AIP and Agentforce have their own packaging. The test is: can an operator inspect the mandate and the required systems before assign.",[190,6734,6735,463,6738,6740,6741,6744],{},[193,6736,6737],{},"Human wait as a step.",[200,6739,1897],{"href":1896}," is not a kill switch in a dashboard. It is quoted payload, named role, fail-closed adapter. ",[200,6742,6743],{"href":241},"HITL approval architecture",". Soft / Hard / Critical matched to blast radius. A six-month zero-reject rate on CRM writes is a finding.",[190,6746,6747,6750,6751,230],{},[193,6748,6749],{},"A ledger of AI operations."," Who briefed, which team, which wiki revision, who signed, what executed. Exportable without the vendor in the room. The warehouse is not this ledger. ",[200,6752,1967],{"href":1966},[190,6754,6755,6758,6759,230],{},[193,6756,6757],{},"Evals that match the job."," Did the executed write match the signed quote. Can you replay. Inner leaderboards are a vendor quality signal for coding. They are not the enterprise eval. See ",[200,6760,2590],{"href":503},[190,6762,6763,6766,6767,6769],{},[193,6764,6765],{},"Economics of the loop."," Routing compact extract vs frontier judgement. Spend quotes. Seat pricing that includes unlimited flagship is an unengineered cost harness. ",[200,6768,1262],{"href":1261},". Nimbus meters NTUs; copilots meter seats. Different jobs.",[190,6771,6772,6775],{},[193,6773,6774],{},"Self-service vs programme."," If every new connector is a six-month SOW, you have bought a deployment, not a harness operators can tighten. That can still be the right buy for Ontology-scale complexity. It is the wrong buy for a standard Salesforce write this quarter.",[255,6777,6779],{"id":6778},"what-it-is-not","What it is not",[190,6781,6782,6783,230],{},"A coding harness with SSO. ",[200,6784,287],{"href":286},[190,6786,6787,6788,230],{},"A copilot with an admin console. ",[200,6789,2751],{"href":2338},[190,6791,6792,6793,6796,6797,230],{},"A framework. LangGraph can ",[270,6794,6795],{},"host"," an enterprise harness if you build grants, quotes, and a ledger. Out of the box it hosts a graph. ",[200,6798,594],{"href":593},[190,6800,6801],{},"“We integrate with Salesforce.” Integration is a slide. A scoped connector plus a blocked unsigned write is a harness.",[190,6803,6804,6805,876,6808,6812],{},"SWE-bench-first marketing. ",[200,6806,765],{"href":411,"rel":6807},[204],[200,6809,6811],{"href":202,"rel":6810},[204],"LangChain"," are writing about coding and general agents. Steal the discipline (stops, artifacts, sensors). Do not steal the benchmark as your control framework.",[255,6814,6816],{"id":6815},"thoughtworks-organisational-harness-in-operator-language","Thoughtworks’ organisational harness, in operator language",[190,6818,3395,6819,6823,6824,6827],{},[200,6820,6822],{"href":1386,"rel":6821},[204],"Thoughtworks OS essay"," (10 July 2026) argues that most AI programmes fail because the organisation never built the operating system around the model: accountability, ownership, measurement, learning. They name four layers. An enterprise agent harness, as this article uses the term, is layers 3–4 made runnable for ",[270,6825,6826],{},"company jobs"," — not only for coding-agent users.",[190,6829,6830,6831,6834],{},"Delegation failures are the tell. The model was fine. The platform ran. Practitioner guides existed. The agent did what it was ",[270,6832,6833],{},"allowed"," to do. The company still took harm. Layer 4 questions: who approved that autonomy, who owns the policy, what was the escalation, how do we prevent the same miss on another team. Layers 1–3 cannot answer those. A chat product cannot either.",[190,6836,6837],{},"Thoughtworks’ control matrix is worth stealing even if you never hire them. Use deterministic controls where the boundary is knowable: allowed actions, residency, spend ceilings, blast-radius limits. Use probabilistic controls only where judgement is required. Pair every guide with a sensor. Temporal constraints — consistency across a multi-step workflow, not a single dropdown — are the ones they say teams miss most. A scheduling agent that is locally plausible on each step and globally inconsistent is not a “hallucination.” It is a missing temporal sensor.",[190,6839,6840],{},"Their public examples (Parloa’s repo-resident rules/skills/commands; Morgan Stanley’s tiered autonomy on CVE triage) are coding-adjacent. Translate them: discount policy as a versioned wiki skill; “what delegation tier does this CRM write require?” instead of “do we trust the agent.” Nimbus’s Soft / Hard / Critical is that tiering in product form. AIP will have a different packaging. The architectural claim is the same.",[190,6842,6843,876,6846,6850],{},[200,6844,211],{"href":209,"rel":6845},[204],[200,6847,6849],{"href":915,"rel":6848},[204],"Wikipedia"," describe the runtime. Thoughtworks describe why a runtime without ownership still fails at scale. You need both descriptions when you buy.",[255,6852,6854],{"id":6853},"what-good-looks-like-on-a-live-job","What “good” looks like on a live job",[190,6856,6857],{},"A renewal write: workstream isolation; Salesforce attached read-only until write is enabled; wiki revision with the cap cited on the run; team cannot start if Legal’s connector requirement is missing; model proposes a quote; Hard gate; reject leaves Stage unchanged; export shows signer without a vendor screen-share. That is an enterprise harness. A demo that only answers “what should we do about Acme” is a copilot with a logo.",[190,6859,6860,6861,230],{},"Spend an hour asking where each Thoughtworks layer lives in the vendor’s product. If layer 4 is “our professional services team,” you are buying a programme. That can be the right buy. Name it. ",[200,6862,6863],{"href":1155},"Self-service vs FDE",[190,6865,6866,6867,6870,6871,6874],{},"Operators already know the human version of this harness. Maker-checker on journals. Segregation of duties on payments. Change-advisory on production. The enterprise agent harness is those instincts encoded so a model cannot talk through them. ",[200,6868,6661],{"href":6659,"rel":6869},[204]," did not wait for LLMs; it waited for a named signer. ",[200,6872,3556],{"href":3554,"rel":6873},[204]," did not wait for MCP; it waits for purpose limitation on the payload. If your AI programme cannot point to the interceptor that enforces those, you have a chatbot with a risk register.",[190,6876,6877],{},"What failure looks like in the first ninety days: every department clones a GPT with the same Salesforce key; Legal’s cap lives in a slide; the only eval is “the demo was impressive”; coding-agent MCP is pointed at production “just for a spike”; the ledger is Slack. What success looks like: one roster of teams, workstream isolation, default read, a Hard refuse on the first PoV, a wiki revision on the graph, inner harnesses still compiling in repos. Nimbus is built to make the success path a product week rather than a services year. Verify that claim with the refuse. AIP may be the right path when Ontology-scale complexity is real — then the harness is a programme, and you should staff it as one.",[255,6879,793],{"id":792},[190,6881,6882,6883,6885,6886,6889,6890,6892,6893,6895],{},"Nimbus’s outer loop is: brief a ",[200,6884,1083],{"href":32}," → assign a ",[200,6887,6888],{"href":20},"team"," whose connector contract is satisfied → retrieve under scope → draft on the canvas (Conflux) → quote writes → ",[200,6891,340],{"href":40}," pause → execute the signed payload → commit to the ",[200,6894,23],{"href":24},". Perception orients; it does not silently write. Routing picks model class per step.",[190,6897,6898,6899,876,6903,6905],{},"That mapping is how we productised harness engineering for operators. It is not a claim that AIP or Agentforce are “not harnesses.” They are different time and scope. ",[200,6900,6902],{"href":6901},"how-to-evaluate-an-enterprise-ai-operating-system","How to evaluate an enterprise AI OS",[200,6904,1485],{"href":251}," are the two sheets; use both.",[255,6907,807],{"id":806},[809,6909,6911],{"id":6910},"do-we-need-this-if-we-already-have-claude-code","Do we need this if we already have Claude Code?",[190,6913,6914],{},"You need it for jobs whose workspace is the company. Keep Claude Code for repos. Do not share production SoR write tokens into the inner harness.",[809,6916,6918],{"id":6917},"is-palantir-aip-an-enterprise-harness","Is Palantir AIP an enterprise harness?",[190,6920,6921],{},"It can be, as a programme-shaped outer runtime. Ask deployment time, who sets a gate without vendor engineers, and whether the ledger is yours. Category yes; evaluation still required.",[809,6923,6925],{"id":6924},"is-agentforce-enough","Is Agentforce enough?",[190,6927,6928],{},"If the job is CRM-anchored and stays there, maybe. Cross-system jobs with Legal on the canvas usually need a harness that is not only Salesforce. Clear scopes; avoid two writers.",[809,6930,6932],{"id":6931},"can-we-build-this-on-langchain","Can we build this on LangChain?",[190,6934,6935,6936,6939],{},"Yes, with time. You will rebuild grants, quoting, roster, and replay. ",[200,6937,6938],{"href":848},"Build vs buy",". Frameworks assemble loops; operators still need a loop they can hire.",[809,6941,6943],{"id":6942},"whats-the-first-proof","What’s the first proof?",[190,6945,6946,6947,6949],{},"A real cross-department write: operator attaches OAuth; unsigned payload blocked; reject leaves SoR unchanged; export shows signer. ",[200,6948,1279],{"href":1278},". A chat demo is not this.",[809,6951,853],{"id":852},[190,6953,6954,858,6956,858,6958,230],{},[200,6955,1659],{"href":251},[200,6957,195],{"href":2140},[200,6959,861],{"href":358},[255,6961,869],{"id":868},[190,6963,6964,876,6966,230],{},[200,6965,3286],{"href":228},[200,6967,6969],{"href":6968},"rfp-questions-for-enterprise-ai-agents","RFP questions for enterprise AI agents",[255,6971,883],{"id":882},[260,6973,6974,6979,6984,6989,6994,7001,7006,7011,7016,7021,7026,7031,7036,7042,7047,7052],{},[263,6975,6976],{},[200,6977,897],{"href":405,"rel":6978},[204],[263,6980,6981],{},[200,6982,910],{"href":209,"rel":6983},[204],[263,6985,6986],{},[200,6987,917],{"href":915,"rel":6988},[204],[263,6990,6991],{},[200,6992,1545],{"href":1386,"rel":6993},[204],[263,6995,6996],{},[200,6997,7000],{"href":6998,"rel":6999},"https://www.thoughtworks.com/insights/podcasts/technology-podcasts/scaling-the-enterprise-harness--how-to-achieve-ai-agent-controll",[204],"Thoughtworks, Scaling the enterprise harness",[263,7002,7003],{},[200,7004,923],{"href":411,"rel":7005},[204],[263,7007,7008],{},[200,7009,983],{"href":383,"rel":7010},[204],[263,7012,7013],{},[200,7014,365],{"href":363,"rel":7015},[204],[263,7017,7018],{},[200,7019,966],{"href":369,"rel":7020},[204],[263,7022,7023],{},[200,7024,4079],{"href":4077,"rel":7025},[204],[263,7027,7028],{},[200,7029,601],{"href":599,"rel":7030},[204],[263,7032,7033],{},[200,7034,3556],{"href":3554,"rel":7035},[204],[263,7037,7038],{},[200,7039,7041],{"href":6659,"rel":7040},[204],"SEC, Sarbanes–Oxley",[263,7043,7044],{},[200,7045,2228],{"href":2041,"rel":7046},[204],[263,7048,7049],{},[200,7050,989],{"href":296,"rel":7051},[204],[263,7053,7054],{},[200,7055,1633],{"href":1631,"rel":7056},[204],{"title":170,"searchDepth":171,"depth":171,"links":7058},[7059,7060,7061,7062,7063,7064,7065,7066,7074,7075],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160},{"id":6682,"depth":171,"text":6683},{"id":6778,"depth":171,"text":6779},{"id":6815,"depth":171,"text":6816},{"id":6853,"depth":171,"text":6854},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":7067},[7068,7069,7070,7071,7072,7073],{"id":6910,"depth":1001,"text":6911},{"id":6917,"depth":1001,"text":6918},{"id":6924,"depth":1001,"text":6925},{"id":6931,"depth":1001,"text":6932},{"id":6942,"depth":1001,"text":6943},{"id":852,"depth":1001,"text":853},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"An enterprise agent harness is the outer runtime for operators: wiki, scoped connectors, agent teams, write gates, and a ledger — not a SWE-bench score and not a chat with every production login.","/blog/what-is-an-enterprise-agent-harness",{"title":6499,"description":7076},"blog/what-is-an-enterprise-agent-harness",[1606,1015,2892,340],"VCI6uaS6V4vGana_omswQGrFpUW2iH4qg59Oz11-AgA",{"id":7083,"title":7084,"archived":164,"authors":7085,"badge":7087,"body":7088,"date":3101,"definedTerm":165,"department":165,"description":7092,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":7533,"relatedHeading":165,"seo":7534,"series":1606,"sitemap":130,"status":165,"stem":7535,"subhead":165,"tags":7536,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":7538},"content/blog/what-is-an-enterprise-ai-operating-system.md","What is an Enterprise AI Operating System",[7086],{"name":184,"to":135},{"label":1024},{"type":167,"value":7089,"toc":7509},[7090,7093,7096,7099,7105,7108,7110,7171,7174,7206,7217,7222,7224,7230,7233,7236,7239,7242,7244,7249,7260,7265,7270,7275,7277,7283,7297,7303,7309,7315,7318,7327,7329,7332,7370,7385,7387,7391,7394,7398,7406,7410,7413,7417,7420,7424,7427,7431,7434,7438,7441,7445,7450,7454,7457,7461,7466,7470,7473,7477,7486,7488,7494,7496],[190,7091,7092],{},"An enterprise AI operating system is the layer between the AI model and how departments actually work — like Windows sits between the chip and your apps.",[190,7094,7095],{},"If your question is “which model should we buy,” you are shopping for a chip. If your question is “how do revenue, legal, and finance run the same loop without a personal-account workaround,” you are shopping for an OS.",[190,7097,7098],{},"It is not a chatbot with company login. It is not a model API with a prompt library.",[190,7100,7101,7104],{},[200,7102,5214],{"href":383,"rel":7103},[204]," found that 88% of companies use AI in at least one function — and that a majority are still piloting. About one in three report that they are scaling. Buying another model does not close that gap. The missing layer is how work actually runs.",[190,7106,7107],{},"The OS metaphor is useful if you keep it honest. An operating system does not replace your spreadsheet or your CRM. It gives applications isolation, permissions, input and output, and a place to keep state after the window closes. An enterprise AI OS does the same for work that uses models: isolation of jobs, rights over tools and data, reads and writes to live systems, budgets, and a record that survives the session.",[255,7109,258],{"id":257},[260,7111,7112,7118,7124,7132,7138,7148,7155,7163],{},[263,7113,7114,7117],{},[193,7115,7116],{},"Copilot."," A high-quality assistant for a person. Admin controls, company login. Not, by itself, how several departments finish one job under a named signer. At work, this is “help me draft.” It is not “release this CRM change.”",[263,7119,7120,7123],{},[193,7121,7122],{},"Operating system (in this sense)."," Process isolation, permissions, input/output to live systems, budgeting, and durable state — the jobs a kernel does for apps.",[263,7125,7126,7128,7129,7131],{},[193,7127,3646],{}," CRM, ERP, HR — still authoritative. The OS is the system of ",[270,7130,3865],{},", not a second CRM.",[263,7133,7134,7137],{},[193,7135,7136],{},"Forward-deployed engineer."," A vendor consultant who sits with you for months. Some programmes need that. Many companies need governed work this quarter without it.",[263,7139,7140,463,7143,7147],{},[193,7141,7142],{},"NIST AI RMF.",[200,7144,7146],{"href":363,"rel":7145},[204],"Govern, Map, Measure, Manage"," — public-sector language for the same kernel idea: identity, tool rights, and a record attached to real actions.",[263,7149,7150,7152,7153,230],{},[193,7151,6554],{}," The process-isolation unit: one job, one scope, one finish line. See ",[200,7154,4992],{"href":215},[263,7156,7157,7160,7161,230],{},[193,7158,7159],{},"Fail-closed writes."," Missing named signer means nothing happens. See ",[200,7162,3286],{"href":228},[263,7164,7165,7168,7169,230],{},[193,7166,7167],{},"NTU."," A normalised work credit so spend can be quoted and capped. See ",[200,7170,3749],{"href":3303},[190,7172,7173],{},"Five jobs cluster around the term:",[547,7175,7176,7182,7188,7194,7200],{},[263,7177,7178,7181],{},[193,7179,7180],{},"Process isolation."," Go-to-market does not silently inherit finance’s ERP login.",[263,7183,7184,7187],{},[193,7185,7186],{},"Resource management."," Inference and tool calls are budgeted.",[263,7189,7190,7193],{},[193,7191,7192],{},"I/O control."," Reads and writes to CRM and ERP are first-class — not “chat that sometimes calls an API.”",[263,7195,7196,7199],{},[193,7197,7198],{},"Permissioning."," Identity and context decide what an agent can see and do. A signed PDF is not enforcement.",[263,7201,7202,7205],{},[193,7203,7204],{},"Durable state."," Outcomes, approvals, and rationale survive the session.",[190,7207,7208,7209,876,7212,7216],{},"Copilots generally fail the last three. ",[200,7210,5408],{"href":7211},"/blog/nimbus-vs-chatgpt-enterprise",[200,7213,7215],{"href":7214},"/blog/nimbus-vs-claude","Claude for Work"," are excellent assistants. They are not this job.",[190,7218,7219,7221],{},[200,7220,5140],{"href":757}," is also not this job. A common plug for tools is USB. USB did not create Windows.",[255,7223,1160],{"id":1159},[190,7225,7226,7227,7229],{},"Operators do not “open the OS” the way they open a model playground. They open ",[193,7228,3865],{},": a brief, a scoped live system, a review, a release.",[190,7231,7232],{},"It affects you if AI is starting to touch revenue, financial close, customer records, or regulated processes. Chat history does not answer “who approved this, against which policy?”",[190,7234,7235],{},"The OS also matters if you refuse a six-to-twelve-month vendor-engineer programme as the only path to production.",[190,7237,7238],{},"The anti-pattern is using an OS as a better chatbot: one user, one thread, no write path, no memory beyond the conversation. If nobody except the original operator can reconstruct what happened, you have a log, not an operating system.",[190,7240,7241],{},"Personal copilots optimise for “the model always answers.” An OS optimises for “the company only acts when the gate says so.” A spend cap or a missing approval is a successful outcome.",[809,7243,3290],{"id":3289},[190,7245,7246,7248],{},[193,7247,3295],{}," The OS is how close and forecast jobs get a budget, a read-only ERP connector, a wiki checklist, and a named signer — without a second ledger. Finance should still own NetSuite. The OS should point at it.",[190,7250,7251,7253,7254,7257,7258,230],{},[193,7252,3310],{}," Reconstructable authorisation, purpose-limited scope, and a place that is not a personal chat vendor. Legal should evaluate whether unapproved writes are ",[270,7255,7256],{},"impossible",", not whether a policy PDF exists. See ",[200,7259,3336],{"href":3335},[190,7261,7262,7264],{},[193,7263,3319],{}," Isolation and durable state are ops problems. Ops should ask whether a paused run is a first-class object, whether connectors default to read-only, and whether Perception (or equivalent) can answer “why did this change?” without a data team reconstructing Slack.",[190,7266,7267,7269],{},[193,7268,3325],{}," Cross-department loops — legal on a renewal, finance on a discount — need a shared job, not a shared inbox. GTM should not have to choose between a copilot that cannot write safely and a spreadsheet export to a consumer model.",[190,7271,7272,7274],{},[193,7273,3331],{}," Identity, least privilege, fail-closed I/O, and not turning the OS into a second store of the whole company. Security also cares that self-service configuration does not mean tenant-wide write keys.",[809,7276,3340],{"id":3339},[190,7278,7279,7282],{},[193,7280,7281],{},"“ChatGPT with integrations.”"," Plugins without scoped work, approval architecture, and durable decision records are plugins. A copilot with automation actions can move data. It cannot, by itself, make unapproved writes impossible.",[190,7284,7285,7288,7289,7292,7293,7296],{},[193,7286,7287],{},"MLOps as a substitute."," MLOps governs ",[270,7290,7291],{},"model production",". An enterprise AI OS governs ",[270,7294,7295],{},"operational work that uses models",". They stack.",[190,7298,7299,7302],{},[193,7300,7301],{},"Replacing the CRM."," Salesforce, NetSuite, Workday, and the warehouse remain authoritative. Duplicating them is a second system of record.",[190,7304,7305,7308],{},[193,7306,7307],{},"OS as chatbot."," One user, one thread, no write path, no memory. That is a copilot with extra vocabulary.",[190,7310,7311,7314],{},[193,7312,7313],{},"Forward-deployed as the only path."," Some warehouses need specialists. Most operators need to attach a connector and set a named signer in the UI.",[190,7316,7317],{},"Good looks like: workstreams, wiki, read-only-default connectors, agent teams, Lifecycle Graph, model routing, NTU quotes, fail-closed writes, self-service configuration. Failure looks like another model contract plus a six-month SOW.",[190,7319,7320,7321,7323,7324,230],{},"For the copilot-versus-OS choice, see ",[200,7322,2339],{"href":2338},". For vendor scoring, ",[200,7325,7326],{"href":6901},"How to evaluate an enterprise AI operating system",[255,7328,793],{"id":792},[190,7330,7331],{},"Nimbus is a self-service enterprise AI OS. Operators configure it in the product.",[260,7333,7334,7341,7346,7352,7357,7365],{},[263,7335,7336,7340],{},[193,7337,7338],{},[200,7339,31],{"href":215}," isolate process.",[263,7342,7343,7345],{},[193,7344,3414],{}," holds asserted policy — approved playbooks, not a dump of PDFs a search might find.",[263,7347,7348,7351],{},[193,7349,7350],{},"Connectors"," attach live systems. Default is read-only. Write-back is opt-in and gated.",[263,7353,7354,7356],{},[193,7355,685],{}," are department-shaped.",[263,7358,7359,7361,7362,7364],{},[193,7360,23],{}," stores the causal record. ",[193,7363,3858],{}," queries it in ordinary language.",[263,7366,7367,7369],{},[193,7368,1262],{}," puts routine extract on cheaper models.",[190,7371,3411,7372,876,7374,7376,7377,217,7379,217,7381,217,7383,230],{},[200,7373,11],{"href":12},[200,7375,7326],{"href":6901},". Product surfaces: ",[200,7378,31],{"href":32},[200,7380,39],{"href":40},[200,7382,23],{"href":24},[200,7384,3858],{"href":36},[255,7386,807],{"id":806},[809,7388,7390],{"id":7389},"is-an-enterprise-ai-os-just-chatgpt-with-integrations","Is an enterprise AI OS just “ChatGPT with integrations”?",[190,7392,7393],{},"No. Integrations without scoped work, approval architecture, and durable decision records are plugins. A copilot with automation actions can move data. It cannot, by itself, make unapproved writes impossible.",[809,7395,7397],{"id":7396},"how-is-this-different-from-mlops","How is this different from MLOps?",[190,7399,7400,7401,7292,7403,7405],{},"MLOps governs ",[270,7402,7291],{},[270,7404,7295],{},". They stack. They do not substitute.",[809,7407,7409],{"id":7408},"do-we-still-need-a-crm-if-we-buy-an-os","Do we still need a CRM if we buy an OS?",[190,7411,7412],{},"Yes. Salesforce, NetSuite, Workday, and the warehouse remain authoritative.",[809,7414,7416],{"id":7415},"does-every-company-need-an-os","Does every company need an OS?",[190,7418,7419],{},"If the job is personal drafting with no writes to live systems, a governed copilot may be enough. The OS becomes the right abstraction when work crosses departments, when writes are material, and when you must reconstruct decisions.",[809,7421,7423],{"id":7422},"is-this-the-same-as-an-integration-platform-ipaas","Is this the same as an integration platform (iPaaS)?",[190,7425,7426],{},"No. iPaaS moves data on schedules and triggers. An AI OS runs language-using jobs with scope, spend, and a human gate. You may still need iPaaS. It does not quote a named signer on a CRM payload.",[809,7428,7430],{"id":7429},"does-operating-system-mean-we-install-software-on-laptops","Does “operating system” mean we install software on laptops?",[190,7432,7433],{},"No. It is a layer for work, not a desktop kernel. The metaphor is isolation, permissions, I/O, and state.",[809,7435,7437],{"id":7436},"can-we-build-this-ourselves-on-a-model-api","Can we build this ourselves on a model API?",[190,7439,7440],{},"You can assemble pieces. You will still need isolation, connectors, gates, spend, and a graph. Most “we built a GPT” programmes stall at the copilot layer. McKinsey’s split between using AI and scaling it is that stall in survey form.",[809,7442,7444],{"id":7443},"where-do-agent-teams-fit","Where do agent teams fit?",[190,7446,7447,7448,230],{},"They are the department-shaped specialists the OS schedules onto workstreams. They are not the OS. See ",[200,7449,494],{"href":493},[809,7451,7453],{"id":7452},"how-does-nists-ai-rmf-map","How does NIST’s AI RMF map?",[190,7455,7456],{},"Govern (owners, policy), Map (inventory of jobs and systems), Measure (evidence, spend, rejects), Manage (fail-closed writes, incident path). A product can make those cheaper. A framework PDF cannot enforce them.",[809,7458,7460],{"id":7459},"what-is-perception-in-this-picture","What is Perception in this picture?",[190,7462,7463,7464,230],{},"Ordinary-language questions over the company’s graph, wiki, and scoped systems — with the next step being a workstream, not another search. See ",[200,7465,3858],{"href":36},[809,7467,7469],{"id":7468},"do-we-need-a-forward-deployed-engineer-to-go-live","Do we need a forward-deployed engineer to go live?",[190,7471,7472],{},"Not as the default path. If operators cannot attach a read-only connector and set a named signer in the UI, you do not have a self-service OS. Specialists belong on genuine exceptions, such as a warehouse with no OAuth.",[809,7474,7476],{"id":7475},"is-search-rag-an-os","Is search (RAG) an OS?",[190,7478,7479,7480,876,7482,230],{},"No. Lookup-then-answer is infrastructure. It does not isolate jobs or gate writes. See ",[200,7481,3468],{"href":438},[200,7483,7485],{"href":7484},"/blog/nimbus-vs-glean","Nimbus vs Glean",[255,7487,869],{"id":868},[190,7489,7490,876,7492,230],{},[200,7491,4992],{"href":215},[200,7493,3336],{"href":3335},[255,7495,883],{"id":882},[260,7497,7498,7504],{},[263,7499,7500],{},[200,7501,7503],{"href":383,"rel":7502},[204],"McKinsey, The state of AI (2025)",[263,7505,7506],{},[200,7507,4100],{"href":363,"rel":7508},[204],{"title":170,"searchDepth":171,"depth":171,"links":7510},[7511,7512,7516,7517,7531,7532],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":7513},[7514,7515],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":7518},[7519,7520,7521,7522,7523,7524,7525,7526,7527,7528,7529,7530],{"id":7389,"depth":1001,"text":7390},{"id":7396,"depth":1001,"text":7397},{"id":7408,"depth":1001,"text":7409},{"id":7415,"depth":1001,"text":7416},{"id":7422,"depth":1001,"text":7423},{"id":7429,"depth":1001,"text":7430},{"id":7436,"depth":1001,"text":7437},{"id":7443,"depth":1001,"text":7444},{"id":7452,"depth":1001,"text":7453},{"id":7459,"depth":1001,"text":7460},{"id":7468,"depth":1001,"text":7469},{"id":7475,"depth":1001,"text":7476},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"/blog/what-is-an-enterprise-ai-operating-system",{"title":7084,"description":7092},"blog/what-is-an-enterprise-ai-operating-system",[1606,2892,7537,340],"operating-system","3g_fqM0BGQmCwQRtpdTdXfYJxVm8Gx12m5O4-20qQxw",{"id":7540,"title":7541,"archived":164,"authors":7542,"badge":7544,"body":7545,"date":3101,"definedTerm":165,"department":165,"description":7962,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":7963,"relatedHeading":165,"seo":7964,"series":1606,"sitemap":130,"status":165,"stem":7965,"subhead":165,"tags":7966,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":7969},"content/blog/what-is-causal-ai-for-operations.md","What is Causal AI for Operations",[7543],{"name":184,"to":135},{"label":1024},{"type":167,"value":7546,"toc":7938},[7547,7553,7568,7575,7578,7581,7583,7634,7637,7651,7653,7656,7659,7683,7690,7704,7713,7715,7724,7729,7734,7739,7744,7746,7752,7758,7764,7770,7776,7779,7782,7794,7796,7804,7807,7816,7818,7822,7825,7829,7832,7836,7839,7843,7846,7850,7853,7857,7863,7867,7870,7874,7877,7881,7884,7888,7891,7895,7900,7904,7907,7909,7917,7919],[190,7548,7549,7550],{},"“Causal AI” is a phrase people type into ChatGPT when they mean: ",[193,7551,7552],{},"can we tell why something happened, or are we guessing?",[190,7554,7555,7556,7561,7562,7567],{},"In statistics, causality is a serious science. Did the discount cause the win, or did seasonality? That needs experiments and careful assumptions, not a model that says “because.” The ",[200,7557,7560],{"href":7558,"rel":7559},"https://plato.stanford.edu/entries/causal-models/",[204],"Stanford Encyclopedia of Philosophy’s entry on causal models"," is a fair orientation to that science. Judea Pearl’s overview, ",[200,7563,7566],{"href":7564,"rel":7565},"https://ftp.cs.ucla.edu/pub/stat_ser/r350.pdf",[204],"Causal inference in statistics",", is the technical companion: identification is a design problem, not a paragraph problem.",[190,7569,7570,7571,7574],{},"In operations, the question is more everyday and more urgent: ",[193,7572,7573],{},"why did this field, journal, or customer message change?"," If you cannot replay the brief, the sources, the named approval, and the live-system result, you have a dashboard, not a cause.",[190,7576,7577],{},"This article is about that second meaning. Nimbus does not claim to estimate market lift from a chatbot. It does claim you should be able to reconstruct the intervention.",[190,7579,7580],{},"Mixing the two meanings is how board decks get written. Two charts rose together; the model wrote a fluent “because”; finance cannot sample the journal. You can have excellent statistics in a notebook and still be unable to say who approved last night’s ERP write. You can have an excellent operations record and still be wrong about the market. Do not let one pretend to be the other.",[255,7582,258],{"id":257},[260,7584,7585,7591,7597,7603,7611,7620,7628],{},[263,7586,7587,7590],{},[193,7588,7589],{},"Cause vs correlation."," Two lines rising together is not proof that one caused the other. At work, AI usage and pipeline in the same quarter is a coincidence until you show the steps.",[263,7592,7593,7596],{},[193,7594,7595],{},"Intervention."," Something you actually did — an approval, a write, a refusal. At work, a fail-closed gate that blocked a write is an intervention with a known counterfactual: nothing would have changed.",[263,7598,7599,7602],{},[193,7600,7601],{},"Identification."," The statistics problem of isolating a true effect. Different from a work record. At work, this is “did signed next-step updates cause wins?” — a question for a designed comparison, not for Perception.",[263,7604,7605,7608,7609,230],{},[193,7606,7607],{},"Lifecycle graph."," The company’s chain of AI work: what was asked, who signed, what changed. See ",[200,7610,3510],{"href":711},[263,7612,7613,7615,7616,7619],{},[193,7614,3634],{}," Who, what, when, derived from what. ",[200,7617,3640],{"href":3638,"rel":7618},[204]," is the open vocabulary for that idea.",[263,7621,7622,7625,7626,230],{},[193,7623,7624],{},"Confounder."," In science, a hidden third factor. In operations, the hidden factor is often “a human pasted a consumer-model answer into CRM.” See ",[200,7627,3250],{"href":3249},[263,7629,7630,7633],{},[193,7631,7632],{},"Rationale."," The model’s English explanation. Often written after the fact. Not a recorded structure.",[190,7635,7636],{},"Keep two layers apart:",[547,7638,7639,7645],{},[263,7640,7641,7644],{},[193,7642,7643],{},"Causal science."," Did the discount cause the win? Needs a design, not a fluent paragraph.",[263,7646,7647,7650],{},[193,7648,7649],{},"Causal operations."," Brief → sources → proposal → approval → write → system response. Needs a record.",[255,7652,1160],{"id":1159},[190,7654,7655],{},"Boards get briefed on “AI caused the pipeline jump” because both charts went up. Finance cannot sample a journal that only exists as a chat. Legal cannot explain a CRM exception that lived in someone’s personal account.",[190,7657,7658],{},"Causal operations affects you if you:",[260,7660,7661,7667,7677],{},[263,7662,7663,7666],{},[193,7664,7665],{},"Have to explain a change."," “Who caused this field to move, against which rule?”",[263,7668,7669,7676],{},[193,7670,7671,7672,7675],{},"Need to know what ",[270,7673,7674],{},"would"," have happened without approval."," In a real gate, the answer is nothing.",[263,7678,7679,7682],{},[193,7680,7681],{},"Are tempted to file a model’s “because” as truth."," Natural-language rationales are often written after the fact.",[190,7684,7685,7686,7689],{},"You do ",[193,7687,7688],{},"not"," need a data-science sprint to ask:",[260,7691,7692,7695,7698,7701],{},[263,7693,7694],{},"Why was this record changed?",[263,7696,7697],{},"Which policy version caused this refusal?",[263,7699,7700],{},"Did a spend cap stop the run?",[263,7702,7703],{},"Did analysis change the CRM, or only produce a draft?",[190,7705,7706,7707,7709,7710,7712],{},"You ",[193,7708,1310],{}," need a statistician if you want to know whether signed next-step updates ",[270,7711,3730],{}," wins. The work record can attach “this account was treated.” Estimation is extra.",[809,7714,3290],{"id":3289},[190,7716,7717,7719,7720,7723],{},[193,7718,3295],{}," Sampling a journal requires the chain, not a story. Spend caps that fire are causes of ",[270,7721,7722],{},"inaction",", which close packs also need to explain. Do not let “AI lift” into a board pack without either an experiment or an honest “we do not know.”",[190,7725,7726,7728],{},[193,7727,3310],{}," Discovery and customer commitments need the payload the signer saw. A model rationale is advocacy, not evidence. Legal should also stop people treating a chatbot explanation as the company’s official why.",[190,7730,7731,7733],{},[193,7732,3319],{}," This is the native question: why did this change, who signed, what was refused. Ops should keep BI for canonical metrics and the graph for AI-work lineage. Dumping bookings into the graph as a fake causal model is a mess.",[190,7735,7736,7738],{},[193,7737,3325],{}," Forecast meetings will try to credit the copilot. GTM needs reconstructable interventions (which opportunities were touched, by which job) and should refuse market-lift claims without a design. Correlation slides train everyone to stop asking.",[190,7740,7741,7743],{},[193,7742,3331],{}," Reconstructability is also incident response: which connector was read-only, which tool was called, whether a jailbreak requested a write that the gate refused. The refusal is a causal fact worth keeping.",[809,7745,3340],{"id":3339},[190,7747,7748,7751],{},[193,7749,7750],{},"The model’s “because.”"," Fluency is not identification and not a recorded structure.",[190,7753,7754,7757],{},[193,7755,7756],{},"Two rising lines."," Correlation. File it as a hypothesis.",[190,7759,7760,7763],{},[193,7761,7762],{},"Using the same AI that proposed the treatment to declare success."," That is marking your own homework.",[190,7765,7766,7769],{},[193,7767,7768],{},"Skipping the operations layer to buy a science platform."," Without reconstructable interventions, the science team inherits Slack folklore.",[190,7771,7772,7775],{},[193,7773,7774],{},"Using the graph as a BI tool."," Canonical commercial metrics stay in the warehouse. The graph answers mixed policy / approval / live-system questions.",[190,7777,7778],{},"Good looks like: a chain you can query, read-only analysis recorded as non-writes, named signers, spend stops as events, and a bright line before anyone claims lift. Failure looks like a dashboard, a chatbot paragraph, and a forecast that nobody can unwind.",[190,7780,7781],{},"Pearl’s identification problem and an operations reconstruction problem share a word and almost nothing else. Keep the word, split the buying decision. You can staff science later. You cannot reconstruct a write you never recorded.",[190,7783,7784,7785,7787,7788,7790,7791,7793],{},"Adjacent: ",[200,7786,3186],{"href":711}," is the product shape of the operations layer. ",[200,7789,3686],{"href":3270}," is what remains after people leave. ",[200,7792,474],{"href":228}," makes “nothing happened” a possible true answer.",[255,7795,793],{"id":792},[190,7797,7798,7799,7801,7802,230],{},"Nimbus implements causal operations as the ",[193,7800,23],{}," plus ",[193,7803,3858],{},[190,7805,7806],{},"Work runs are chains you can query in ordinary language. The company wiki is often the parent of a refusal (“this playbook caused the flag”). Connectors default to read-only, which is itself a causal fact: analysis did not change the CRM. Spend quotes make cost an explicit stop, not an ambient cloud bill. Fail-closed writes mean a missing named signer is a recorded non-event with a known counterfactual.",[190,7808,3411,7809,876,7811,7813,7814,230],{},[200,7810,23],{"href":24},[200,7812,3858],{"href":36},". Writes that cannot happen without a signer are ",[200,7815,1079],{"href":228},[255,7817,807],{"id":806},[809,7819,7821],{"id":7820},"if-the-model-explains-why-is-that-causal-ai","If the model explains “why,” is that causal AI?",[190,7823,7824],{},"No. A fluent paragraph is not a recorded structure, and it is not a statistical identification.",[809,7826,7828],{"id":7827},"do-we-need-advanced-causal-statistics-to-buy-an-operating-layer","Do we need advanced causal statistics to buy an operating layer?",[190,7830,7831],{},"No. You need reconstructable interventions. If you later staff a science team, they will thank you for not storing decisions as Slack folklore.",[809,7833,7835],{"id":7834},"can-the-graph-estimate-lift","Can the graph estimate lift?",[190,7837,7838],{},"Only if you design an experiment or a credible comparison and collect the right outcomes. Beware of using the same AI that proposed the treatment to declare the treatment a success.",[809,7840,7842],{"id":7841},"where-does-this-end-and-a-bi-tool-start","Where does this end and a BI tool start?",[190,7844,7845],{},"The graph is for AI-work lineage and mixed policy / approval / live-system questions. BI remains for canonical commercial metrics. Dumping bookings into the graph as a fake causal model is a mess.",[809,7847,7849],{"id":7848},"what-is-an-intervention-in-this-sense","What is an intervention in this sense?",[190,7851,7852],{},"An approval, a write, a refusal, or a spend stop — something the company actually did (or refused to do) in software. Not a correlation on a slide.",[809,7854,7856],{"id":7855},"why-does-a-fail-closed-gate-matter-for-causality","Why does a fail-closed gate matter for causality?",[190,7858,7859,7860,7862],{},"Because the counterfactual is clean: without the named signer, the live system does not change. Fail-open systems cannot say what ",[270,7861,7674],{}," have happened; they can only hope someone noticed.",[809,7864,7866],{"id":7865},"is-w3c-prov-the-same-as-a-lifecycle-graph","Is W3C PROV the same as a lifecycle graph?",[190,7868,7869],{},"PROV is a standard for provenance concepts. A lifecycle graph is an operational record of AI-mediated work. You can be inspired by PROV without claiming a full W3C implementation.",[809,7871,7873],{"id":7872},"can-we-reconstruct-causes-from-crm-field-history-plus-slack","Can we reconstruct causes from CRM field history plus Slack?",[190,7875,7876],{},"Field history says the value changed. Slack may contain a rumour. Neither joins playbook version, quoted payload, and signer identity as a single chain.",[809,7878,7880],{"id":7879},"does-causal-ai-mean-the-model-uses-causal-graphs-internally","Does “causal AI” mean the model uses causal graphs internally?",[190,7882,7883],{},"Sometimes, in research marketing. In this article it means operations can answer why a change happened. Ask vendors which meaning they are selling.",[809,7885,7887],{"id":7886},"how-should-we-talk-to-the-board","How should we talk to the board?",[190,7889,7890],{},"Separate “we can reconstruct what we did” from “we can estimate market lift.” The first is a control. The second is a study.",[809,7892,7894],{"id":7893},"where-does-the-wiki-fit","Where does the wiki fit?",[190,7896,7897,7898,230],{},"Asserted policy is often the parent of a refusal or a draft. “Which playbook version caused this flag?” is a causal-operations question. See ",[200,7899,2404],{"href":443},[809,7901,7903],{"id":7902},"how-is-this-different-from-audit-logging","How is this different from audit logging?",[190,7905,7906],{},"Audit logs are often thin events. Causal operations needs the join: job, sources, proposal, person, system response. A log that cannot join is a pile.",[255,7908,869],{"id":868},[190,7910,7911,217,7913,3536,7915,230],{},[200,7912,3510],{"href":711},[200,7914,4992],{"href":215},[200,7916,3271],{"href":3270},[255,7918,883],{"id":882},[260,7920,7921,7927,7933],{},[263,7922,7923],{},[200,7924,7926],{"href":7558,"rel":7925},[204],"Stanford Encyclopedia of Philosophy, Causal Models",[263,7928,7929],{},[200,7930,7932],{"href":7564,"rel":7931},[204],"Pearl, Causal inference in statistics: An overview",[263,7934,7935],{},[200,7936,4002],{"href":3638,"rel":7937},[204],{"title":170,"searchDepth":171,"depth":171,"links":7939},[7940,7941,7945,7946,7960,7961],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":7942},[7943,7944],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":7947},[7948,7949,7950,7951,7952,7953,7954,7955,7956,7957,7958,7959],{"id":7820,"depth":1001,"text":7821},{"id":7827,"depth":1001,"text":7828},{"id":7834,"depth":1001,"text":7835},{"id":7841,"depth":1001,"text":7842},{"id":7848,"depth":1001,"text":7849},{"id":7855,"depth":1001,"text":7856},{"id":7865,"depth":1001,"text":7866},{"id":7872,"depth":1001,"text":7873},{"id":7879,"depth":1001,"text":7880},{"id":7886,"depth":1001,"text":7887},{"id":7893,"depth":1001,"text":7894},{"id":7902,"depth":1001,"text":7903},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Causal AI for operations means you can answer “why did this change happen?” with the actual steps and approval — not a guess that “the chatbot caused a lift.”","/blog/what-is-causal-ai-for-operations",{"title":7541,"description":7962},"blog/what-is-causal-ai-for-operations",[1606,7967,4045,7968],"causal-ai","operations","dnOrlkbDj2AQQzXrVNtI2mJOYeFhCAfQGr0mUrKmvmM",{"id":7971,"title":7972,"archived":164,"authors":7973,"badge":7975,"body":7976,"date":3101,"definedTerm":165,"department":165,"description":8415,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":8416,"relatedHeading":165,"seo":8417,"series":1606,"sitemap":130,"status":165,"stem":8418,"subhead":165,"tags":8419,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":8422},"content/blog/what-is-enterprise-rag.md","What is Enterprise RAG",[7974],{"name":184,"to":135},{"label":1024},{"type":167,"value":7977,"toc":8391},[7978,7988,7997,8006,8013,8016,8018,8080,8085,8087,8090,8093,8118,8125,8127,8132,8137,8142,8147,8155,8157,8163,8171,8177,8183,8189,8195,8201,8207,8209,8220,8226,8234,8236,8240,8243,8247,8250,8254,8265,8269,8272,8276,8281,8285,8292,8296,8299,8303,8311,8315,8326,8330,8333,8337,8344,8348,8354,8356,8362,8364],[190,7979,7980,7981,7984,7985,230],{},"RAG stands for ",[193,7982,7983],{},"retrieval-augmented generation",". In plain language: ",[193,7986,7987],{},"look up, then answer",[190,7989,7990,7991,7996],{},"The model does not rely only on what it was trained on. It first fetches supporting documents from a company corpus, then writes the answer using those documents. The original research paper is ",[200,7992,7995],{"href":7993,"rel":7994},"https://arxiv.org/abs/2005.11401",[204],"Lewis et al., Retrieval-Augmented Generation (2020)",". The idea is older than ChatGPT: give the generator evidence at question time so it is less likely to invent.",[190,7998,7999,8001,8002,8005],{},[193,8000,439],{}," is that move with permissions respected. Search runs as a named person or team, not as an admin crawler of everything. Citations include a document, version, and date. It is how you reduce hallucination on ",[270,8003,8004],{},"company"," facts.",[190,8007,8008,8009,8012],{},"It is not, by itself, an operating system, a write gate, or a memory of decisions. ",[200,8010,3556],{"href":3554,"rel":8011},[204]," does not pause because the “user” of the files is an AI. Retrieval is still processing personal data.",[190,8014,8015],{},"The demo omitted the hard parts. A laptop search over a folder of PDFs is not enterprise RAG. Neither is a chatbot that sometimes browses the public web. Enterprise lookup has to survive access lists, freshness SLAs, poisoned documents, and the difference between “this file exists” and “this is policy.”",[255,8017,258],{"id":257},[260,8019,8020,8026,8032,8037,8043,8052,8058,8064,8070],{},[263,8021,8022,8025],{},[193,8023,8024],{},"Corpus."," The set of files and records the AI is allowed to search. At work, this should be the job’s corpus, not the company’s entire Drive.",[263,8027,8028,8031],{},[193,8029,8030],{},"Embedding / vector store."," A numerical fingerprint of text, used to find similar passages. Similarity search fails on invoice IDs and clause numbers unless you also use keywords. At work, “find contract 88421” is a keyword problem pretending to be a semantic one.",[263,8033,8034,8036],{},[193,8035,3203],{}," A clickable source Legal can check — not “according to our documents.” At work, the citation needs a version and a date, or it is a vibe.",[263,8038,8039,8042],{},[193,8040,8041],{},"Hallucination."," Fluent invention. RAG reduces it on company facts. It does not eliminate it, and it does not stop an ungoverned write.",[263,8044,8045,8048,8049,8051],{},[193,8046,8047],{},"Asserted policy."," What the company currently wants. That belongs in a ",[200,8050,4397],{"href":443},", not in whichever PDF sounded closest.",[263,8053,8054,8057],{},[193,8055,8056],{},"Chunking."," Splitting files so search can retrieve a passage. Bad chunking is how a table’s header parts company from its numbers.",[263,8059,8060,8063],{},[193,8061,8062],{},"Freshness."," When the index sees a change. At work, “we changed the vendor template yesterday” is an SLA question.",[263,8065,8066,8069],{},[193,8067,8068],{},"Permission-aware search."," The retriever sees what the user (or the job) may see. A superuser crawler is not enterprise; it is a new data store.",[263,8071,8072,8075,8076,8079],{},[193,8073,8074],{},"Prompt injection via documents."," Retrieved text that instructs the model to ignore policy. The ",[200,8077,977],{"href":375,"rel":8078},[204]," treats that as a security surface.",[190,8081,8082,8083,230],{},"What enterprise RAG is not: a chatbot that sometimes browses the public web; a dump of all tickets into a vector database; a replacement for official playbooks; or permission-aware search sold as a work OS. Search that respects permissions is still search. It does not gate a write. See ",[200,8084,7485],{"href":7484},[255,8086,1160],{"id":1159},[190,8088,8089],{},"It affects you if answers about policy, customers, or finance will be trusted — and if those answers later need a source you can click.",[190,8091,8092],{},"Enterprise lookup adds:",[260,8094,8095,8101,8106,8112],{},[263,8096,8097,8100],{},[193,8098,8099],{},"Permissions."," SharePoint, Salesforce, and Drive access lists still apply.",[263,8102,8103,8105],{},[193,8104,8062],{}," If the index updates on Sundays, your SLA is weekly.",[263,8107,8108,8111],{},[193,8109,8110],{},"Poisoned documents."," Retrieved text can instruct the model to ignore policy.",[263,8113,8114,8117],{},[193,8115,8116],{},"Purpose."," Indexing everything “just in case” is a privacy and quality problem.",[190,8119,8120,8121,8124],{},"Treat RAG as ",[193,8122,8123],{},"infrastructure with an SLA",", not as a magic brain. Separate asserted versus retrieved. Scope retrieval to the job. Demand citations. Assign owners the way you would for a search service. GDPR erasure is harder if you forgot the index.",[809,8126,3290],{"id":3289},[190,8128,8129,8131],{},[193,8130,3295],{}," Retrieval of last year’s close pack is not the close checklist. Numbers in retrieved slides go stale. Finance should insist that thresholds live in asserted wiki tables, and that RAG citations are dated. A fluent answer about recognition policy without a clickable source is not usable in a close.",[190,8133,8134,8136],{},[193,8135,3310],{}," Citations are the point. “According to our documents” is not reviewable. Legal also owns the processing question: indexing HR files into a shared vector store is a new copy of personal data. Erasure requests have to hit the index, not only the source system.",[190,8138,8139,8141],{},[193,8140,3319],{}," Freshness and owners. Ops should treat the retriever like any other search service: uptime, lag, and who gets paged when the wrong SOP is served. Chunking errors show up as “the agent missed the table.”",[190,8143,8144,8146],{},[193,8145,3325],{}," Competitive decks and old playbooks are semantically close to this quarter’s question. Without a conflict rule that wiki wins, GTM will ship last year’s discount floor because it matched the query. RAG without assertion is folklore with better ranking.",[190,8148,8149,8151,8152,8154],{},[193,8150,3331],{}," Superuser crawlers, poisoned documents, and a second store of sensitive text. Security should ask who the retriever authenticates as, whether ",[200,8153,758],{"href":757}," helpers search as a superuser, and whether prompt injection in a PDF can change tool behaviour. Network search products are not write gates.",[809,8156,3340],{"id":3339},[190,8158,8159,8162],{},[193,8160,8161],{},"Indexing everything."," Quality falls. Privacy rises. Purpose disappears.",[190,8164,8165,8168,8169,230],{},[193,8166,8167],{},"RAG as an OS."," Lookup does not isolate jobs, quote writes, or store decisions. See ",[200,8170,4519],{"href":4371},[190,8172,8173,8176],{},[193,8174,8175],{},"RAG as the wiki."," Retrieved files are what exists. The wiki is what is in force.",[190,8178,8179,8182],{},[193,8180,8181],{},"Citations without versions."," Legal cannot check “the wiki” or “our Drive.”",[190,8184,8185,8188],{},[193,8186,8187],{},"Warehouse SQL as a substitute."," “What is our revenue recognition policy?” is retrieval. “What was Q4 revenue by region?” is structured query. Many jobs need both.",[190,8190,8191,8194],{},[193,8192,8193],{},"Assuming hallucination is solved."," Missing files still produce fluent guesses. Ungoverned writes still land.",[190,8196,8197,8198,8200],{},"Good looks like: permission-aware retrieval scoped to the ",[200,8199,1083],{"href":215},", hybrid keyword plus similarity, dated citations, a wiki conflict rule, an index SLA, and a write gate that does not care how good the retrieval was. Failure looks like a tenant-wide vector lake labelled “the brain.”",[190,8202,8203,8204,8206],{},"Lewis et al. (2020) showed that lookup-then-answer reduces invention on facts in the corpus. Enterprise buyers still have to decide which corpus, whose permissions, and whether a retrieved PDF is allowed to outrank the ",[200,8205,326],{"href":443},". The paper does not answer those questions. Your runtime must.",[255,8208,793],{"id":792},[190,8210,8211,8212,8215,8216,8219],{},"Nimbus uses RAG-like retrieval ",[193,8213,8214],{},"inside"," a work OS, not as a standalone search SKU. Wiki is asserted policy. Connectors supply live context. Workstreams pre-scope the corpus. Governance still gates any write. The Lifecycle Graph stores which sources were used for a decision — retrieval becomes part of ",[200,8217,8218],{"href":3270},"institutional memory",", not a forgotten context window.",[190,8221,8222,8223,8225],{},"A common plug so AI apps can use the same tools — ",[200,8224,5140],{"href":757}," — can standardise access to repositories. It does not implement access lists for you. A tool that searches Drive as a superuser is still a superuser.",[190,8227,3411,8228,217,8230,3536,8232,230],{},[200,8229,3414],{"href":28},[200,8231,31],{"href":32},[200,8233,39],{"href":40},[255,8235,807],{"id":806},[809,8237,8239],{"id":8238},"will-rag-stop-the-model-making-things-up","Will RAG stop the model making things up?",[190,8241,8242],{},"It reduces invention on facts that exist in authorised files. It does not make the model honest about missing files, and it does not replace a person on a live-system change.",[809,8244,8246],{"id":8245},"is-indexing-everything-just-in-case-a-good-idea","Is indexing everything “just in case” a good idea?",[190,8248,8249],{},"No. Indexing without a purpose is a privacy and quality problem. Scope the corpus to the job.",[809,8251,8253],{"id":8252},"how-is-this-different-from-a-company-wiki","How is this different from a company wiki?",[190,8255,8256,8257,8260,8261,8264],{},"The wiki is what the company ",[270,8258,8259],{},"wants"," to be true. RAG is what ",[270,8262,8263],{},"exists"," in files. If they conflict, the wiki should win unless a human promotes a change.",[809,8266,8268],{"id":8267},"can-warehouse-sql-replace-rag","Can warehouse SQL replace RAG?",[190,8270,8271],{},"They answer different questions. Policy prose is retrieval. Regional revenue is a query. Many real jobs need both.",[809,8273,8275],{"id":8274},"is-glean-or-similar-an-enterprise-ai-os","Is Glean (or similar) an enterprise AI OS?",[190,8277,8278,8279,230],{},"Permission-aware search is still search. It does not, by itself, quote a CRM write or bind a named signer. See ",[200,8280,7485],{"href":7484},[809,8282,8284],{"id":8283},"why-do-invoice-numbers-fail-in-vector-search","Why do invoice numbers fail in vector search?",[190,8286,8287,8288,8291],{},"Embeddings capture similarity of meaning, not identity of tokens. Hybrid search — keywords plus vectors — is how you find ",[422,8289,8290],{},"INV-88421"," instead of a semantically nearby invoice.",[809,8293,8295],{"id":8294},"how-fast-should-the-index-update","How fast should the index update?",[190,8297,8298],{},"As fast as the decision you are supporting. If a template changed yesterday and the agent still cites last month, your SLA is wrong. Publish the lag.",[809,8300,8302],{"id":8301},"what-is-document-based-prompt-injection","What is document-based prompt injection?",[190,8304,8305,8306,8310],{},"A retrieved file that says, in effect, “ignore previous instructions.” Treat retrieved text as untrusted input. The ",[200,8307,8309],{"href":375,"rel":8308},[204],"OWASP LLM list"," is the starting point. A wiki conflict rule and a write gate still matter.",[809,8312,8314],{"id":8313},"does-gdpr-apply-to-the-vector-index","Does GDPR apply to the vector index?",[190,8316,8317,8318,876,8322,230],{},"Yes, if it holds personal data. The index is another copy. Erasure, purpose, and access control apply. See the ",[200,8319,8321],{"href":3554,"rel":8320},[204],"GDPR text",[200,8323,8325],{"href":3547,"rel":8324},[204],"ICO AI guidance",[809,8327,8329],{"id":8328},"can-mcp-make-retrieval-respect-permissions","Can MCP make retrieval respect permissions?",[190,8331,8332],{},"Only if the helper is built that way. The protocol will happily pass superuser results.",[809,8334,8336],{"id":8335},"should-customer-facing-chatbots-use-rag-on-the-public-website-plus-internal-policy","Should customer-facing chatbots use RAG on the public website plus internal policy?",[190,8338,8339,8340,8343],{},"Internal policy in a customer bot is how invented fares happen unless a human still owns the commitment. Air Canada’s case — ",[200,8341,2043],{"href":2041,"rel":8342},[204]," — is retrieval-plus-generation without a working gate.",[809,8345,8347],{"id":8346},"how-do-we-know-which-sources-a-decision-used","How do we know which sources a decision used?",[190,8349,8350,8351,8353],{},"Record them on the ",[200,8352,3186],{"href":711},". A context window that evaporates is not memory.",[255,8355,869],{"id":868},[190,8357,8358,876,8360,230],{},[200,8359,2404],{"href":443},[200,8361,3271],{"href":3270},[255,8363,883],{"id":882},[260,8365,8366,8371,8376,8381,8386],{},[263,8367,8368],{},[200,8369,7995],{"href":7993,"rel":8370},[204],[263,8372,8373],{},[200,8374,3556],{"href":3554,"rel":8375},[204],[263,8377,8378],{},[200,8379,977],{"href":375,"rel":8380},[204],[263,8382,8383],{},[200,8384,3549],{"href":3547,"rel":8385},[204],[263,8387,8388],{},[200,8389,2228],{"href":2041,"rel":8390},[204],{"title":170,"searchDepth":171,"depth":171,"links":8392},[8393,8394,8398,8399,8413,8414],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":8395},[8396,8397],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":8400},[8401,8402,8403,8404,8405,8406,8407,8408,8409,8410,8411,8412],{"id":8238,"depth":1001,"text":8239},{"id":8245,"depth":1001,"text":8246},{"id":8252,"depth":1001,"text":8253},{"id":8267,"depth":1001,"text":8268},{"id":8274,"depth":1001,"text":8275},{"id":8283,"depth":1001,"text":8284},{"id":8294,"depth":1001,"text":8295},{"id":8301,"depth":1001,"text":8302},{"id":8313,"depth":1001,"text":8314},{"id":8328,"depth":1001,"text":8329},{"id":8335,"depth":1001,"text":8336},{"id":8346,"depth":1001,"text":8347},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Enterprise RAG is looking up authorised company files before the AI answers, with permissions respected — lookup-then-answer, not an operating system.","/blog/what-is-enterprise-rag",{"title":7972,"description":8415},"blog/what-is-enterprise-rag",[1606,8420,8421,3588],"rag","retrieval","109PE-bl9ovP1Q_x37YcWXhp1Jc_btmr5yC1dDSCQG0",{"id":8424,"title":8425,"archived":164,"authors":8426,"badge":8428,"body":8429,"date":1008,"definedTerm":165,"department":165,"description":9016,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":9017,"relatedHeading":165,"seo":9018,"series":1606,"sitemap":130,"status":165,"stem":9019,"subhead":165,"tags":9020,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":9021},"content/blog/what-is-harness-engineering.md","What is Harness Engineering",[8427],{"name":184,"to":135},{"label":1024},{"type":167,"value":8430,"toc":8998},[8431,8436,8460,8463,8479,8481,8555,8561,8563,8570,8572,8592,8605,8619,8622,8626,8635,8654,8667,8679,8691,8703,8706,8710,8716,8722,8735,8744,8752,8756,8763,8769,8775,8781,8795,8798,8810,8814,8817,8831,8833,8837,8840,8844,8850,8854,8861,8865,8872,8876,8893,8895,8906,8908,8916,8918],[190,8432,8433,8435],{},[193,8434,359],{}," is the practice of treating the runtime around a model as the system you design, test, and tighten — so that when an agent fails, you change the environment, not only the prompt.",[190,8437,8438,8441,8442,8445,8446,8449,8450,8454,8455,8459],{},[200,8439,6811],{"href":405,"rel":8440},[204]," defines the object: Agent = Model + Harness. Harness engineering is what you ",[270,8443,8444],{},"do"," to that object. ",[200,8447,2307],{"href":953,"rel":8448},[204]," puts the payoff in one line: a decent model with a great harness beats a great model with a bad harness. ",[200,8451,8453],{"href":946,"rel":8452},[204],"Birgitta Böckeler’s article on martinfowler.com"," is the user’s-side map for coding agents: guides in, sensors back. Thoughtworks then asked the organisational question: ",[200,8456,8458],{"href":6998,"rel":8457},[204],"how you scale that harness across a company"," without turning every team into a snowflake of markdown files.",[190,8461,8462],{},"The practice showed up because prompt engineering hit a wall that everyone could see and nobody wanted to name. You can spend a week on a system prompt. The agent will still skip the test, ignore the style guide, or report the task finished. The model is non-deterministic. The prompt is interpreted, not executed. The harness is code. That is the whole discipline.",[190,8464,8465,8466,8470,8471,8475,8476,8478],{},"This is not a replacement for ",[200,8467,8469],{"href":411,"rel":8468},[204],"prompt"," or ",[200,8472,8474],{"href":307,"rel":8473},[204],"context"," work. Those live ",[270,8477,8214],{}," the harness. Harness engineering is the wider loop: every failure becomes a rule, a hook, a test, or a denied tool — the ratchet Osmani describes — so the same mistake is cheaper the second time and impossible the tenth.",[255,8480,258],{"id":257},[260,8482,8483,8489,8506,8516,8526,8532,8538,8547],{},[263,8484,8485,8488],{},[193,8486,8487],{},"Ratchet."," A failure updates the harness. Commented-out test → pre-commit hook and a reviewer check. Invented CRM field → schema quote and a Hard gate. If you only fix the artefact by hand, you did operations. You did not do harness engineering.",[263,8490,8491,8494,8495,8497,8498,217,8500,8502,8503,8505],{},[193,8492,8493],{},"Guides (feed-forward)."," Context the agent gets ",[270,8496,3299],{}," it acts: ",[422,8499,434],{},[422,8501,646],{},", architecture notes, ",[200,8504,4397],{"href":443}," playbooks. Böckeler’s term. Advice. Necessary. Not a stop.",[263,8507,8508,8511,8512,8515],{},[193,8509,8510],{},"Sensors (feedback)."," Deterministic checks (compiler, linter, schema, pytest) and inferential checks (LLM reviewer, specialist critic). ",[200,8513,2565],{"href":1699,"rel":8514},[204],". Without sensors the agent grades its own homework.",[263,8517,8518,8521,8522,8525],{},[193,8519,8520],{},"Hooks."," Lifecycle intercepts that always run. ",[200,8523,462],{"href":460,"rel":8524},[204]," can block a tool with exit code 2. LangChain middleware is the library form. A guide that says “never run rm -rf” is not a hook.",[263,8527,8528,8531],{},[193,8529,8530],{},"Harness-as-a-service."," Osmani’s HaaS framing: you used to build on completion APIs; you now build on runtime APIs (Claude Agent SDK, Codex SDK, OpenAI Agents SDK) that already own the loop, sandbox, and hooks. You configure; you do not re-implement ReAct.",[263,8533,8534,8537],{},[193,8535,8536],{},"Skill issue."," HumanLayer’s joke with a serious edge: most agent failures are configuration. Blaming the model first is how teams wait for the next release instead of adding a sensor.",[263,8539,8540,463,8542,8546],{},[193,8541,6548],{},[200,8543,8545],{"href":1386,"rel":8544},[204],"Thoughtworks’ enterprise layer",": who may build which harness, how exceptions work, identity, economics, learning. The gap after builder harnesses (Claude Code, Cursor) and user harnesses (guides and sensors on a repo).",[263,8548,8549,8552,8553,230],{},[193,8550,8551],{},"Eval loop."," Independent verification that does not take the model’s word. SWE-bench and Terminal-Bench for code. Quoted payload vs executed write for operations. See ",[200,8554,5335],{"href":503},[190,8556,8557,8558,8560],{},"In Nimbus, harness engineering for operators looks like: wiki revisions as guides, connector scopes as tool policy, Soft / Hard / Critical as hooks on the write plane, and the ",[200,8559,23],{"href":24}," as the sensor log you can query. That is the same discipline as adding a linter. The artefact is a signed CRM change rather than a green CI job.",[255,8562,1160],{"id":1159},[190,8564,8565,8566,8569],{},"If you only tune prompts, every incident is a conversation. If you engineer the harness, incidents become tests. ",[200,8567,5252],{"href":363,"rel":8568},[204]," Measure and Manage steps assume you can change controls after you observe harm. A prompt history is not a control change. A hook that now fires is.",[190,8571,1171],{},[260,8573,8574,8577,8583,8586,8589],{},[263,8575,8576],{},"agents already write code or propose writes to live systems",[263,8578,8579,8580,8582],{},"two teams have two ",[422,8581,646],{}," files that contradict Legal",[263,8584,8585],{},"you cannot say which harness version ran last Tuesday",[263,8587,8588],{},"spend is “the model was verbose” rather than “the loop had no budget”",[263,8590,8591],{},"auditors ask who could have stopped the action, and the answer is “the model was supposed to ask”",[190,8593,8594,8597,8598,8600,8601,8604],{},[200,8595,2464],{"href":383,"rel":8596},[204]," keeps showing usage without redesign. Harness engineering ",[270,8599,272],{}," the redesign for agentic work: not a new department named AI, a runtime with stops. ",[200,8602,966],{"href":369,"rel":8603},[204]," wants named AI actors and documented operational controls. You cannot name actors if every operator’s personal GPT is a different harness.",[190,8606,8607,8608,8611,8612,8614,8615,8618],{},"Coding teams already have half of this and do not always notice. Types, tests, CI, CODEOWNERS — Böckeler’s point is that those ",[270,8609,8610],{},"are"," sensors. The work is to point the agent at them and to add the ones that are missing (architecture fitness, behaviour: did it do what was asked). Operations teams usually have the human version — maker-checker, SoD, SOX — and have not yet wired those instincts into a loop. ",[200,8613,474],{"href":228}," is that wiring. ",[200,8616,8617],{"href":241},"Human-in-the-loop approval architecture"," is the state machine.",[190,8620,8621],{},"Air Canada’s chatbot and the sanctioned ChatGPT brief are what happens when generation reaches a system of record with no ratchet. The fix is not a sterner system prompt. The fix is a harness that cannot emit a commitment or a filing until a named person has seen the artefact.",[255,8623,8625],{"id":8624},"the-practice-not-the-slogan","The practice, not the slogan",[190,8627,8628,8631,8632,8634],{},[193,8629,8630],{},"1. Work backward from the behaviour you cannot afford to miss once."," Inner loop: never merge without tests; never ",[422,8633,2662],{}," to main. Outer loop: never PATCH Opportunity.Amount without a Hard quote. Write those as hooks, not as paragraphs.",[190,8636,8637,463,8640,8644,8645,8647,8648,8650,8651,8653],{},[193,8638,8639],{},"2. Separate advice from invariants.",[200,8641,8643],{"href":836,"rel":8642},[204],"Anthropic’s steering note for Claude Code"," is unusually clear: ",[422,8646,646],{}," is always-on context; hooks fire on events and can block. If a rule must hold when the model is tired, it graduates from markdown to a hook. Enterprise equivalent: playbooks in the ",[200,8649,326],{"href":28}," versus the interceptor in ",[200,8652,340],{"href":40},". If they conflict, the interceptor wins.",[190,8655,8656,8659,8660,8663,8664,230],{},[193,8657,8658],{},"3. Put verification outside the generator."," Anthropic’s long-running harness uses incremental commits and end-to-end checks so later sessions cannot declare victory by vibes. Coding sensors: pytest, tsc, lint. Enterprise sensors: schema of the quote, identity of the signer, hash of the payload that executed, connector grant still attached. The model may ",[270,8661,8662],{},"propose"," that it is done. The harness ",[270,8665,8666],{},"decides",[190,8668,8669,8672,8673,8675,8676,8678],{},[193,8670,8671],{},"4. Version the harness."," Which ",[422,8674,434],{},", which wiki revision, which team contract, which approval tier ran. ",[200,8677,1955],{"href":1954}," already treats workflow version as an input. Harness engineering extends that to tools and gates. Hot-patching production prompts without a change record is how Tuesday becomes unexplained.",[190,8680,8681,8684,8685,876,8687,8690],{},[193,8682,8683],{},"5. Budget the loop."," Max steps and a cost cap that do not depend on the model’s judgement. Seat licences hide this; metered work makes it visible. See ",[200,8686,3520],{"href":1261},[200,8688,8689],{"href":524},"AI cost control architecture",". Always-flagship is not careful. It is an unengineered harness.",[190,8692,8693,8696,8697,8699,8700,8702],{},[193,8694,8695],{},"6. Do not fork a harness per person."," User-owned bots are how mandates drift. Org-level ",[200,8698,221],{"href":220}," assigned to ",[200,8701,216],{"href":215}," is the enterprise form of “one CI config per repo, not one per intern.” Thoughtworks’ organisational harness is this ownership question: who is allowed to add a write tool.",[190,8704,8705],{},"Nimbus encodes several of these as product defaults — read-only connectors until you enable write, quoted payloads, graph on the way out — because operators should not have to re-implement ReAct to get a ratchet. You can still fail the practice: a wiki that is never updated, a Critical tier nobody uses, a graph nobody queries. The product is not the practice. The practice is whether last month’s incident produced a new gate.",[255,8707,8709],{"id":8708},"how-this-differs-from-adjacent-crafts","How this differs from adjacent crafts",[190,8711,8712,8715],{},[193,8713,8714],{},"Prompt engineering"," improves a single call. Necessary for tone, tool descriptions, and “what good looks like.” Insufficient for tool dispatch, identity, and replay.",[190,8717,8718,8721],{},[193,8719,8720],{},"Context engineering"," governs what the model sees this turn: compaction, retrieval, files. Anthropic’s initializer agent is context engineering in a harness. It is not permission to write NetSuite.",[190,8723,8724,8727,8728,8731,8732,230],{},[193,8725,8726],{},"Platform / DevOps."," CI, sandboxes, secrets. Harness engineering ",[270,8729,8730],{},"reuses"," those as sensors and execution environments. It adds the fact that the component in the loop is non-deterministic, so “the job returned zero” is not enough: you need independent tests of the ",[270,8733,8734],{},"claim",[190,8736,8737,8740,8741,8743],{},[193,8738,8739],{},"Governance-as-PDF."," Policy. Harness engineering is whether the tool call is reachable. ",[200,8742,4407],{"href":4406}," is the buying cousin.",[190,8745,8746,8749,8750,230],{},[193,8747,8748],{},"Framework assembly."," Writing LangGraph nodes is building a harness in code. Harness engineering is the ongoing discipline after the graph exists: sensors, ownership, eval. See ",[200,8751,5400],{"href":593},[255,8753,8755],{"id":8754},"four-layers-one-ratchet","Four layers, one ratchet",[190,8757,8758,8762],{},[200,8759,8761],{"href":1386,"rel":8760},[204],"Thoughtworks’ July 2026 essay"," is the organisational map most engineering blogs skip. They split enterprise AI into four harness layers. Most companies have built one, maybe two. The gap is not a smarter model.",[190,8764,8765,8768],{},[193,8766,8767],{},"Layer 1 — the model."," Substrate. Choice still matters for cost, residency, and task fit. It is the wrong unit of analysis for a programme. Teams that prototype, hit a failure, and buy the next flagship are looping on layer 1.",[190,8770,8771,8774],{},[193,8772,8773],{},"Layer 2 — the builder harness."," Frameworks, tool access, memory, where inference runs. LangChain, Claude Agent SDK, AIP-style platforms, Nimbus’s hosted loop. Without layer 3, every team invents naming and review. Without layer 4, nobody owns failure.",[190,8776,8777,8780],{},[193,8778,8779],{},"Layer 3 — the user harness."," Guides and sensors on the job. Böckeler’s taxonomy lives here. Thoughtworks add a useful matrix: feed-forward vs feedback, crossed with deterministic vs probabilistic. Deterministic feed-forward is a whitelist and a spend ceiling — cheap, auditable, default. Probabilistic feed-forward is a runbook retrieved at decision time. Deterministic feedback is schema validation after the act. Probabilistic feedback is an eval model on a rubric — expensive, use on critical paths only. A guide with no sensor is theatre.",[190,8782,8783,8786,8787,8790,8791,8794],{},[193,8784,8785],{},"Layer 4 — the organisational harness."," Who may grant which autonomy, escalation, accountability when layers 1–3 all “worked” and the company still took harm. Thoughtworks’ public cases: Parloa, where versioned rules, skills, commands, and helpers lived ",[270,8788,8789],{},"in the repo"," (they report p95 latency drops they attribute to harness architecture, not a new model); Morgan Stanley, where hygiene and CVE triage used a ",[270,8792,8793],{},"delegation tier"," instead of a yes/no “do we trust the agent.” You do not need those vendors to accept the lesson: governance that is not versioned next to the work decays.",[190,8796,8797],{},"Harness engineering is the steering loop across those layers. Sensor data reveals a miss. Guides update. Hooks graduate. Templates change. The next job is cheaper. An organisation with that loop has a compounding harness. An organisation without one has markdown that rots while models improve.",[190,8799,8800,8801,8803,8804,8806,8807,8809],{},"A concrete week: Monday the agent comments out a flaky test (inner) or proposes Amount without CloseDate (outer). Tuesday a human fixes the artefact. That is operations. Harness engineering is Tuesday’s hook or schema sensor, Wednesday’s wiki or ",[422,8802,434],{}," line, Thursday’s replay that the new control fired. Friday you run the job ten times and count refuses. Nimbus makes the outer version of that week a product surface — ",[200,8805,340],{"href":40}," queues, ",[200,8808,778],{"href":24}," export — so operators are not waiting on a platform sprint to add the sensor. You still have to look at the refuse count. A product without a steering cadence is layer 2 with a nicer UI.",[255,8811,8813],{"id":8812},"what-good-looks-like","What good looks like",[190,8815,8816],{},"Good: a named owner for the harness (not “AI working group”), a cadence that turns incidents into controls, deterministic gates on knowable bounds, inferential checks only where judgement is required, versioned guides, exportable traces. Failure: a new system prompt after every incident; sensors the agent can skip; no owner; SWE-bench as the only score for a CRM job; layer 4 as a PDF.",[190,8818,8819,8823,8824,8826,8827,8830],{},[200,8820,8822],{"href":953,"rel":8821},[204],"Osmani’s ratchet"," and Thoughtworks’ steering loop are the same instinct. ",[200,8825,1659],{"href":251}," asks whether your vendor lets you ",[270,8828,8829],{},"run"," that instinct.",[255,8832,807],{"id":806},[809,8834,8836],{"id":8835},"who-coined-harness-engineering","Who coined “harness engineering”?",[190,8838,8839],{},"The phrase circulated in early 2026 across OpenAI engineering notes (Ryan Lopopolo’s line of work), LangChain’s anatomy posts, Böckeler at Thoughtworks, and Osmani’s synthesis. Treat it as a shared 2026 name for work teams were already doing, not a trademarked method.",[809,8841,8843],{"id":8842},"is-this-only-for-coding-agents","Is this only for coding agents?",[190,8845,8846,8847,8849],{},"The literature is densest there because tests already exist. The discipline is the same for RevOps and Finance: independent sensors, fail-closed writes, versioned context. An ",[200,8848,1337],{"href":864}," is that application.",[809,8851,8853],{"id":8852},"do-we-wait-for-a-better-model-instead","Do we wait for a better model instead?",[190,8855,8856,8857,8860],{},"You still buy better models. You do not pause the ratchet. Stronger models attempt larger jobs and fail in new ways. Anthropic’s long-running work exists ",[270,8858,8859],{},"because"," models got good enough to outlast a window.",[809,8862,8864],{"id":8863},"how-do-we-start-this-quarter","How do we start this quarter?",[190,8866,8867,8868,8871],{},"Pick one job that already has a finish line. Encode guides. Attach one deterministic sensor. Add one hook that can refuse. Run it ten times. Every failure updates the harness. That is a ",[200,8869,8870],{"href":1278},"proof of value"," for the practice, not a chat demo.",[809,8873,8875],{"id":8874},"how-does-nimbus-fit-without-becoming-the-definition","How does Nimbus fit without becoming the definition?",[190,8877,8878,8879,217,8881,217,8884,217,8887,8889,8890,8892],{},"Nimbus is an outer harness you can hire: ",[200,8880,216],{"href":32},[200,8882,8883],{"href":20},"teams",[200,8885,8886],{"href":40},"gates",[200,8888,778],{"href":24},". Score it the way you score Claude Code: can you add a sensor, refuse a write, and replay who signed. ",[200,8891,1659],{"href":251}," is the sheet.",[809,8894,853],{"id":852},[190,8896,8897,8899,8900,8902,8903,8905],{},[200,8898,5095],{"href":286}," for the repo/company cut. ",[200,8901,195],{"href":2140}," for the parts. ",[200,8904,1106],{"href":246}," if you still need the noun.",[255,8907,869],{"id":868},[190,8909,8910,876,8912,230],{},[200,8911,4519],{"href":4371},[200,8913,8915],{"href":8914},"how-to-solve-ai-that-cannot-write-back-safely","How to solve AI that cannot write back safely",[255,8917,883],{"id":882},[260,8919,8920,8925,8930,8935,8940,8946,8951,8956,8963,8968,8973,8978,8983,8988,8993],{},[263,8921,8922],{},[200,8923,897],{"href":405,"rel":8924},[204],[263,8926,8927],{},[200,8928,891],{"href":202,"rel":8929},[204],[263,8931,8932],{},[200,8933,948],{"href":946,"rel":8934},[204],[263,8936,8937],{},[200,8938,2202],{"href":1699,"rel":8939},[204],[263,8941,8942],{},[200,8943,8945],{"href":6998,"rel":8944},[204],"Thoughtworks, Scaling the enterprise harness (podcast)",[263,8947,8948],{},[200,8949,1545],{"href":1386,"rel":8950},[204],[263,8952,8953],{},[200,8954,955],{"href":953,"rel":8955},[204],[263,8957,8958],{},[200,8959,8962],{"href":8960,"rel":8961},"https://www.oreilly.com/radar/agent-harness-engineering/",[204],"O’Reilly Radar, Agent harness engineering",[263,8964,8965],{},[200,8966,923],{"href":411,"rel":8967},[204],[263,8969,8970],{},[200,8971,929],{"href":307,"rel":8972},[204],[263,8974,8975],{},[200,8976,935],{"href":836,"rel":8977},[204],[263,8979,8980],{},[200,8981,941],{"href":460,"rel":8982},[204],[263,8984,8985],{},[200,8986,983],{"href":383,"rel":8987},[204],[263,8989,8990],{},[200,8991,365],{"href":363,"rel":8992},[204],[263,8994,8995],{},[200,8996,966],{"href":369,"rel":8997},[204],{"title":170,"searchDepth":171,"depth":171,"links":8999},[9000,9001,9002,9003,9004,9005,9006,9014,9015],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160},{"id":8624,"depth":171,"text":8625},{"id":8708,"depth":171,"text":8709},{"id":8754,"depth":171,"text":8755},{"id":8812,"depth":171,"text":8813},{"id":806,"depth":171,"text":807,"children":9007},[9008,9009,9010,9011,9012,9013],{"id":8835,"depth":1001,"text":8836},{"id":8842,"depth":1001,"text":8843},{"id":8852,"depth":1001,"text":8853},{"id":8863,"depth":1001,"text":8864},{"id":8874,"depth":1001,"text":8875},{"id":852,"depth":1001,"text":853},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Harness engineering is the 2026 practice of fixing the environment when an agent fails — tools, hooks, tests, and stops — instead of rewriting the prompt and hoping the next model call behaves.","/blog/what-is-harness-engineering",{"title":8425,"description":9016},"blog/what-is-harness-engineering",[1606,5649,1015,2269],"J7WO8MGgOG9nFjnGjs25nfHH5sz-UYeKtt3NvK0eueE",{"id":9023,"title":9024,"archived":164,"authors":9025,"badge":9027,"body":9028,"date":3101,"definedTerm":165,"department":165,"description":9454,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":9455,"relatedHeading":165,"seo":9456,"series":1606,"sitemap":130,"status":165,"stem":9457,"subhead":165,"tags":9458,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":9461},"content/blog/what-is-human-in-the-loop-ai.md","What is Human-in-the-Loop AI",[9026],{"name":184,"to":135},{"label":1024},{"type":167,"value":9029,"toc":9430},[9030,9036,9039,9052,9055,9058,9060,9121,9128,9130,9133,9136,9147,9154,9160,9167,9170,9172,9177,9182,9190,9195,9207,9209,9215,9220,9226,9232,9238,9244,9250,9260,9272,9275,9277,9280,9287,9292,9299,9301,9305,9308,9312,9315,9319,9322,9326,9329,9333,9336,9340,9347,9351,9356,9360,9363,9367,9370,9374,9377,9381,9384,9388,9391,9393,9403,9405],[190,9031,9032,9033,230],{},"Human-in-the-loop AI, in everyday language, means ",[193,9034,9035],{},"a person must approve before the AI can finish the job",[190,9037,9038],{},"Not “a human might read the chat.” Not a footer that says this content was generated. A gate the software cannot skip.",[190,9040,9041,9042,9045,9046,9051],{},"In February 2024, a British Columbia tribunal held Air Canada responsible for a chatbot that invented a bereavement-fare policy. ",[200,9043,4209],{"href":2041,"rel":9044},[204]," that the airline’s argument — the chatbot is a separate legal entity — failed. The decision is ",[200,9047,9050],{"href":9048,"rel":9049},"https://decisions.civilresolutionbc.ca/crt/crtd/en/525448/1/document.do",[204],"Moffatt v. Air Canada",". A customer relied on the invented fare. A human did not catch the fiction before it became a commitment.",[190,9053,9054],{},"That is the class of failure this article is about.",[190,9056,9057],{},"The phrase is older than ChatGPT. Safety engineering already distinguished a signer on every payload from a supervisor with a kill switch. Generative AI borrowed the label and diluted it. Vendors now say “human in the loop” for a thumbs-up on a chat, a weekly review of logs, or a prompt that says “ask the user first.” Only one of those is a gate.",[255,9059,258],{"id":257},[260,9061,9062,9068,9077,9082,9091,9097,9102,9108,9115],{},[263,9063,9064,9067],{},[193,9065,9066],{},"In the loop."," The process cannot proceed past a gate without a human act. At work, the CRM write does not execute until a named person signs the quoted fields.",[263,9069,9070,9073,9074,9076],{},[193,9071,9072],{},"On the loop."," The system runs; a human ",[270,9075,1229],{}," stop it. Intervention is possible. It is not required per action. At work, this may be acceptable for read-only monitoring. It is not a write control.",[263,9078,9079,9081],{},[193,9080,4349],{}," A checkbox “I understand this is AI,” or a prompt that says “ask the user first,” while the model may still act.",[263,9083,9084,463,9087,9090],{},[193,9085,9086],{},"Effective oversight.",[200,9088,4296],{"href":599,"rel":9089},[204]," Article 14: for higher-risk systems, people must be able to interpret outputs, stay aware that automation can lull them, and interrupt the system.",[263,9092,9093,9096],{},[193,9094,9095],{},"Rubber stamp."," A gate that fires so often people auto-click. That is not oversight. It is fatigue.",[263,9098,9099,9101],{},[193,9100,4138],{}," Identity bound to the decision. Shared inboxes destroy this.",[263,9103,9104,9107],{},[193,9105,9106],{},"Quote / payload."," The exact change in the language of the live system — opportunity fields, journal lines, email body — not a wall of prompt text.",[263,9109,9110,9112,9113,230],{},[193,9111,4144],{}," Missing approval means nothing happens. See ",[200,9114,3286],{"href":228},[263,9116,9117,9120],{},[193,9118,9119],{},"Maker-checker."," An older control: one person proposes, another authorises. HITL for AI is that instinct when the proposer is a model.",[190,9122,9123,9127],{},[200,9124,9126],{"href":1904,"rel":9125},[204],"Mata v. Avianca"," is the cousin case on the legal side: fluent fiction entered a court record because no working check caught invented citations. The loop failed before filing, not after.",[255,9129,1160],{"id":1159},[190,9131,9132],{},"Enterprise buyers should demand a person at the gate for writes to live business systems and for customer-facing commitments. They may accept “on the loop” for read-only monitoring. They should reject theatre.",[190,9134,9135],{},"It affects you if AI can:",[260,9137,9138,9141,9144],{},[263,9139,9140],{},"change records or money",[263,9142,9143],{},"send a customer a message that asserts a policy, price, or term",[263,9145,9146],{},"affect employment, credit, or people’s rights",[190,9148,9149,9150,9153],{},"Place people where ",[193,9151,9152],{},"risk and reversibility"," change: before writes, before external messages, and at exception thresholds (amount, region, data class).",[190,9155,9156,9157,9159],{},"Do ",[193,9158,7688],{}," put humans on every sentence. A gate that fires fifty times a day will be auto-clicked.",[190,9161,9162,9163,9166],{},"The person who already owns that class of change in the analogue process should sign it here. Inventing an “AI champion” who approves finance journals ",[270,9164,9165],{},"and"," legal emails is how you get a rubber stamp.",[190,9168,9169],{},"Show the change in the language of the live system. A person cannot oversee what they cannot parse.",[809,9171,3290],{"id":3289},[190,9173,9174,9176],{},[193,9175,3295],{}," Journals, forecast overrides, and material fields need the same owner who would sign in the analogue close. A champion who does not own the ledger will click through. Rejects are success: they prove the gate. A six-month zero reject rate is a finding.",[190,9178,9179,9181],{},[193,9180,3310],{}," Customer commitments and filings need a signer who can interpret the payload. Air Canada is customer-facing fiction. Mata v. Avianca is professional fiction entering a record. Legal should also refuse “Act compliant” claims that rest only on a button. Article 14 is a bundle of duties, not a widget.",[190,9183,9184,9186,9187,9189],{},[193,9185,3319],{}," Place gates at reversibility boundaries. Ops should measure time-to-approved-write and reject rate, and should treat human wait as a first-class ",[200,9188,5666],{"href":1954}," step, not a Slack nudge.",[190,9191,9192,9194],{},[193,9193,3325],{}," Friction is real. The honest comparison is unreviewed mutation versus incident response, not versus a demo that writes instantly. GTM should not be asked to approve legal emails, and legal should not be asked to approve Amount.",[190,9196,9197,9199,9200,9202,9203,9206],{},[193,9198,3331],{}," The gate must be unskippable by the model, including after prompt injection. A jailbreak can trick the model into ",[270,9201,5847],{}," a bad write. It should not be able to ",[270,9204,9205],{},"execute"," without a quote and a signer. Identity binding matters: a generic “approve” in a shared inbox is not a control.",[809,9208,3340],{"id":3339},[190,9210,9211,9214],{},[193,9212,9213],{},"On the loop as in the loop."," A kill switch is not a per-action signer.",[190,9216,9217,9219],{},[193,9218,4349],{}," Footers, checkboxes, and “shall I proceed?” in unbound chat.",[190,9221,9222,9225],{},[193,9223,9224],{},"Too many gates."," Fatigue produces rubber stamps. Fewer gates, better quotes.",[190,9227,9228,9231],{},[193,9229,9230],{},"Wrong human."," Whoever is online, or an AI champion spanning domains.",[190,9233,9234,9237],{},[193,9235,9236],{},"Chat as the quote."," Prompt text is not field-level change.",[190,9239,9240,9243],{},[193,9241,9242],{},"HITL as sufficient for the EU AI Act."," Oversight is necessary, not sufficient, for higher-risk systems.",[190,9245,9246,9247,9249],{},"Good looks like: read-only analysis without a click per sentence; quoted writes; named roles; fail-closed execution; rejects stored on the ",[200,9248,3186],{"href":711},"; metrics on reject rates. Failure looks like a prompt, a footer, and a customer who relied on the bot.",[190,9251,9252,9253,9256,9257,9259],{},"The person should sit at ",[193,9254,9255],{},"release",", not at every internal hand-off between ",[200,9258,221],{"href":493},". Internal critics can reduce garbage. They are not the signer.",[190,9261,9262,9263,9265,9266,9268,9269,9271],{},"Adjacent ideas are easy to mix. ",[200,9264,474],{"href":228}," is the fail-closed property of the write. HITL is the human act that satisfies it. A ",[200,9267,3186],{"href":711}," is how you prove the act later. A ",[200,9270,1083],{"href":215}," is whose job the gate belongs to. None of those is a footer on a chatbot.",[190,9273,9274],{},"Fatigue is the operational enemy. If every sentence needs a click, people will click. If only irreversible steps need a click, people can still read. Design the quote so a finance owner can say yes or no in the language of the journal, and a legal owner can say yes or no in the language of the email body. Mixed payloads produce mixed, tired humans.",[255,9276,793],{"id":792},[190,9278,9279],{},"Read-only connectors mean the loop can analyse without a human per sentence.",[190,9281,9282,9283,9286],{},"When a write is proposed, governance ",[193,9284,9285],{},"quotes"," it and stops. Named roles must sign. Agent teams can draft. They cannot waive the gate. Missing approval is fail-closed.",[190,9288,3395,9289,9291],{},[200,9290,23],{"href":711}," stores the human act: who signed, what they saw, what happened next — including rejects. Perception can list rejected items.",[190,9293,3411,9294,9296,9297,230],{},[200,9295,39],{"href":40},". Companion: ",[200,9298,3286],{"href":228},[255,9300,807],{"id":806},[809,9302,9304],{"id":9303},"isnt-this-just-slower-ai","Isn’t this just slower AI?",[190,9306,9307],{},"It is slower than ungoverned writes and faster than incident response. Invented policy is cheaper to catch in a quote than in a tribunal.",[809,9309,9311],{"id":9310},"who-should-be-the-human","Who should be the human?",[190,9313,9314],{},"The owner of the live-system change or the customer commitment, not “whoever is online.” Shared inboxes destroy accountability.",[809,9316,9318],{"id":9317},"does-a-person-at-the-gate-satisfy-eu-ai-law-by-itself","Does a person-at-the-gate satisfy EU AI law by itself?",[190,9320,9321],{},"No. Higher-risk systems have a bundle of duties. Oversight is necessary, not sufficient. Do not claim “Act compliant” because you have a button.",[809,9323,9325],{"id":9324},"how-do-we-stop-rubber-stamping","How do we stop rubber-stamping?",[190,9327,9328],{},"Fewer gates, better quotes, metrics on reject rates. A six-month zero reject rate on CRM writes is a finding: either you are perfect, or nobody is reading.",[809,9330,9332],{"id":9331},"is-a-chat-saying-shall-i-proceed-enough","Is a chat saying “shall I proceed?” enough?",[190,9334,9335],{},"Only if it is bound to identity, shows the payload, and cannot be skipped.",[809,9337,9339],{"id":9338},"what-is-the-difference-between-in-the-loop-and-on-the-loop","What is the difference between in the loop and on the loop?",[190,9341,9342,9343,9346],{},"In the loop: the job cannot finish the risky step without a human act. On the loop: a human ",[270,9344,9345],{},"may"," intervene. Vendors blur them because the second is cheaper to ship.",[809,9348,9350],{"id":9349},"can-agent-teams-approve-each-others-work","Can agent teams approve each other’s work?",[190,9352,9353,9354,230],{},"They can criticise drafts. Release still needs a named human. Multi-agent review is not a signer. See ",[200,9355,494],{"href":493},[809,9357,9359],{"id":9358},"do-read-only-jobs-need-a-person-every-time","Do read-only jobs need a person every time?",[190,9361,9362],{},"Usually not. That is the point of connectors defaulting to read-only. Put people where reversibility changes.",[809,9364,9366],{"id":9365},"how-does-this-relate-to-air-canada","How does this relate to Air Canada?",[190,9368,9369],{},"A customer-facing chatbot made a commitment with no working human catch. The tribunal did not treat the bot as a separate legal person. If your loop can send or display a policy, price, or term, you need a gate or you own the fiction.",[809,9371,9373],{"id":9372},"what-about-mata-v-avianca","What about Mata v. Avianca?",[190,9375,9376],{},"Lawyers filed invented case law from ChatGPT. The failure was the missing check before the record changed. The same pattern waits in CRM and ERP.",[809,9378,9380],{"id":9379},"can-we-batch-approve-200-records","Can we batch-approve 200 records?",[190,9382,9383],{},"Not as one click with no visible set. Bulk without inspection is a rubber stamp with worse radius. Show the set.",[809,9385,9387],{"id":9386},"does-logging-approvals-in-slack-count","Does logging approvals in Slack count?",[190,9389,9390],{},"Only if identity, payload, and outcome are bound and retained as a control record. A thumbs-up emoji is theatre.",[255,9392,869],{"id":868},[190,9394,9395,876,9397,9399,9400,9402],{},[200,9396,3336],{"href":3335},[200,9398,494],{"href":493}," — the person should sit at ",[193,9401,9255],{},", not at every internal hand-off.",[255,9404,883],{"id":882},[260,9406,9407,9412,9418,9424],{},[263,9408,9409],{},[200,9410,2228],{"href":2041,"rel":9411},[204],[263,9413,9414],{},[200,9415,9417],{"href":9048,"rel":9416},[204],"Civil Resolution Tribunal, Moffatt v. Air Canada",[263,9419,9420],{},[200,9421,9423],{"href":599,"rel":9422},[204],"EU AI Act (Regulation 2024/1689), including Article 14",[263,9425,9426],{},[200,9427,9429],{"href":1904,"rel":9428},[204],"Reuters, New York lawyers sanctioned for ChatGPT fake cases",{"title":170,"searchDepth":171,"depth":171,"links":9431},[9432,9433,9437,9438,9452,9453],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":9434},[9435,9436],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":9439},[9440,9441,9442,9443,9444,9445,9446,9447,9448,9449,9450,9451],{"id":9303,"depth":1001,"text":9304},{"id":9310,"depth":1001,"text":9311},{"id":9317,"depth":1001,"text":9318},{"id":9324,"depth":1001,"text":9325},{"id":9331,"depth":1001,"text":9332},{"id":9338,"depth":1001,"text":9339},{"id":9349,"depth":1001,"text":9350},{"id":9358,"depth":1001,"text":9359},{"id":9365,"depth":1001,"text":9366},{"id":9372,"depth":1001,"text":9373},{"id":9379,"depth":1001,"text":9380},{"id":9386,"depth":1001,"text":9387},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Human-in-the-loop AI means a person must approve before the AI can finish the job — seeing the exact change, signing with their identity, and leaving a record.","/blog/what-is-human-in-the-loop-ai",{"title":9024,"description":9454},"blog/what-is-human-in-the-loop-ai",[1606,9459,340,9460],"human-in-the-loop","approvals","LXIbsAoAn6ZbCxvPeO0OjvcsNqJVWTnKQh_OvTdlM20",{"id":9463,"title":9464,"archived":164,"authors":9465,"badge":9467,"body":9468,"date":3101,"definedTerm":165,"department":165,"description":9884,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":9885,"relatedHeading":165,"seo":9886,"series":1606,"sitemap":130,"status":165,"stem":9887,"subhead":165,"tags":9888,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":9889},"content/blog/what-is-institutional-memory-in-enterprise-ai.md","What is Institutional Memory in Enterprise AI",[9466],{"name":184,"to":135},{"label":1024},{"type":167,"value":9469,"toc":9860},[9470,9473,9476,9479,9486,9488,9491,9540,9543,9545,9570,9576,9578,9581,9584,9604,9607,9609,9616,9625,9630,9635,9642,9644,9650,9656,9662,9672,9678,9684,9687,9706,9716,9719,9721,9724,9730,9741,9743,9747,9750,9754,9757,9761,9766,9770,9773,9777,9782,9786,9789,9793,9796,9800,9803,9807,9810,9814,9817,9821,9824,9828,9831,9833,9841,9843],[190,9471,9472],{},"Institutional memory is what the company still knows after the person who did the work leaves — after the chat vendor changes, after the model version rolls.",[190,9474,9475],{},"Individual memory is a hallway conversation and a personal ChatGPT thread. Company memory is playbooks, signed decisions, and live systems, with access control.",[190,9477,9478],{},"Organisations have always had memory: filing cabinets, shared drives, ERP history, “ask the person who was here last year.” Generative AI created a new amnesia: high-value reasoning happens in disposable threads, on personal accounts, in tools with the wrong retention, or in a vendor’s silo the company cannot query.",[190,9480,9481,9482,9485],{},"This is an evidence topic, not a nostalgia topic. Financial reporting changes have needed reconstructable authorisation for decades. Records-management programmes ask for metadata and assigned responsibility. None of those regimes is satisfied by a personal chat thread the predecessor took with them. ",[200,9483,3806],{"href":3547,"rel":9484},[204]," still wants purpose and retention thinking when the “user” is a model.",[255,9487,258],{"id":257},[190,9489,9490],{},"Keep four kinds of memory separate on purpose:",[260,9492,9493,9505,9514,9527],{},[263,9494,9495,9497,9498,9501,9502,9504],{},[193,9496,8047],{}," What we ",[270,9499,9500],{},"want"," to be true: playbooks, guardrails, approved language. See ",[200,9503,2404],{"href":443},". At work, this is the current discount floor, not last year’s slide.",[263,9506,9507,9510,9511,9513],{},[193,9508,9509],{},"Systems of record."," What ",[270,9512,272],{}," true in operations: CRM, ERP, HR, the warehouse. These are memory of the business, not of AI work. At work, the opportunity Amount is here. The reason it changed may not be.",[263,9515,9516,9519,9520,9523,9524,9526],{},[193,9517,9518],{},"Decision memory."," Why we ",[270,9521,9522],{},"changed"," something with AI in the loop: briefs, approvals, rejected options, source versions. A ",[200,9525,3186],{"href":711},". At work, this is “who signed this exception, against which playbook version.”",[263,9528,9529,9532,9533,9536,9537,9539],{},[193,9530,9531],{},"Retrieved knowledge."," Documents we ",[270,9534,9535],{},"might"," use. That is ",[200,9538,3172],{"href":438},". Lookup without policy and decisions is a search engine, not memory.",[190,9541,9542],{},"If you collapse all four into “one vector store” — a database of text fingerprints used to find similar documents — you get sludge that cannot tell policy from a brainstorm. You also get a new store of sensitive data.",[190,9544,5709],{},[260,9546,9547,9553,9559,9564],{},[263,9548,9549,9552],{},[193,9550,9551],{},"Hallway knowledge."," The unofficial version of the rule. It leaves with people. Agents will invent a cousin if it is not asserted.",[263,9554,9555,9558],{},[193,9556,9557],{},"Provider logs."," The vendor’s artefact. Not scoped to your jobs, not your access-controlled ledger.",[263,9560,9561,9563],{},[193,9562,3664],{}," How long a class of record is kept. Completeness is reconstructability, not hoarding.",[263,9565,9566,9569],{},[193,9567,9568],{},"Perception."," Asking that memory in ordinary language, with permissions still applied.",[190,9571,9572,9575],{},[200,9573,9574],{"href":3734},"Causal operations"," is the “why did this change?” slice of decision memory. It is not a claim about market lift.",[255,9577,1160],{"id":1159},[190,9579,9580],{},"It affects you the first Monday after someone leaves, and the first time an auditor asks “why is this exception in the CRM when the playbook still says otherwise?”",[190,9582,9583],{},"Three verbs:",[260,9585,9586,9592,9598],{},[263,9587,9588,9591],{},[193,9589,9590],{},"Assert."," Put the rule into a controlled surface. If it only lives in a slide, agents will invent a cousin.",[263,9593,9594,9597],{},[193,9595,9596],{},"Record."," Store the decision chain when AI is in the loop — not every token, the links that let you reconstruct a change.",[263,9599,9600,9603],{},[193,9601,9602],{},"Ask."," Let the next operator query that memory in ordinary language, with permissions still applied.",[190,9605,9606],{},"Causal operations questions (“why did this change?”) need decision memory. Remember outcomes, quotes, approvals, and citations — not every failed token. Wiki needs owners; memory without freshness is last year’s discount floor.",[809,9608,3290],{"id":3289},[190,9610,9611,9613,9614,230],{},[193,9612,3295],{}," Close packs inherit exceptions. Finance needs the playbook version and the signer, not a rumour that “we always accrue this way.” Provider ChatGPT exports are not a SOX-style trail. Spend history also belongs in memory: which job consumed the units, which run stopped on a cap. See ",[200,9615,3749],{"href":3303},[190,9617,9618,9620,9621,9624],{},[193,9619,3310],{}," Discovery, customer commitments, and erasure. Legal should insist that decision memory points at systems of record rather than duplicating them, and that retention is typed. Infinite chat fails a privacy review. ",[200,9622,3556],{"href":3554,"rel":9623},[204]," erasure is harder if you indexed everything into sludge.",[190,9626,9627,9629],{},[193,9628,3319],{}," Handoffs. The next shift should query “why did this pause?” without reconstructing Slack. Ops should refuse a design that stores every token “because AI” and then cannot delete it.",[190,9631,9632,9634],{},[193,9633,3325],{}," Win/loss reasons and discount exceptions walk out the door with account owners. GTM should put asserted playbooks in the wiki and signed exceptions on the graph — not in a personal Claude project.",[190,9636,9637,9639,9640,230],{},[193,9638,3331],{}," Memory is a sensitive store. Access control on the graph and wiki is as important as on the CRM. Shadow AI is amnesia by design: the work happened on an account the company cannot query. See ",[200,9641,3250],{"href":3249},[809,9643,3340],{"id":3339},[190,9645,9646,9649],{},[193,9647,9648],{},"CRM as sufficient memory."," CRM remembers the current field. It does not remember which playbook version, which AI run, or which person signed the exception.",[190,9651,9652,9655],{},[193,9653,9654],{},"Exporting ChatGPT threads."," Vendor artefact. Wrong scope. Wrong access control.",[190,9657,9658,9661],{},[193,9659,9660],{},"One vector store for everything."," Policy, brainstorms, tickets, and decisions become an undifferentiated similarity soup.",[190,9663,9664,9667,9668,9671],{},[193,9665,9666],{},"A business knowledge graph as a substitute."," That graph models customers and products. Institutional memory for AI work models ",[193,9669,9670],{},"what we did with models"," — and why.",[190,9673,9674,9677],{},[193,9675,9676],{},"Keeping everything forever."," Hoarding is not completeness. It is a privacy and cost failure.",[190,9679,9680,9683],{},[193,9681,9682],{},"Remembering every token."," Reconstruct the change. Do not archive the model’s scratch reasoning by default.",[190,9685,9686],{},"Good looks like four layers kept apart, owners on wiki pages, a lifecycle graph of decisions, permissions on ask, typed retention, and pointers to live systems. Failure looks like a personal thread, a vendor log, and a vector lake.",[190,9688,4359,9689,9691,9692,9694,9695,9697,9698,9701,9702,9705],{},[200,9690,3172],{"href":438}," is lookup, not memory of what we decided. A ",[200,9693,4397],{"href":443}," is asserted policy, which goes stale without owners. A ",[200,9696,3186],{"href":711}," is decision memory of AI-mediated work. ",[200,9699,9700],{"href":3734},"Causal AI for operations"," is the “why did this change?” question that memory should be able to answer. ",[200,9703,9704],{"href":3249},"Shadow AI"," is how memory never starts.",[190,9707,9708,9709,4394,9712,9715],{},"Do not confuse this with a second CRM. Point at the opportunity; do not copy the pipeline. Copies become conflicting official numbers and an erasure problem under ",[200,9710,3556],{"href":3554,"rel":9711},[204],[200,9713,3640],{"href":3638,"rel":9714},[204]," idea — entities, activities, agents — is the right instinct for the decision layer: enough structure to reconstruct, not a lake of tokens.",[190,9717,9718],{},"A Monday-morning test is enough. Can the next operator, with the right permissions, find the playbook version, the signed exception, and the live field — without the predecessor’s laptop? If the answer depends on a personal chat vendor, you do not have institutional memory. You have a coincidence that the person has not left yet.",[255,9720,793],{"id":792},[190,9722,9723],{},"Nimbus combines wiki (asserted policy), connectors (systems of record), the Lifecycle Graph (decision memory), Perception (ask), and workstream scoping (who may see what).",[190,9725,9726,9727,9729],{},"Connectors default to read-only, so analysis can be remembered as ",[270,9728,7688],{}," having written. Named signers and fail-closed writes make refusals part of memory, not missing events.",[190,9731,3869,9732,217,9734,3536,9736,9738,9739,230],{},[200,9733,23],{"href":24},[200,9735,3414],{"href":28},[200,9737,3858],{"href":36},". The job boundary is a ",[200,9740,1083],{"href":215},[255,9742,807],{"id":806},[809,9744,9746],{"id":9745},"isnt-crm-already-our-memory","Isn’t CRM already our memory?",[190,9748,9749],{},"CRM remembers the current field. It does not remember which playbook version, which AI run, or which person signed the exception.",[809,9751,9753],{"id":9752},"can-we-just-export-chatgpt-threads","Can we just export ChatGPT threads?",[190,9755,9756],{},"Provider logs are the vendor’s artefact. They are not scoped to your jobs, and they are not your access-controlled ledger.",[809,9758,9760],{"id":9759},"how-is-this-different-from-a-knowledge-graph-of-customers-and-products","How is this different from a knowledge graph of customers and products?",[190,9762,9763,9764,9671],{},"That graph models the business domain. Institutional memory for AI work models ",[193,9765,9670],{},[809,9767,9769],{"id":9768},"does-this-mean-storing-everything-forever","Does this mean storing everything forever?",[190,9771,9772],{},"No. Retention follows the type of record. Completeness is reconstructability, not hoarding.",[809,9774,9776],{"id":9775},"how-is-this-different-from-enterprise-rag","How is this different from enterprise RAG?",[190,9778,9779,9780,230],{},"RAG retrieves what exists. Memory of work is what we asserted, what we decided, and what the live system holds. Retrieval without those layers is search. See ",[200,9781,3468],{"href":438},[809,9783,9785],{"id":9784},"what-should-we-remember-from-a-run","What should we remember from a run?",[190,9787,9788],{},"The brief, sources (including wiki version), quoted payload, named signer, live-system result, and spend stop if any. Not every failed token, and not secrets in transcripts by default.",[809,9790,9792],{"id":9791},"how-do-we-stop-last-years-policy-living-forever","How do we stop last year’s policy living forever?",[190,9794,9795],{},"Owners and review cadence on the wiki. Archives must not win retrieval against current policy. Freshness is part of memory, not a nice-to-have.",[809,9797,9799],{"id":9798},"can-perception-see-other-departments-decisions","Can Perception see other departments’ decisions?",[190,9801,9802],{},"Only with the same least privilege as the workstream. A go-to-market question should not surface People Ops briefs.",[809,9804,9806],{"id":9805},"is-hallway-knowledge-always-bad","Is hallway knowledge always bad?",[190,9808,9809],{},"It is how work actually happens until you assert it. The failure is leaving it only in hallways once agents are in the loop.",[809,9811,9813],{"id":9812},"how-does-switching-model-vendors-affect-memory","How does switching model vendors affect memory?",[190,9815,9816],{},"If memory lived in the vendor’s chat product, you lost it. If it lived in your wiki, graph, and systems of record, you kept it. That is a buying criterion.",[809,9818,9820],{"id":9819},"where-does-shadow-ai-fit","Where does shadow AI fit?",[190,9822,9823],{},"Personal accounts are institutional amnesia: the company cannot assert, record, or ask. Substitution onto a governed path is how memory starts.",[809,9825,9827],{"id":9826},"do-we-need-a-data-team-to-ask-the-memory","Do we need a data team to ask the memory?",[190,9829,9830],{},"Not if the product has an ordinary-language query surface over the graph and wiki, with permissions. That is Perception in Nimbus. A data team is still right for warehouse metrics.",[255,9832,869],{"id":868},[190,9834,9835,217,9837,3536,9839,230],{},[200,9836,3510],{"href":711},[200,9838,3468],{"href":438},[200,9840,2404],{"href":443},[255,9842,883],{"id":882},[260,9844,9845,9850,9855],{},[263,9846,9847],{},[200,9848,3549],{"href":3547,"rel":9849},[204],[263,9851,9852],{},[200,9853,3556],{"href":3554,"rel":9854},[204],[263,9856,9857],{},[200,9858,4002],{"href":3638,"rel":9859},[204],{"title":170,"searchDepth":171,"depth":171,"links":9861},[9862,9863,9867,9868,9882,9883],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":9864},[9865,9866],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":9869},[9870,9871,9872,9873,9874,9875,9876,9877,9878,9879,9880,9881],{"id":9745,"depth":1001,"text":9746},{"id":9752,"depth":1001,"text":9753},{"id":9759,"depth":1001,"text":9760},{"id":9768,"depth":1001,"text":9769},{"id":9775,"depth":1001,"text":9776},{"id":9784,"depth":1001,"text":9785},{"id":9791,"depth":1001,"text":9792},{"id":9798,"depth":1001,"text":9799},{"id":9805,"depth":1001,"text":9806},{"id":9812,"depth":1001,"text":9813},{"id":9819,"depth":1001,"text":9820},{"id":9826,"depth":1001,"text":9827},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Institutional memory is what the company still knows after the person who did the work leaves: official playbooks, signed decisions, and live systems — with access control.","/blog/what-is-institutional-memory-in-enterprise-ai",{"title":9464,"description":9884},"blog/what-is-institutional-memory-in-enterprise-ai",[1606,4046,4045,326],"YGpBB8KLQ6hQUeeTwqJ1ARCyXSZoB5Y1P_GiTN-wZkM",{"id":9891,"title":1141,"archived":164,"authors":9892,"badge":9894,"body":9895,"date":3101,"definedTerm":165,"department":165,"description":10331,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":10332,"relatedHeading":165,"seo":10333,"series":1606,"sitemap":130,"status":165,"stem":10334,"subhead":165,"tags":10335,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":10338},"content/blog/what-is-model-context-protocol.md",[9893],{"name":184,"to":135},{"label":1024},{"type":167,"value":9896,"toc":10307},[9897,9900,9923,9930,9939,9942,9944,9992,10001,10003,10006,10012,10031,10034,10043,10046,10048,10057,10062,10067,10072,10081,10083,10089,10095,10101,10111,10119,10128,10131,10150,10155,10157,10165,10168,10174,10176,10180,10183,10187,10190,10194,10200,10202,10205,10209,10212,10216,10219,10223,10226,10230,10233,10237,10242,10246,10251,10255,10262,10266,10269,10271,10277,10279],[190,9898,9899],{},"USB did not create a data-governance programme. A common plug let keyboards, cameras, and drives talk to any computer. It did not decide who may copy the finance drive, or whether a change to the ledger needs a signer.",[190,9901,9902,9905,9906,9911,9912,9917,9918,9922],{},[193,9903,9904],{},"Model Context Protocol (MCP)"," is the same kind of open standard for AI. ",[200,9907,9910],{"href":9908,"rel":9909},"https://www.anthropic.com/news/model-context-protocol",[204],"Anthropic announced it"," as a ",[200,9913,9916],{"href":9914,"rel":9915},"https://modelcontextprotocol.io/docs/2026-07-28/getting-started/intro",[204],"common plug"," so AI apps can use the same tools and files, instead of every vendor inventing a one-off connection. The ",[200,9919,9921],{"href":296,"rel":9920},[204],"specification"," standardises how a host calls tools and reads resources.",[190,9924,9925,9926,9929],{},"In one sentence: MCP is ",[193,9927,9928],{},"plumbing, not a company strategy",". It does not decide who may update Salesforce.",[190,9931,9932,9933,9938],{},"Developers already know this pattern from the ",[200,9934,9937],{"href":9935,"rel":9936},"https://microsoft.github.io/language-server-protocol/",[204],"Language Server Protocol",": one language server, many editors, instead of rewriting autocomplete for every IDE. MCP is that idea for tools an AI can call. LSP made language servers interchangeable. It did not make every language server a safe place for customer lists.",[190,9940,9941],{},"Before this standard, every AI product invented its own way to “use a tool.” Teams spent months redoing the same wiring. That cost was real. So is the over-read: “we support MCP” is not “we have enterprise governance.”",[255,9943,258],{"id":257},[260,9945,9946,9952,9958,9964,9970,9976,9984],{},[263,9947,9948,9951],{},[193,9949,9950],{},"Protocol / standard."," Agreed wiring so products can interoperate. At work, this is the USB cable, not the access-control list on the share.",[263,9953,9954,9957],{},[193,9955,9956],{},"MCP server / helper."," A small programme that says “here are the actions I can take, and here are the files I can show you.” At work, a helper that searches Drive as a superuser is still a superuser.",[263,9959,9960,9963],{},[193,9961,9962],{},"Host / client."," The AI application that calls the helper. At work, several hosts can speak MCP and still have completely different write gates — or none.",[263,9965,9966,9969],{},[193,9967,9968],{},"Tool call."," The AI asking that helper to search a folder, look up a ticket, post a message, or query a database.",[263,9971,9972,9975],{},[193,9973,9974],{},"Resource."," A file or record the helper can expose for reading.",[263,9977,9978,9981,9982,230],{},[193,9979,9980],{},"Connector (Nimbus)."," A supported, company-controlled integration to a live system — OAuth, scoped to the job, read-only by default. That is the operator-facing story. MCP may sit at a developer edge. It is not a substitute for connectors plus ",[200,9983,340],{"href":3335},[263,9985,9986,9988,9989,9991],{},[193,9987,4183],{}," Which tools this ",[200,9990,1083],{"href":215}," may call. Importing every available helper is how a demo becomes one actor with every production login.",[190,9993,9994,9995,9998,9999,230],{},"A tool call can still change production data. The protocol will happily pass that change along. The company still has to decide whether that is allowed. Fail-closed writes, named signers, and quoted payloads live ",[270,9996,9997],{},"above"," the plug. See ",[200,10000,3286],{"href":228},[255,10002,1160],{"id":1159},[190,10004,10005],{},"The plug is useful. It is also easy to over-read.",[190,10007,10008,10009,10011],{},"MCP does ",[193,10010,7688],{}," decide:",[260,10013,10014,10017,10020,10023,10026],{},[263,10015,10016],{},"whose login is used",[263,10018,10019],{},"whether the AI may only read, or also change a live system",[263,10021,10022],{},"who must approve a change",[263,10024,10025],{},"how the company remembers what happened",[263,10027,10028,10029],{},"which model is used for the step — see ",[200,10030,3520],{"href":1261},[190,10032,10033],{},"Choosing a model is choosing a brain. This standard is choosing hands. A cheap model with dangerous tools is worse than a strong model with none. Decide them separately.",[190,10035,1029,10036,10038,10039,10042],{},[200,10037,5353],{"href":1954}," that imports every available tool is a confused workflow. Plumbing is not a stop condition. ",[200,10040,4860],{"href":411,"rel":10041},[204]," is about bounding tools and stops, not about collecting helpers.",[190,10044,10045],{},"It affects you if a vendor says “we support MCP” and you hear “we have enterprise governance.” Those are different sentences.",[809,10047,3290],{"id":3289},[190,10049,10050,10052,10053,10056],{},[193,10051,3295],{}," A helper that can post a journal is a write path, protocol or not. Finance should ask whether the host quotes the payload and requires a named signer, not whether the wiring is MCP. Spend also sits above the plug: tool loops can burn ",[200,10054,10055],{"href":3303},"NTUs"," without a ceiling.",[190,10058,10059,10061],{},[193,10060,3310],{}," Processing agreements, purpose, and customer data in helpers running on laptops. Legal should not treat “open standard” as “safe.” A standard plug does not create a DPIA.",[190,10063,10064,10066],{},[193,10065,3319],{}," Bounded tool belts per job. Ops should refuse workflows that attach every helper “for flexibility,” and should keep human wait and budget as stops regardless of how tools are wired.",[190,10068,10069,10071],{},[193,10070,3325],{}," Faster wiring to CRM and Drive can be good — if the connector is still read-only by default. GTM should not confuse a demo that updates an opportunity via MCP with a governed release.",[190,10073,10074,10076,10077,10080],{},[193,10075,3331],{}," This is the sharp edge. Helpers run with some identity. Superuser search is still superuser search. Prompt injection can trick a model into requesting a tool call; the ",[200,10078,977],{"href":375,"rel":10079},[204]," is the relevant list. The protocol will not save you. Least privilege, read-only defaults, and fail-closed writes will.",[809,10082,3340],{"id":3339},[190,10084,10085,10088],{},[193,10086,10087],{},"MCP as governance."," Wiring is not a named signer.",[190,10090,10091,10094],{},[193,10092,10093],{},"MCP as the Salesforce strategy."," You still need identity, read versus write, an approver, and a record.",[190,10096,10097,10100],{},[193,10098,10099],{},"Refusing products that do not speak MCP."," Interoperable tools are a plus. Absence of MCP is not absence of a connector. Presence of MCP is not presence of governance.",[190,10102,10103,10106,10107,10110],{},[193,10104,10105],{},"Replacing the integration platform."," MCP standardises how an AI ",[270,10108,10109],{},"talks"," to a helper. Your identity, iPaaS, and change-control stack still have to exist.",[190,10112,10113,10116,10117,230],{},[193,10114,10115],{},"Assuming retrieval will respect permissions."," Only if the helper is built that way. See ",[200,10118,3468],{"href":438},[190,10120,10121,10124,10125,10127],{},[193,10122,10123],{},"Collecting every server."," A large tool belt is a confused ",[200,10126,5353],{"href":1954}," and a larger attack surface.",[190,10129,10130],{},"Good looks like: MCP where it reduces duplicate wiring; operator-facing connectors that stay scoped, encrypted, and read-only by default; writes only after sign-off; no belief that the spec implemented your control framework. Failure looks like a laptop running a superuser helper pointed at production.",[190,10132,10133,10134,10137,10138,10140,10141,10143,10144,10146,10147,10149],{},"Think of the stack in layers, or you will buy the wrong layer. MCP is how a host talks to a helper. A ",[200,10135,10136],{"href":51},"connector"," is how operators attach a live system to a ",[200,10139,1083],{"href":215}," with OAuth and a read-only default. ",[200,10142,474],{"href":228}," is whether a tool call that mutates production is allowed to execute. The ",[200,10145,3186],{"href":711}," is whether you can still explain the call next quarter. ",[200,10148,1262],{"href":1261}," is which brain issued the call. None of those jobs is in the spec, and that is fine — specs should stay thin. Trouble starts when a thin spec is sold as the thick programme.",[190,10151,1029,10152,10154],{},[200,10153,4372],{"href":4371}," sits above plumbing the way an OS sits above USB: isolation, permissions, I/O policy, and state. USB made accessories interchangeable. It did not decide who may format the finance drive.",[255,10156,793],{"id":792},[190,10158,10159,10160,10162,10163,230],{},"Nimbus’s operator-facing integrations are ",[193,10161,225],{},": scoped per workstream, encrypted per tenant, read-only by default. Action connectors write only after human sign-off. See ",[200,10164,50],{"href":51},[190,10166,10167],{},"MCP can be useful at developer edges. It is not the product’s answer to “who may change CRM.” Governance, wiki, and the Lifecycle Graph still sit above any plug.",[190,10169,3411,10170,876,10172,230],{},[200,10171,39],{"href":40},[200,10173,31],{"href":32},[255,10175,807],{"id":806},[809,10177,10179],{"id":10178},"is-mcp-how-we-should-connect-salesforce","Is MCP how we should connect Salesforce?",[190,10181,10182],{},"Not by itself. You still need identity, read vs write rights, an approver, and a record. A standard plug does not provide those.",[809,10184,10186],{"id":10185},"should-we-refuse-products-that-dont-speak-mcp","Should we refuse products that don’t speak MCP?",[190,10188,10189],{},"No. Interoperable tools are a plus. Absence of MCP is not absence of a connector. Presence of MCP is not presence of governance.",[809,10191,10193],{"id":10192},"does-mcp-replace-our-integration-platform","Does MCP replace our integration platform?",[190,10195,10196,10197,10199],{},"No. It standardises how an AI ",[270,10198,10109],{}," to a helper. Your integration, identity, and change-control stack still has to exist.",[809,10201,8329],{"id":8328},[190,10203,10204],{},"Only if the helper is built that way. A tool that searches Drive as a superuser is still a superuser.",[809,10206,10208],{"id":10207},"is-mcp-the-same-as-a-nimbus-connector","Is MCP the same as a Nimbus connector?",[190,10210,10211],{},"No. A connector is the operator-facing, company-controlled integration: OAuth, workstream scope, read-only default. MCP is a developer wiring standard that might sit at an edge.",[809,10213,10215],{"id":10214},"does-the-spec-require-fail-closed-writes","Does the spec require fail-closed writes?",[190,10217,10218],{},"No. The spec does not require a quoted Salesforce payload, a named approver, or a fail-closed write. Those are product and policy choices.",[809,10220,10222],{"id":10221},"how-does-this-relate-to-usb-and-lsp","How does this relate to USB and LSP?",[190,10224,10225],{},"USB and LSP are the right analogies: interoperability of accessories and language servers. Neither is an access-control programme. Do not buy MCP as if it were.",[809,10227,10229],{"id":10228},"can-we-let-every-agent-team-install-their-own-mcp-servers","Can we let every agent team install their own MCP servers?",[190,10231,10232],{},"That is how you get overlapping write rights and no inventory. Treat helpers like production integrations: owners, scope, and a default of read-only.",[809,10234,10236],{"id":10235},"does-mcp-choose-the-model","Does MCP choose the model?",[190,10238,10239,10240,230],{},"No. Routing is which brain you pay for. MCP is which hands that brain can use. Decide them separately. See ",[200,10241,3520],{"href":1261},[809,10243,10245],{"id":10244},"is-we-support-mcp-a-good-rfp-answer-for-governance","Is “we support MCP” a good RFP answer for governance?",[190,10247,10248,10249,230],{},"It is a good answer for tool interoperability. For governance, ask about quotes, named signers, workstream scope, and the ",[200,10250,3186],{"href":711},[809,10252,10254],{"id":10253},"what-is-the-security-failure-mode","What is the security failure mode?",[190,10256,10257,10258,10261],{},"A helper with broad credentials, a host with no gate, and a model tricked into calling ",[422,10259,10260],{},"update_record",". The protocol did its job. Your company did not.",[809,10263,10265],{"id":10264},"should-customer-facing-bots-get-mcp-tools-to-internal-crm","Should customer-facing bots get MCP tools to internal CRM?",[190,10267,10268],{},"That is how a public conversation inherits production hands. Scope tools as tightly as you would scope a workstream — usually, do not.",[255,10270,869],{"id":868},[190,10272,10273,876,10275,230],{},[200,10274,1955],{"href":1954},[200,10276,3286],{"href":228},[255,10278,883],{"id":882},[260,10280,10281,10287,10292,10297,10302],{},[263,10282,10283],{},[200,10284,10286],{"href":9908,"rel":10285},[204],"Anthropic, Introducing the Model Context Protocol",[263,10288,10289],{},[200,10290,989],{"href":296,"rel":10291},[204],[263,10293,10294],{},[200,10295,9937],{"href":9935,"rel":10296},[204],[263,10298,10299],{},[200,10300,977],{"href":375,"rel":10301},[204],[263,10303,10304],{},[200,10305,923],{"href":411,"rel":10306},[204],{"title":170,"searchDepth":171,"depth":171,"links":10308},[10309,10310,10314,10315,10329,10330],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":10311},[10312,10313],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":10316},[10317,10318,10319,10320,10321,10322,10323,10324,10325,10326,10327,10328],{"id":10178,"depth":1001,"text":10179},{"id":10185,"depth":1001,"text":10186},{"id":10192,"depth":1001,"text":10193},{"id":8328,"depth":1001,"text":8329},{"id":10207,"depth":1001,"text":10208},{"id":10214,"depth":1001,"text":10215},{"id":10221,"depth":1001,"text":10222},{"id":10228,"depth":1001,"text":10229},{"id":10235,"depth":1001,"text":10236},{"id":10244,"depth":1001,"text":10245},{"id":10253,"depth":1001,"text":10254},{"id":10264,"depth":1001,"text":10265},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Model Context Protocol is a common plug so AI apps can use the same tools, like USB for accessories. It is plumbing, not a company strategy, and it does not decide who may update Salesforce.","/blog/what-is-model-context-protocol",{"title":1141,"description":10331},"blog/what-is-model-context-protocol",[1606,10336,225,10337],"mcp","tools","lkEL_If4pUc4pwb2Rsta1gK8p6fIEc3BE-Z43jRbCOs",{"id":10340,"title":10341,"archived":164,"authors":10342,"badge":10344,"body":10345,"date":3101,"definedTerm":165,"department":165,"description":10730,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":10731,"relatedHeading":165,"seo":10732,"series":1606,"sitemap":130,"status":165,"stem":10733,"subhead":165,"tags":10734,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":10736},"content/blog/what-is-model-routing.md","What is Model Routing",[10343],{"name":184,"to":135},{"label":1024},{"type":167,"value":10346,"toc":10706},[10347,10353,10356,10359,10368,10370,10427,10434,10439,10441,10444,10451,10473,10482,10484,10489,10497,10502,10507,10512,10514,10520,10528,10534,10540,10546,10552,10558,10570,10576,10578,10581,10584,10594,10596,10600,10603,10607,10610,10614,10617,10621,10624,10628,10631,10635,10640,10644,10647,10651,10654,10658,10661,10665,10670,10674,10677,10681,10684,10686,10692,10694],[190,10348,10349,10350,10352],{},"Labs ship a ladder of models: small and cheap, large and expensive. ",[193,10351,1262],{}," is the policy above that ladder: use a cheaper, faster model for simple steps, and a stronger model only when the task needs it.",[190,10354,10355],{},"It is not a dropdown labelled “best.” Someone typing “use the best model” for a classify-this-ticket step is how a flagship invoice gets burned on work a compact model could have finished in a second.",[190,10357,10358],{},"Done well, extract runs on compact models and hard reasoning runs on frontier models. Done poorly, every step hits the most expensive model, spend becomes a surprise, and “we use the best model” becomes an unexamined religion.",[190,10360,10361,876,10364,10367],{},[200,10362,4616],{"href":4614,"rel":10363},[204],[200,10365,765],{"href":4619,"rel":10366},[204]," publish those ladders in public. The prices change. The shape does not: input, output, and sometimes tools meter differently, and the top rung is many times the compact rung. Finance cannot treat “always flagship” as a quality culture. It is an unbudgeted preference.",[255,10369,258],{"id":257},[260,10371,10372,10378,10384,10390,10396,10405,10414,10420],{},[263,10373,10374,10377],{},[193,10375,10376],{},"Frontier / flagship model."," The strongest (and usually most expensive) model a lab currently sells. At work, this is for judgment: does this clause violate the playbook?",[263,10379,10380,10383],{},[193,10381,10382],{},"Compact / small model."," Faster and cheaper. Often enough for extract, classify, and summarise. At work, this is “pull the fields from the export.”",[263,10385,10386,10389],{},[193,10387,10388],{},"Cascade."," Try cheap first; spend the expensive call only when the cheap one is not enough.",[263,10391,10392,10395],{},[193,10393,10394],{},"Fallback."," If a provider is down or over budget, send the step somewhere else.",[263,10397,10398,10400,10401,10404],{},[193,10399,5340],{}," What steps exist. Different from routing, which is ",[270,10402,10403],{},"which brain"," each step uses. You can orchestrate a brilliant multi-agent graph and still send every node to the flagship.",[263,10406,10407,10410,10411,10413],{},[193,10408,10409],{},"Quality bar."," The reject-rate or rework threshold that decides whether a compact model is good enough on ",[270,10412,4187],{}," job.",[263,10415,10416,10419],{},[193,10417,10418],{},"Data residency / data class."," A cheap endpoint may be forbidden for a class of records. Routing is then a compliance table, not only a cost table.",[263,10421,10422,10424,10425,230],{},[193,10423,7167],{}," The normalised unit routing is trying to protect. See ",[200,10426,3749],{"href":3303},[190,10428,10429,10430,10433],{},"Routing is also not ",[193,10431,10432],{},"fine-tuning"," (changing a model’s weights). Fine-tuning is a research and ops programme. Routing is an operating policy over models you already buy.",[190,10435,10436,10437,230],{},"Constraints that belong in the route table: data residency, evaluation (you cannot route on vibes), and security (a model with web tools is a different actor than a model with none). Choosing a model is choosing a brain. Choosing tools is choosing hands. Decide them separately. See ",[200,10438,1141],{"href":757},[255,10440,1160],{"id":1159},[190,10442,10443],{},"It affects you if you pay the bill, or if quality on a step is load-bearing.",[190,10445,10446,10447,10450],{},"Talk about it as a ",[193,10448,10449],{},"budget and quality conversation",", not as an ML research project:",[260,10452,10453,10459,10465],{},[263,10454,10455,10458],{},[193,10456,10457],{},"Tag the steps."," Extracting fields from an export is not the same as arguing whether a clause violates policy. If your platform cannot name steps, it cannot route them.",[263,10460,10461,10464],{},[193,10462,10463],{},"Set a quality bar per step."," “Compact model until human reject rate exceeds X on this job.” Without a bar, routing becomes “always escalate because someone was once unhappy.”",[263,10466,10467,10470,10471,230],{},[193,10468,10469],{},"Keep the gate regardless of model."," A cheap model with a write tool is still a write tool. See ",[200,10472,3286],{"href":228},[190,10474,10475,10476,10478,10479,10481],{},"A spend ceiling without routing still lets every step hit the flagship until the ceiling kills the run. Routing is how you stay under the ceiling ",[270,10477,9165],{}," finish the job. See ",[200,10480,1955],{"href":1954}," for why loops without stops dominate the bill.",[809,10483,3290],{"id":3289},[190,10485,10486,10488],{},[193,10487,3295],{}," Routing is the practical lever on unit cost. Quotes should assume the policy, not the flagship. Finance should ask for approved-updates per NTU, and for evidence that extract steps are not on the top rung. Locking one vendor forever is a pricing and outage choice; routing across providers is a second tape measure.",[190,10490,10491,10493,10494,10496],{},[193,10492,3310],{}," Data class and residency can forbid the cheap endpoint. Legal should sit on the route table for those classes, not discover them on an invoice. Customer-facing language may need a stronger model ",[270,10495,9165],{}," a named signer; routing does not replace the gate.",[190,10498,10499,10501],{},[193,10500,3319],{}," Steps must be named or you cannot route them. Ops should own fallbacks when a provider is down, and should refuse a single “best” toggle that bypasses the table.",[190,10503,10504,10506],{},[193,10505,3325],{}," Quality anxiety is strongest here. Measure reject rates on the job. A compact model that extracts next steps may be fine; a compact model that invents a concession is not. Routing on one unhappy anecdote will pin every step to flagship.",[190,10508,10509,10511],{},[193,10510,3331],{}," A model with browsing or unconstrained tools is a different actor. Routing should not silently add hands. Prompt injection plus a flagship model plus write tools is a worse combination than a compact extract-only step behind a fail-closed gate.",[809,10513,3340],{"id":3339},[190,10515,10516,10519],{},[193,10517,10518],{},"Always the smartest model."," Use the weakest model that meets the quality bar for that step. Flagship is for judgment, not for labelling.",[190,10521,10522,10525,10526,230],{},[193,10523,10524],{},"Routing as multi-agent."," Several agents is a cast. Routing is which brain each step pays for. See ",[200,10527,494],{"href":493},[190,10529,10530,10533],{},[193,10531,10532],{},"Routing as fine-tuning."," Different programme.",[190,10535,10536,10539],{},[193,10537,10538],{},"Dropdown labelled “best.”"," That is not a policy. It is a preference that cannot be audited.",[190,10541,10542,10545],{},[193,10543,10544],{},"Dropping the write gate for a “trusted” model."," Trust the gate. Models change weekly.",[190,10547,10548,10551],{},[193,10549,10550],{},"Routing on vibes."," One anecdote becomes a permanent escalate. Measure rework.",[190,10553,10554,10555,10557],{},"Good looks like: named steps, a route table with cost, quality bar, and data class, cascade where it helps, fallback across providers, gates independent of model, falling unit cost as the ",[200,10556,326],{"href":443}," reduces re-derivation. Failure looks like flagship-everywhere and a board slide about the bill.",[190,10559,10560,10561,10563,10564,10566,10567,10569],{},"Adjacent ideas worth keeping separate: ",[200,10562,3304],{"href":3303}," is quote, cap, and attribute. Routing is which rung of the ladder a named step is allowed to use. ",[200,10565,5344],{"href":493}," is how many specialist roles run. You can route a single agent, and you can send a whole agent team to the flagship by mistake. ",[200,10568,758],{"href":757}," is hands, not brains: do not let a compact extract step inherit a write tool because “the helper was available.”",[190,10571,10572,10573,10575],{},"Evaluation has to live on the job, not in a model-arena screenshot. A compact model that extracts fields with a low reject rate is a success even if it would lose a public chatbot bake-off. A flagship model that drafts a concession the wiki forbids is a failure even if it is eloquent. Tie routing reviews to ",[200,10574,1083],{"href":215}," outcomes — approved writes, rejects, rework — the same way you would review any other operating policy.",[255,10577,793],{"id":792},[190,10579,10580],{},"Nimbus treats routing as an operating decision tied to workstream steps: task type, sensitivity, and cost — not “best everywhere.” Release gates apply regardless of which model drafted the payload.",[190,10582,10583],{},"NTU quotes and ceilings sit around that policy so operators see a number before they commit. Everyday extract should not consume flagship credits.",[190,10585,3411,10586,10588,10589,10591,10592,230],{},[200,10587,44],{"href":45},". For the unit of account routing sits inside, ",[200,10590,3749],{"href":3303},". Product: ",[200,10593,31],{"href":32},[255,10595,807],{"id":806},[809,10597,10599],{"id":10598},"should-we-always-use-the-smartest-model","Should we always use the smartest model?",[190,10601,10602],{},"No. Use the weakest model that meets the quality bar for that step. Flagship is for judgment, not for labelling.",[809,10604,10606],{"id":10605},"will-routing-make-answers-worse","Will routing make answers worse?",[190,10608,10609],{},"It can, if you under-route hard steps. Measure rejects and rework on the job. Do not route on a single anecdote.",[809,10611,10613],{"id":10612},"is-this-the-same-as-having-several-agents","Is this the same as having several agents?",[190,10615,10616],{},"No. Several agents is a cast. Routing is which brain each step pays for.",[809,10618,10620],{"id":10619},"can-we-lock-one-vendor-forever","Can we lock one vendor forever?",[190,10622,10623],{},"You can. You will pay for it in price, outages, and lock-in. Routing across providers is how finance keeps a second tape measure.",[809,10625,10627],{"id":10626},"what-is-a-cascade","What is a cascade?",[190,10629,10630],{},"Try the cheap model first. Escalate only when a confidence or quality check says the cheap pass is not enough. It is a tactic inside a policy, not a substitute for naming steps.",[809,10632,10634],{"id":10633},"does-a-better-model-remove-the-need-for-a-wiki","Does a better model remove the need for a wiki?",[190,10636,10637,10638,230],{},"No. Stronger models are better at sounding like policy. Asserted playbooks still win over Drive folklore. See ",[200,10639,2404],{"href":443},[809,10641,10643],{"id":10642},"how-do-we-set-a-quality-bar","How do we set a quality bar?",[190,10645,10646],{},"Start with human reject rate and rework on that step. “Compact until rejects exceed X on this workstream” is a bar. “People like the flagship” is not.",[809,10648,10650],{"id":10649},"should-customer-facing-copy-always-use-the-flagship","Should customer-facing copy always use the flagship?",[190,10652,10653],{},"Not always. It should always use a human gate if it asserts a term or a price. Model size does not absorb Air Canada-style risk.",[809,10655,10657],{"id":10656},"what-if-the-cheap-endpoint-is-in-the-wrong-region","What if the cheap endpoint is in the wrong region?",[190,10659,10660],{},"Then it is not cheap; it is forbidden. Put residency in the route table beside price.",[809,10662,10664],{"id":10663},"does-routing-replace-spend-caps","Does routing replace spend caps?",[190,10666,10667,10668,230],{},"No. Caps stop unbounded loops. Routing makes legitimate work affordable under the cap. You want both. See ",[200,10669,3749],{"href":3303},[809,10671,10673],{"id":10672},"can-the-model-choose-its-own-successor","Can the model choose its own successor?",[190,10675,10676],{},"Letting the model always escalate is how every step becomes flagship. Escalation should be a policy check, not a preference the model expresses.",[809,10678,10680],{"id":10679},"how-does-this-show-up-in-an-rfp","How does this show up in an RFP?",[190,10682,10683],{},"Ask whether steps are named, whether gates apply regardless of model, and whether finance sees a normalised unit. “We use the best models” is not an answer.",[255,10685,869],{"id":868},[190,10687,10688,876,10690,230],{},[200,10689,3749],{"href":3303},[200,10691,494],{"href":493},[255,10693,883],{"id":882},[260,10695,10696,10701],{},[263,10697,10698],{},[200,10699,5002],{"href":4614,"rel":10700},[204],[263,10702,10703],{},[200,10704,5008],{"href":4619,"rel":10705},[204],{"title":170,"searchDepth":171,"depth":171,"links":10707},[10708,10709,10713,10714,10728,10729],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":10710},[10711,10712],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":10715},[10716,10717,10718,10719,10720,10721,10722,10723,10724,10725,10726,10727],{"id":10598,"depth":1001,"text":10599},{"id":10605,"depth":1001,"text":10606},{"id":10612,"depth":1001,"text":10613},{"id":10619,"depth":1001,"text":10620},{"id":10626,"depth":1001,"text":10627},{"id":10633,"depth":1001,"text":10634},{"id":10642,"depth":1001,"text":10643},{"id":10649,"depth":1001,"text":10650},{"id":10656,"depth":1001,"text":10657},{"id":10663,"depth":1001,"text":10664},{"id":10672,"depth":1001,"text":10673},{"id":10679,"depth":1001,"text":10680},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Model routing is using a cheaper, faster model for simple steps and a stronger model only when the task needs it — a policy, not a dropdown labelled “best.”","/blog/what-is-model-routing",{"title":10341,"description":10730},"blog/what-is-model-routing",[1606,5045,5043,10735],"cost","K7yUQFVeuwQEHFO9maOLrChccWdVbLG9XKu9GuGsvB4",{"id":10738,"title":10739,"archived":164,"authors":10740,"badge":10742,"body":10743,"date":3101,"definedTerm":165,"department":165,"description":11146,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":11147,"relatedHeading":165,"seo":11148,"series":1606,"sitemap":130,"status":165,"stem":11149,"subhead":165,"tags":11150,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":11153},"content/blog/what-is-multi-agent-ai.md","What is Multi-Agent AI",[10741],{"name":184,"to":135},{"label":1024},{"type":167,"value":10744,"toc":11122},[10745,10748,10751,10761,10763,10817,10831,10836,10838,10841,10843,10854,10857,10877,10882,10888,10890,10895,10903,10908,10913,10918,10920,10926,10932,10938,10944,10950,10956,10959,10965,10978,10984,10989,10991,11000,11003,11013,11015,11019,11022,11026,11029,11033,11038,11042,11045,11049,11052,11056,11059,11063,11068,11072,11075,11079,11082,11086,11091,11095,11098,11102,11105,11107,11113,11115],[190,10746,10747],{},"Multi-agent AI is more than one AI specialist handing work to each other, the way legal already reviews a go-to-market draft. They share a goal, pass intermediate work, and stop for a human when duties collide.",[190,10749,10750],{},"It is not “more than one model call.” If you can rename your agents to “saved prompts” and nothing breaks, you do not have multi-agent AI. You have prompt folders. Three chat tabs labelled Research, CRM, and Legal are still one person copying between windows.",[190,10752,10753,10754,10756,10757,10760],{},"Companies already hand work between departments. Multi-agent AI is useful when those hand-offs ",[270,10755,8610],{}," the job. It is not useful as a prestige multiplier on a task one specialist should finish. ",[200,10758,4860],{"href":411,"rel":10759},[204]," is mostly about workflows and stops, not about collecting a zoo of bots.",[255,10762,258],{"id":257},[260,10764,10765,10771,10777,10783,10788,10794,10802,10808],{},[263,10766,10767,10770],{},[193,10768,10769],{},"Single-agent."," One policy, one tool set, one conversation. The human is the only coordinator. Right for many tasks: rewrite this email, explain this clause.",[263,10772,10773,10776],{},[193,10774,10775],{},"Multi-agent."," Role specialisation, shared state scoped to the job, arbitration when agents disagree, and a stop — including “wait for approval.”",[263,10778,10779,10782],{},[193,10780,10781],{},"Orchestrator."," A coordinator that assigns work to specialists. Not a licence to give every specialist the same production login.",[263,10784,10785,10787],{},[193,10786,6139],{}," Nimbus’s product name for a department-shaped specialist group (finance, revenue, operations) that persists, rather than a zoo of user-owned bots.",[263,10789,10790,10793],{},[193,10791,10792],{},"Separation of duties."," The specialist that recommends a CRM update is not the same principal that executes it without a quote.",[263,10795,10796,10799,10800,230],{},[193,10797,10798],{},"Shared state."," The job’s brief, wiki sections, and artefacts — not a pile of private chats. At work, this is the ",[200,10801,1083],{"href":215},[263,10803,10804,10807],{},[193,10805,10806],{},"Arbitration."," What happens when specialists disagree. At work, legal’s “do not send” should beat go-to-market’s “looks fine,” and a named human still releases.",[263,10809,10810,10813,10814,10816],{},[193,10811,10812],{},"Cast versus brain."," Several agents is a cast. ",[200,10815,1262],{"href":1261}," is which brain each step pays for.",[190,10818,10819,10820,10823,10824,10827,10828,10830],{},"Tools are ",[193,10821,10822],{},"hands",". They are not ",[193,10825,10826],{},"roles",". A common plug so AI apps can use the same tools is useful plumbing — see ",[200,10829,1141],{"href":757}," — and it is also how a “multi-agent” demo quietly becomes one actor with every tool on the belt.",[190,10832,1029,10833,10835],{},[200,10834,5353],{"href":1954}," is the sequence. Multi-agent AI is the cast. Mixing those words is how vendors sell extra model calls as organisation design.",[255,10837,1160],{"id":1159},[190,10839,10840],{},"Coordination is an org-chart problem, not a model problem.",[190,10842,1171],{},[260,10844,10845,10848,10851],{},[263,10846,10847],{},"one “god agent” would need every production login",[263,10849,10850],{},"legal must review a draft before anyone writes CRM",[263,10852,10853],{},"next quarter, “why did we change this?” must still be answerable",[190,10855,10856],{},"Known failure modes:",[547,10858,10859,10865,10871],{},[263,10860,10861,10864],{},[193,10862,10863],{},"Parallel single-agent."," Three chat tabs. No shared state. The user is the message bus.",[263,10866,10867,10870],{},[193,10868,10869],{},"Agent sprawl."," Dozens of custom agents with overlapping tools and unclear write rights.",[263,10872,10873,10876],{},[193,10874,10875],{},"Orchestration without memory."," A beautiful run that discards the outcome when the worker exits.",[190,10878,10879,10880,230],{},"Multi-agent does not reduce accountability. It concentrates it on the release gate. See ",[200,10881,3923],{"href":1896},[190,10883,10884,10885,230],{},"Do not think in “number of agents.” Think in ",[193,10886,10887],{},"jobs that already have hand-offs",[809,10889,3290],{"id":3289},[190,10891,10892,10894],{},[193,10893,3295],{}," A finance-shaped agent team can draft a journal against the close checklist without inheriting GTM’s CRM write connector. Separation of duties is the point. Finance should still be the named signer on the ledger. Extra agents are not extra authorisation.",[190,10896,10897,10899,10900,10902],{},[193,10898,3310],{}," Review-before-send is a real hand-off. Legal-shaped specialists should not need People Ops files “for context.” Legal also cares that internal agent debate is not treated as a signature. The ",[200,10901,3186],{"href":711}," should show the human at release.",[190,10904,10905,10907],{},[193,10906,3319],{}," Persist teams, do not spawn a bot per user. Ops should refuse sprawl, insist on shared workstream state, and keep fail-closed writes outside the cast. Incident reviews need one chain, not three private transcripts.",[190,10909,10910,10912],{},[193,10911,3325],{}," Cross-functional launches already look like this: GTM drafts, legal redlines, finance checks the discount. Encode that. Do not encode a god agent that can do all three logins. Time-to-approved-write still beats number-of-agents as a metric.",[190,10914,10915,10917],{},[193,10916,3331],{}," Sprawl is an identity problem. Each specialist with overlapping write tools is another path to production. Prompt injection that turns one specialist into a tool-caller should still die at the gate. Least privilege applies per role, not “the swarm is trusted.”",[809,10919,3340],{"id":3339},[190,10921,10922,10925],{},[193,10923,10924],{},"Saved prompts as agents."," If renaming them changes nothing, they were prompts.",[190,10927,10928,10931],{},[193,10929,10930],{},"Chat tabs as multi-agent."," The user is still the bus.",[190,10933,10934,10937],{},[193,10935,10936],{},"More agents as more quality."," Coordination cost is real. Start from existing hand-offs.",[190,10939,10940,10943],{},[193,10941,10942],{},"Agents as signers."," Internal critics reduce garbage. They are not the named human.",[190,10945,10946,10949],{},[193,10947,10948],{},"Every specialist gets every tool."," That recreates the god agent with extra steps.",[190,10951,10952,10955],{},[193,10953,10954],{},"Orchestration without a workstream."," No scope, no budget, no memory.",[190,10957,10958],{},"Good looks like: department-shaped teams that persist, inherit workstream scope (wiki, connectors, NTU budget), disagree in the open, and stop for a named signer. Failure looks like a folder of user-owned bots and a demo where five helpers share one production key.",[190,10960,10961,10962,10964],{},"A useful test: draw the analogue hand-off first. If legal already reviews a go-to-market draft before a customer sees it, you have a candidate for two specialist roles on one ",[200,10963,1083],{"href":215},". If one analyst extracts a table, you have a candidate for a single tool-using agent. If nobody can name the hand-off, you are inventing a cast for a play that does not exist — and you will invent overlapping tools to keep them busy.",[190,10966,10967,10968,10970,10971,10974,10975,10977],{},"Spend follows the same test. Extra specialists mean extra model calls. Without ",[200,10969,5360],{"href":1261}," and an ",[200,10972,10973],{"href":3303},"NTU"," ceiling, “let them debate” is an unbounded loop. Debate that never reaches a named signer is also not ",[200,10976,9459],{"href":1896},"; it is theatre with more speakers.",[190,10979,10980,10981,10983],{},"Memory is the other test. If the hand-off is not on the ",[200,10982,3186],{"href":711},", next quarter’s question — “why did we change this?” — has no answer except whoever still remembers the swarm. That is not multi-agent AI. That is parallel chat.",[190,10985,10986,10988],{},[200,10987,685],{"href":20}," in Nimbus are meant to look like the departments you already have, not like a prompt gallery. If your org chart does not contain a role, do not invent an agent for it. If your org chart does contain a role that must review before release, do not skip it because a single flagship model offered to “do it all.” Number of agents is a vanity metric. Named hand-offs are not.",[255,10990,793],{"id":792},[190,10992,10993,10994,10996,10997,10999],{},"Nimbus implements multi-agent AI as ",[193,10995,221],{},", not as a folder of user-owned bots. Teams persist. They inherit ",[200,10998,1083],{"href":215}," scope — wiki sections, connectors, spend budget — and they participate in the same release process as any other actor.",[190,11001,11002],{},"Connectors stay read-only by default. Agent teams can draft. They cannot waive the gate. The Lifecycle Graph records the hand-offs as work, not as a swarm mystery.",[190,11004,3411,11005,11007,11008,11010,11011,230],{},[200,11006,685],{"href":20},". An ",[200,11009,5353],{"href":1954}," is the sequence. Multi-agent AI is the cast. The OS around them is ",[200,11012,4519],{"href":4371},[255,11014,807],{"id":806},[809,11016,11018],{"id":11017},"isnt-this-just-several-chatgpts-talking","Isn’t this just several ChatGPTs talking?",[190,11020,11021],{},"Not if they share one job, one scope, and one stop. Several chats with no shared state is still you, copying.",[809,11023,11025],{"id":11024},"do-we-need-multi-agent-ai-for-everything","Do we need multi-agent AI for everything?",[190,11027,11028],{},"No. A single tool-using agent is enough for many tasks. Add specialists when duties already split in the organisation.",[809,11030,11032],{"id":11031},"does-each-agent-need-its-own-model","Does each agent need its own model?",[190,11034,11035,11036,230],{},"Often yes for cost and quality. Classification rarely needs the flagship. Tricky policy interpretation often does. See ",[200,11037,3520],{"href":1261},[809,11039,11041],{"id":11040},"who-is-accountable-when-several-agents-worked-on-it","Who is accountable when several agents worked on it?",[190,11043,11044],{},"The named human at release — not “the swarm.” Internal critics can reduce garbage that reaches the person. They are not the signer.",[809,11046,11048],{"id":11047},"how-is-this-different-from-an-agentic-workflow","How is this different from an agentic workflow?",[190,11050,11051],{},"The workflow is the sequence of steps and stops. Multi-agent is whether more than one specialist role executes those steps. You can have a workflow with one agent.",[809,11053,11055],{"id":11054},"what-is-an-agent-team-in-nimbus","What is an agent team in Nimbus?",[190,11057,11058],{},"A department-shaped specialist group that persists and inherits the workstream’s wiki, connectors, and budget — not a user-owned custom GPT.",[809,11060,11062],{"id":11061},"can-agents-approve-each-others-writes","Can agents approve each other’s writes?",[190,11064,11065,11066,230],{},"They can flag problems. Execution still needs a named signer and a fail-closed gate. See ",[200,11067,3286],{"href":228},[809,11069,11071],{"id":11070},"why-not-one-god-agent-with-every-connector","Why not one god agent with every connector?",[190,11073,11074],{},"Because least privilege and separation of duties already exist in the company. Encoding the org chart is safer than encoding a superuser.",[809,11076,11078],{"id":11077},"how-do-we-avoid-agent-sprawl","How do we avoid agent sprawl?",[190,11080,11081],{},"One team per function that already exists, assigned onto jobs, with overlapping tools treated as an incident. Do not let every operator publish a bot.",[809,11083,11085],{"id":11084},"does-mcp-make-us-multi-agent","Does MCP make us multi-agent?",[190,11087,11088,11089,230],{},"No. MCP is how a host calls tools. Many helpers on one belt can still be one actor. See ",[200,11090,1141],{"href":757},[809,11092,11094],{"id":11093},"how-do-disagreements-get-recorded","How do disagreements get recorded?",[190,11096,11097],{},"On the decision chain: what was proposed, what was objected to, what the human signed. If disagreement evaporates with the session, you have orchestration without memory.",[809,11099,11101],{"id":11100},"will-more-agents-stop-hallucinations","Will more agents stop hallucinations?",[190,11103,11104],{},"They can catch some errors the way a second reader can. They do not replace asserted wiki, citations, or a person on commitments. Air Canada-style fiction is a gate problem, not a cast-size problem.",[255,11106,869],{"id":868},[190,11108,11109,876,11111,230],{},[200,11110,4992],{"href":215},[200,11112,4519],{"href":4371},[255,11114,883],{"id":882},[260,11116,11117],{},[263,11118,11119],{},[200,11120,923],{"href":411,"rel":11121},[204],{"title":170,"searchDepth":171,"depth":171,"links":11123},[11124,11125,11129,11130,11144,11145],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":11126},[11127,11128],{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":11131},[11132,11133,11134,11135,11136,11137,11138,11139,11140,11141,11142,11143],{"id":11017,"depth":1001,"text":11018},{"id":11024,"depth":1001,"text":11025},{"id":11031,"depth":1001,"text":11032},{"id":11040,"depth":1001,"text":11041},{"id":11047,"depth":1001,"text":11048},{"id":11054,"depth":1001,"text":11055},{"id":11061,"depth":1001,"text":11062},{"id":11070,"depth":1001,"text":11071},{"id":11077,"depth":1001,"text":11078},{"id":11084,"depth":1001,"text":11085},{"id":11093,"depth":1001,"text":11094},{"id":11100,"depth":1001,"text":11101},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Multi-agent AI is more than one AI specialist handing work to each other — the way legal already reviews a go-to-market draft — with a shared job, a stop, and a person who must approve before a live system changes.","/blog/what-is-multi-agent-ai",{"title":10739,"description":11146},"blog/what-is-multi-agent-ai",[1606,11151,11152,2892],"multi-agent","agent-teams","19f70xeEoO2YudV92-7To-KyrfQ5x2mD0IDacUDBGbE",{"id":11155,"title":11156,"archived":164,"authors":11157,"badge":11159,"body":11160,"date":11333,"definedTerm":11334,"department":165,"description":11335,"extension":173,"eyebrow":165,"faqHeader":11336,"faqs":11338,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":11348,"relatedHeading":165,"seo":11349,"series":1606,"sitemap":130,"status":165,"stem":11350,"subhead":165,"tags":11351,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":11353},"content/blog/rbac-for-enterprise-ai.md","What is RBAC for enterprise AI, and why should you care?",[11158],{"name":184,"to":135},{"label":1024},{"type":167,"value":11161,"toc":11327},[11162,11165,11168,11174,11178,11187,11198,11201,11215,11218,11222,11231,11234,11246,11249,11253,11256,11259,11279,11284,11288,11295,11307,11315],[190,11163,11164],{},"RBAC means role-based access control: who is allowed to do what. For enterprise AI, the “who” is not only people. It is also the model acting with someone’s credentials — reading files, and sometimes changing a live system.",[190,11166,11167],{},"You should care because a fluent answer can still be the wrong change in the wrong place. Access rules are how you keep AI useful without pretending every user should see every record.",[190,11169,11170,11171,11173],{},"This guide explains the idea, why it shows up in vendor conversations, and a practical way to start. It is not a claim that one product has solved it. ",[200,11172,3336],{"href":3335}," is the parent definition.",[255,11175,11177],{"id":11176},"what-is-rbac-for-enterprise-ai","What is RBAC for enterprise AI?",[190,11179,11180,11181,11186],{},"Classic RBAC, described by Ferraiolo and Kuhn in a ",[200,11182,11185],{"href":11183,"rel":11184},"https://csrc.nist.gov/files/pubs/conference/1992/10/13/rolebased-access-controls/final/docs/ferraiolo-kuhn-92.pdf",[204],"NIST paper"," (1992), assigns permissions to roles, then roles to people. Enterprise AI adds three extra questions:",[260,11188,11189,11192,11195],{},[263,11190,11191],{},"Which jobs and files can this person (and this model) see?",[263,11193,11194],{},"Which tools can it call?",[263,11196,11197],{},"If it can change a live system, who must approve, and is that approval stored?",[190,11199,11200],{},"A chatbot login answers “may this person talk to the bot?” That is necessary. It is not the same as answering the three questions above.",[190,11202,11203,11204,11208,11209,11214],{},"NIST’s ",[200,11205,11207],{"href":363,"rel":11206},[204],"AI Risk Management Framework"," (2023) and ",[200,11210,11213],{"href":11211,"rel":11212},"https://csrc.nist.gov/pubs/sp/800-207/final",[204],"SP 800-207"," (2020) on zero trust are the public-sector language for the same idea: do not assume a session is trusted just because it authenticated.",[190,11216,11217],{},"Guests, members, and admins are the people side of the same idea: who is on the job. The model side is which tools that session may call. Both belong in RBAC. Do not treat a chatbot login as the whole answer.",[255,11219,11221],{"id":11220},"why-should-you-care-about-rbac-for-ai","Why should you care about RBAC for AI?",[190,11223,11224,11225,11230],{},"IBM’s ",[200,11226,11229],{"href":11227,"rel":11228},"https://newsroom.ibm.com/2024-07-30-ibm-report-escalating-data-breach-disruption-pushes-costs-to-new-highs",[204],"Cost of a Data Breach"," report (2024) put the global average breach cost at $4.88 million. You do not need a breach for RBAC to matter. You need a customer record changed without a name next to the change, or a contractor who still sees a workstream after the project ended.",[190,11232,11233],{},"A simple example: a guest from an agency is invited to a campaign workstream. The model in that room can read the CRM export because a member pasted it. When the campaign ends, the guest login is forgotten. The export is still in the history. Roles that follow the job — not only the person — are how you close that gap.",[190,11235,11236,11237,11242,11243,11245],{},"Microsoft and LinkedIn’s ",[200,11238,11241],{"href":11239,"rel":11240},"https://www.microsoft.com/en-us/worklab/work-trend-index/ai-at-work-is-here-now-comes-the-hard-part",[204],"Work Trend Index"," (2024) found that 78% of AI users bring their own tools (BYOAI). That is ",[200,11244,4279],{"href":3249},": useful, and outside the roles you think you assigned.",[190,11247,11248],{},"You should care if you have guests on a job, if AI can write to CRM or finance systems, or if an auditor might ask who approved a machine-initiated change. If AI only summarises public wiki pages, the stakes are lower — you can still use roles so the wiki is not everyone’s dump of customer data.",[255,11250,11252],{"id":11251},"how-do-you-apply-it-when-ai-can-change-records","How do you apply it when AI can change records?",[190,11254,11255],{},"Write-back means the AI changes a live system. Fail-closed means if nobody approves, nothing happens. Payload means the exact change, shown before it goes out.",[190,11257,11258],{},"A practical sequence:",[547,11260,11261,11267,11270,11276],{},[263,11262,11263,11264,11266],{},"Keep the model from writing until you can name the object class and the signer. ",[200,11265,474],{"href":228}," is the checklist.",[263,11268,11269],{},"For each write, name the approver role — not “the channel”.",[263,11271,11272,11273,11275],{},"Store the payload and the decision so you can reopen them. ",[200,11274,3036],{"href":3035}," is the evidence pack.",[263,11277,11278],{},"When someone leaves the job, remove them from the roster the same week.",[190,11280,11281,11283],{},[200,11282,9704],{"href":3249}," is what happens when the unofficial path never got those roles.",[255,11285,11287],{"id":11286},"what-should-you-ask-a-vendor","What should you ask a vendor?",[190,11289,11290,11291,11294],{},"A short list of demo questions lives in ",[200,11292,11293],{"href":3035},"what auditors are asking for",". In one sentence: can they show who could see a job, which tool ran, and who approved a write — without a screenshot hunt?",[190,11296,3395,11297,11301,11302,11306],{},[200,11298,601],{"href":11299,"rel":11300},"https://eur-lex.europa.eu/legal-content/EN/TXT/?uri=CELEX:32024R1689",[204]," (2024/1689) and ",[200,11303,966],{"href":11304,"rel":11305},"https://www.iso.org/standard/81230.html",[204]," are reasons those questions are showing up in procurement. You do not have to implement every clause on day one. You do need an answer you could give an auditor.",[190,11308,2603,11309,876,11311,11314],{},[200,11310,340],{"href":40},[200,11312,11313],{"href":54},"security"," pages describe how we approach this. Other vendors will have their own. The useful test is the same: roles on the job, not only on the chat login.",[190,11316,11317,11318,11322,11323,230],{},"For how teams share the job once access is clear, see ",[200,11319,11321],{"href":11320},"what-is-collaborative-ai","what is collaborative AI",". For where the decision should live after the thread ends, see ",[200,11324,11326],{"href":11325},"search-is-not-memory","search is not memory",{"title":170,"searchDepth":171,"depth":171,"links":11328},[11329,11330,11331,11332],{"id":11176,"depth":171,"text":11177},{"id":11220,"depth":171,"text":11221},{"id":11251,"depth":171,"text":11252},{"id":11286,"depth":171,"text":11287},"2026-08-27","RBAC","RBAC is who is allowed to do what. For enterprise AI it has to cover the model as well as the people — what it can read, what it can change, and who can stop it. A plain-language guide.",{"eyebrow":3104,"title":11337},"Roles when the user is a model",[11339,11342,11345],{"question":11340,"answer":11341},"Is a shared chatbot login the same as RBAC?","No. A shared login says who can open the chat. RBAC says who can see which jobs, which tools, and which live systems — and whether the model may write at all.",{"question":11343,"answer":11344},"Do we need RBAC if AI is read-only?","You still need it for what the model can see. Read-only reduces the chance of a bad write. It does not decide which customer files belong in whose session.",{"question":11346,"answer":11347},"Where should we start?","Name who can approve a change to a live system, keep AI from writing until that is clear, and list unofficial tools. The auditors guide on this site is a first evidence pack.","/blog/rbac-for-enterprise-ai",{"title":11156,"description":11335},"blog/rbac-for-enterprise-ai",[1606,11334,11352],"access","g2P_DB_QD94yq1LoWlZBKGVnScVo4TxpXZx40Kjx12Y",{"id":11355,"title":11356,"archived":164,"authors":11357,"badge":11359,"body":11360,"date":3101,"definedTerm":165,"department":165,"description":11779,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":11780,"relatedHeading":165,"seo":11781,"series":1606,"sitemap":130,"status":165,"stem":11782,"subhead":165,"tags":11783,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":11785},"content/blog/what-is-shadow-ai.md","What is Shadow AI",[11358],{"name":184,"to":135},{"label":1024},{"type":167,"value":11361,"toc":11754},[11362,11365,11368,11375,11383,11386,11388,11442,11450,11452,11455,11498,11501,11505,11508,11511,11521,11523,11528,11537,11542,11547,11552,11554,11560,11566,11572,11578,11584,11590,11597,11599,11605,11608,11614,11616,11620,11626,11630,11633,11637,11644,11648,11651,11655,11658,11662,11665,11669,11679,11683,11690,11694,11697,11701,11706,11710,11713,11717,11720,11722,11728,11730],[190,11363,11364],{},"Shadow AI is employees using personal ChatGPT, Claude, Gemini, or similar tools for work because the official company tool is too slow, too locked down, or missing.",[190,11366,11367],{},"The work is real. The risk is off the books. Security often hears about it first as an incident.",[190,11369,11370,11371,11374],{},"It is the AI-era cousin of ",[193,11372,11373],{},"shadow IT",": unsanctioned software people adopt because it helps them finish the job. The pattern is older than ChatGPT — personal Dropbox, unsanctioned notebooks, Excel macros that became load-bearing. Generative AI sped it up because the tools are excellent, cheap, and one paste away from a customer list.",[190,11376,11377,11382],{},[200,11378,11381],{"href":11379,"rel":11380},"https://newsroom.ibm.com/2025-07-30-ibm-report-13-of-organizations-reported-breaches-of-ai-models-or-applications,-97-of-which-reported-lacking-proper-ai-access-controls",[204],"IBM’s 2025 Cost of a Data Breach research"," found that 20% of organisations reported security incidents involving shadow AI, and that organisations with high levels of it paid $670,000 more per breach. Sixty-three percent lacked AI governance policies.",[190,11384,11385],{},"Shame does not fix those numbers. Substitution does. People paste work into personal accounts because the deadline is tonight and the official programme is a waitlist. If the approved tool cannot see Salesforce, they will export a spreadsheet. If the approved tool is ten times slower than paste, shadow wins.",[255,11387,258],{"id":257},[260,11389,11390,11400,11410,11416,11422,11428,11436],{},[263,11391,11392,11395,11396,11399],{},[193,11393,11394],{},"Shadow IT."," Unsanctioned systems. Shadow AI is often unsanctioned ",[270,11397,11398],{},"generation"," on a sanctioned laptop — a personal account, not a new product install. At work, the browser is allowed; the tenant is not yours.",[263,11401,11402,11405,11406,11409],{},[193,11403,11404],{},"Sanctioned tool."," The company’s official AI, with company login and a vendor agreement. At work, ChatGPT Enterprise on the company tenant can be sanctioned and still be ",[193,11407,11408],{},"ungoverned for writes"," if people copy output into CRM.",[263,11411,11412,11415],{},[193,11413,11414],{},"Data leakage."," Prompts become logs at a vendor you have no processing agreement with. At work, a customer list in a consumer chat is a processing event you cannot inventory.",[263,11417,11418,11421],{},[193,11419,11420],{},"Acceptable-use policy."," A PDF. Necessary. Not a substitute for a tool people can actually use.",[263,11423,11424,11427],{},[193,11425,11426],{},"DLP / CASB."," Network or cloud tools that watch paste-out and unsanctioned apps. Useful. They do not quote a CRM change or bind a named signer.",[263,11429,11430,11433,11434,230],{},[193,11431,11432],{},"Personal API key."," A pass-through that looks like engineering hygiene and is often shadow AI with a credit card. See ",[200,11435,3749],{"href":3303},[263,11437,11438,11441],{},[193,11439,11440],{},"Pressure valve."," A logged sandbox with fake data and no production writes. Not shadow. A way to experiment without a customer list.",[190,11443,11444,11449],{},[200,11445,11448],{"href":11446,"rel":11447},"https://www.enisa.europa.eu/publications/enisa-threat-landscape-2025",[204],"ENISA’s Threat Landscape 2025"," notes fake AI-tool sites and malware posing as AI installers. People hunting for “a free assistant” are the audience. Blocking the official vendors without a substitute trains that hunt.",[255,11451,1160],{"id":1159},[190,11453,11454],{},"That can mean:",[260,11456,11457,11467,11476,11484,11492],{},[263,11458,11459,11462,11463,11466],{},[193,11460,11461],{},"Customer or internal data sitting in a consumer vendor’s logs."," Legal later asks which model saw it. Nobody can say. ",[200,11464,3556],{"href":3554,"rel":11465},[204]," does not pause because the employee used a personal account.",[263,11468,11469,11472,11473,11475],{},[193,11470,11471],{},"Changes with no record."," Someone types model output into CRM. No approver of record. That is ungoverned ",[200,11474,229],{"href":228}," with extra steps.",[263,11477,11478,11481,11482,230],{},[193,11479,11480],{},"Two versions of policy."," The official playbook says one thing. A shadow chat invented another. See ",[200,11483,2404],{"href":443},[263,11485,11486,11489,11490,230],{},[193,11487,11488],{},"Institutional amnesia."," The reasoning lived in a thread the company cannot query. See ",[200,11491,3271],{"href":3270},[263,11493,11494,11497],{},[193,11495,11496],{},"A wider attack surface."," Fake installer sites, prompt leakage, and keys in plugins.",[190,11499,11500],{},"Blocking websites without offering a sanctioned path does not end shadow AI. It trains people to use personal phones.",[809,11502,11504],{"id":11503},"what-actually-reduces-it","What actually reduces it",[190,11506,11507],{},"Find the jobs people are already doing in personal chats: drafting, summarising, extracting tables, writing the email. Put those jobs on an official path that can see the right files without a paste.",[190,11509,11510],{},"Read-only links to live systems in an approved product are how you stop the spreadsheet. Make the official path not much slower than paste. The honest metric is time to finish the job.",[190,11512,11513,11514,11516,11517,11520],{},"Perimeter blocking can tighten ",[270,11515,4463],{}," a real path exists — not before. ",[200,11518,11519],{"href":3335},"AI governance"," that is only a block list is guidance with extra steps.",[809,11522,3290],{"id":3289},[190,11524,11525,11527],{},[193,11526,3295],{}," Shadow spend hides on personal cards and departmental tools. Shadow output typed into the ledger has no trail. Finance should want a sanctioned path with quotes and caps, not a ban that moves the bill onto expenses.",[190,11529,11530,11532,11533,11536],{},[193,11531,3310],{}," Processing without an agreement, invented customer commitments, and no inventory of what left the tenant. Air Canada’s chatbot was official and still made a false commitment — ",[200,11534,2043],{"href":2041,"rel":11535},[204],". Shadow tools add the same class of fiction with even less control. Legal should not allow “non-sensitive only” personal accounts; employees are bad at classifying.",[190,11538,11539,11541],{},[193,11540,3319],{}," Deadline pressure is the demand signal. Ops should treat missing connectors and waitlists as root causes, and should offer sandboxes so experimentation does not need production data.",[190,11543,11544,11546],{},[193,11545,3325],{}," Fastest to shadow, because the consumer tools are excellent at email and decks. GTM needs read-only CRM in the official path or they will export. They also copy invented pricing into the opportunity — a write-back problem dressed as productivity.",[190,11548,11549,11551],{},[193,11550,3331],{}," Detection (surveys, DLP, key scanning) plus substitution. Punishment first yields dishonest surveys. IBM’s uplift in breach cost is the board-level argument; ENISA’s fake-tool landscape is the practical one. A secure web gateway is not a named signer.",[809,11553,3340],{"id":3339},[190,11555,11556,11559],{},[193,11557,11558],{},"Blocking as strategy."," Phones exist.",[190,11561,11562,11565],{},[193,11563,11564],{},"Sanctioned equals governed."," Company ChatGPT can still be copy-paste into Salesforce.",[190,11567,11568,11571],{},[193,11569,11570],{},"Allowing personal accounts for “non-sensitive” work."," Classification fails under deadline.",[190,11573,11574,11577],{},[193,11575,11576],{},"Shame."," Drives better hiding, not better behaviour.",[190,11579,11580,11583],{},[193,11581,11582],{},"Assuming shadow is a people problem."," It is usually a missing-path problem: no connectors, no speed, no permission to try.",[190,11585,11586,11587,11589],{},"Good looks like: self-service ",[200,11588,216],{"href":215},", wiki, read-only connectors, a named signer on writes, time-to-job close to paste, perimeter controls after substitution, sandboxes with fake data. Failure looks like a blocked URL, a PDF, and a personal Claude project full of customers.",[190,11591,11592,11596],{},[200,11593,11595],{"href":3547,"rel":11594},[204],"ICO guidance on AI and data protection"," still applies when the employee is the one pasting. Lawful basis and purpose do not wait for an official rollout. That is why substitution is a legal control as well as a security one: the unofficial path is still processing.",[255,11598,793],{"id":792},[190,11600,11601,11602,11604],{},"Nimbus is built so the legitimate path is the easy path: operators open ",[200,11603,216],{"href":215}," themselves, use approved playbooks and read-only connectors, and only write to live systems after a named person signs.",[190,11606,11607],{},"Nimbus does not “detect shadow AI” the way a network tool that watches cloud apps would. Those perimeter tools still matter. The product bet is gravitational: if governed work is live quickly, shadow has less to do.",[190,11609,3411,11610,876,11612,230],{},[200,11611,39],{"href":40},[200,11613,31],{"href":32},[255,11615,807],{"id":806},[809,11617,11619],{"id":11618},"is-using-chatgpt-enterprise-still-shadow-ai","Is using ChatGPT Enterprise still shadow AI?",[190,11621,11622,11623,11625],{},"If it is the organisation’s tenant, with company login, a processing agreement, and a defined use policy, it is sanctioned — not shadow. It can still be ",[193,11624,11408],{}," (people copy output into CRM). Sanctioned is not the same as sufficient.",[809,11627,11629],{"id":11628},"does-blocking-openai-at-the-office-network-solve-it","Does blocking OpenAI at the office network solve it?",[190,11631,11632],{},"It reduces one channel. It does not stop phones, home networks, or other vendors. Without a substitute, it also reduces productivity.",[809,11634,11636],{"id":11635},"can-we-allow-personal-accounts-for-non-sensitive-work","Can we allow personal accounts for “non-sensitive” work?",[190,11638,11639,11640,11643],{},"Employees are bad at classifying. If you allow it, assume leakage of whatever they ",[270,11641,11642],{},"think"," is non-sensitive.",[809,11645,11647],{"id":11646},"how-do-we-find-existing-shadow-ai","How do we find existing shadow AI?",[190,11649,11650],{},"Anonymous surveys, credit-card review, data-loss monitoring, scanning for personal API keys, and talking to the teams under deadline pressure. Do not start with punishment if you want honest answers.",[809,11652,11654],{"id":11653},"why-do-people-prefer-the-unofficial-tools","Why do people prefer the unofficial tools?",[190,11656,11657],{},"Speed, quality, missing connectors in the official tool, and fear of “the AI team.” Treat those as requirements, not as moral failure.",[809,11659,11661],{"id":11660},"is-a-logged-sandbox-shadow-ai","Is a logged sandbox shadow AI?",[190,11663,11664],{},"No. Fake data, no production writes, company login: that is a pressure valve. Production customer lists in a personal account are not.",[809,11666,11668],{"id":11667},"how-is-this-different-from-shadow-it","How is this different from shadow IT?",[190,11670,11671,11672,11675,11676,11678],{},"Shadow IT is often an unsanctioned ",[270,11673,11674],{},"system",". Shadow AI is often unsanctioned ",[270,11677,11398],{}," on a laptop you issued. Your asset inventory will look clean while the prompts leave.",[809,11680,11682],{"id":11681},"what-does-ibms-research-actually-say-here","What does IBM’s research actually say here?",[190,11684,11685,11686,230],{},"IBM reported shadow-AI incidents, higher average breach cost where shadow AI was high, and a large share of organisations lacking AI governance policies. Use it as evidence that this is a control topic, not a manners topic. Read the ",[200,11687,11689],{"href":11379,"rel":11688},[204],"newsroom summary",[809,11691,11693],{"id":11692},"will-a-better-acceptable-use-policy-be-enough","Will a better acceptable-use policy be enough?",[190,11695,11696],{},"Write it. Then put the same rules in a product people can finish the job with. PDFs do not see Salesforce.",[809,11698,11700],{"id":11699},"how-do-writes-sneak-in","How do writes sneak in?",[190,11702,11703,11704,230],{},"The model never calls Salesforce. A human pastes the answer. That is still a change to a live system with no quote and no named signer. See ",[200,11705,3286],{"href":228},[809,11707,11709],{"id":11708},"should-we-ban-plugins-and-personal-api-keys","Should we ban plugins and personal API keys?",[190,11711,11712],{},"Treat them as unsanctioned processing until they sit on a company path with a ceiling. Keys in wikis are an unmetered utility and a credential incident.",[809,11714,11716],{"id":11715},"what-is-the-first-sanctioned-path-worth-shipping","What is the first sanctioned path worth shipping?",[190,11718,11719],{},"Read-only connectors on the jobs people already paste — email, extract, summarise — plus a wiki they can cite. Writes come later, fail-closed. Speed matters more than a perfect platform launch.",[255,11721,869],{"id":868},[190,11723,11724,876,11726,230],{},[200,11725,3336],{"href":3335},[200,11727,3286],{"href":228},[255,11729,883],{"id":882},[260,11731,11732,11738,11744,11749],{},[263,11733,11734],{},[200,11735,11737],{"href":11379,"rel":11736},[204],"IBM newsroom, Cost of a Data Breach 2025",[263,11739,11740],{},[200,11741,11743],{"href":11446,"rel":11742},[204],"ENISA Threat Landscape 2025",[263,11745,11746],{},[200,11747,3556],{"href":3554,"rel":11748},[204],[263,11750,11751],{},[200,11752,2228],{"href":2041,"rel":11753},[204],{"title":170,"searchDepth":171,"depth":171,"links":11755},[11756,11757,11762,11763,11777,11778],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":11758},[11759,11760,11761],{"id":11503,"depth":1001,"text":11504},{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":11764},[11765,11766,11767,11768,11769,11770,11771,11772,11773,11774,11775,11776],{"id":11618,"depth":1001,"text":11619},{"id":11628,"depth":1001,"text":11629},{"id":11635,"depth":1001,"text":11636},{"id":11646,"depth":1001,"text":11647},{"id":11653,"depth":1001,"text":11654},{"id":11660,"depth":1001,"text":11661},{"id":11667,"depth":1001,"text":11668},{"id":11681,"depth":1001,"text":11682},{"id":11692,"depth":1001,"text":11693},{"id":11699,"depth":1001,"text":11700},{"id":11708,"depth":1001,"text":11709},{"id":11715,"depth":1001,"text":11716},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Shadow AI is people using personal ChatGPT or similar for work because the official tool is too slow or missing — which leaks data and leaves no record of what changed.","/blog/what-is-shadow-ai",{"title":11356,"description":11779},"blog/what-is-shadow-ai",[1606,11784,340,11313],"shadow-ai","MIpmteaB8bIplpgHCcPotzgOMNe8x4IDeegPm_cOF6k",{"id":11787,"title":11788,"archived":164,"authors":11789,"badge":11791,"body":11792,"date":3101,"definedTerm":165,"department":165,"description":12274,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":12275,"relatedHeading":165,"seo":12276,"series":1606,"sitemap":130,"status":165,"stem":12277,"subhead":165,"tags":12278,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":12279},"content/blog/what-is-write-back-governance.md","What is Write-Back Governance",[11790],{"name":184,"to":135},{"label":1024},{"type":167,"value":11793,"toc":12249},[11794,11800,11807,11810,11821,11832,11834,11878,11889,11891,11898,11900,11914,11917,11945,11948,11955,11959,11965,11972,11975,11978,11981,11983,11990,11999,12007,12012,12027,12029,12035,12041,12047,12053,12059,12065,12073,12079,12085,12087,12093,12098,12104,12106,12110,12116,12120,12123,12127,12130,12134,12142,12146,12149,12153,12160,12164,12167,12171,12174,12178,12181,12185,12190,12194,12200,12204,12207,12209,12215,12217],[190,11795,11796,11797,11799],{},"Write-back means the AI is allowed to ",[193,11798,4120],{}," a live business system — a CRM field, an ERP journal, a customer message — not just draft a suggestion.",[190,11801,11802,11803,11806],{},"Write-back governance is the control that decides whether that is allowed, exactly what will change, who must sign, and how you can prove it later. Its defining property is ",[193,11804,11805],{},"fail-closed",": if approval is missing, the write does not occur.",[190,11808,11809],{},"A prompt that says “please ask first” is etiquette. It is not a control.",[190,11811,11812,11813,11817,11818,11820],{},"In June 2023, a New York federal judge sanctioned two lawyers who filed a brief citing cases ",[200,11814,11816],{"href":1904,"rel":11815},[204],"ChatGPT had invented"," — the ",[270,11819,9126],{}," episode. Fiction had been written into a court record. The same failure mode is waiting in CRM and ERP: fluent output that becomes an operational fact.",[190,11822,11823,11824,11827,11828,11831],{},"Air Canada’s chatbot invented a bereavement fare and the company was held to the commitment — ",[200,11825,2043],{"href":2041,"rel":11826},[204],", decision ",[200,11829,9050],{"href":9048,"rel":11830},[204],". A customer-facing message is a write to the relationship even when no CRM API fired.",[255,11833,258],{"id":257},[260,11835,11836,11842,11848,11853,11858,11863,11867,11872],{},[263,11837,11838,11841],{},[193,11839,11840],{},"Live system / system of record."," Salesforce, NetSuite, Workday — the official place the number or record lives. At work, this is where other teams will inherit the new value.",[263,11843,11844,11847],{},[193,11845,11846],{},"Payload / quote."," The exact change, shown before it runs: fields, values, line items — not “update CRM.” At work, Amount and a next-step note are not the same quote.",[263,11849,11850,11852],{},[193,11851,4138],{}," The identity that authorised the payload. Shared inboxes and “whoever is online” destroy this.",[263,11854,11855,11857],{},[193,11856,9119],{}," An older control: one person proposes, another authorises. Write-back governance is that instinct for AI, when the proposer is a model.",[263,11859,11860,11862],{},[193,11861,6113],{}," A secure link from the AI product to a live system. Read-only is a control. A production write login is a different risk class.",[263,11864,11865,4145],{},[193,11866,4144],{},[263,11868,11869,11871],{},[193,11870,4183],{}," The connected identity should not be able to edit every object “because setup was easier.” Application-level quotes do not shrink a superuser token.",[263,11873,11874,11877],{},[193,11875,11876],{},"Rollback."," Sometimes possible for a field; often impossible for a sent message. Rollback is not a substitute for a gate.",[190,11879,3395,11880,11885,11886,11888],{},[200,11881,11884],{"href":11882,"rel":11883},"https://www.nist.gov/cyberframework",[204],"NIST Cybersecurity Framework"," is useful vocabulary here — identify, protect, detect, respond, recover — but it does not, by itself, quote a Salesforce payload. ",[200,11887,11519],{"href":3335}," is the broader programme. Write-back governance is the write subset.",[255,11890,1160],{"id":1159},[190,11892,11893,11894,11897],{},"Most enterprise software already has write controls: CRM validation, ERP posting periods, maker-checker in banking. Generative AI added a new writer that does not respect those cultures unless the ",[193,11895,11896],{},"runtime"," is bound to them.",[190,11899,9135],{},[260,11901,11902,11905,11908,11911],{},[263,11903,11904],{},"change revenue, pipeline, or customer records",[263,11906,11907],{},"post journals",[263,11909,11910],{},"send a customer a message that asserts a term or a price",[263,11912,11913],{},"bulk-update hundreds of rows",[190,11915,11916],{},"Industry maturity, in plain steps:",[547,11918,11919,11927,11933,11939],{},[263,11920,11921,11924,11925,230],{},[193,11922,11923],{},"Copy-paste."," A human types model output into the live system. Unlogged as AI. See ",[200,11926,4279],{"href":3249},[263,11928,11929,11932],{},[193,11930,11931],{},"Tools plus manners."," The model has an “update opportunity” button. Instructions say to confirm. Haste and bugs bypass.",[263,11934,11935,11938],{},[193,11936,11937],{},"Admin toggles."," “Allow writes” per user. No preview of the exact change.",[263,11940,11941,11944],{},[193,11942,11943],{},"Quoted, named, fail-closed."," The product shows the change, binds the signer’s identity, records the outcome.",[190,11946,11947],{},"Only (4) is write-back governance in the sense auditors mean.",[190,11949,11950,11951,11954],{},"Salesforce validation will stop some nonsense. It will not record that an ",[270,11952,11953],{},"AI"," proposed the change, which playbook was cited, or that finance rejected an earlier payload.",[809,11956,11958],{"id":11957},"how-to-turn-it-on-without-starting-in-production","How to turn it on without starting in production",[190,11960,11961,11962,11964],{},"Start ",[193,11963,4384],{},". Prove retrieval and drafts. Count how often humans would have written — and how often the draft contains something you would not file.",[190,11966,11967,11968,11971],{},"Enable writes ",[193,11969,11970],{},"per job and per object type",". A next-step note is not Amount. A journal is not Slack.",[190,11973,11974],{},"Always show the payload. Rejects are success: a stored rejection proves the gate. Never hide bulk in a single “approve 200 records” with no visible set.",[190,11976,11977],{},"Rollback is not a substitute. Some writes are messages you cannot unsend.",[190,11979,11980],{},"Do not give the agent a full production key “because the proof of value was read-only” and promise to add gates later. Later is when the first bad write ships.",[809,11982,3290],{"id":3289},[190,11984,11985,11987,11988,230],{},[193,11986,3295],{}," Journals and material fields need the owner of the analogue posting, a field-level quote, and a chain that shows the wiki version cited. A champion who does not own the ledger will rubber-stamp. Spend caps stop looping proposers; they do not replace the signer. See ",[200,11989,3749],{"href":3303},[190,11991,11992,11994,11995,11998],{},[193,11993,3310],{}," Customer messages, terms, and anything that could become a commitment. Air Canada is the caution. Legal should also treat copy-paste from a consumer model as a write that bypassed the programme. ",[200,11996,9126],{"href":1904,"rel":11997},[204]," is fluent fiction entering a record — the CRM analogue is a next-step that never happened, or a clause the company does not offer.",[190,12000,12001,12003,12004,12006],{},[193,12002,3319],{}," Object-class rollout, visible bulk, human wait as a ",[200,12005,5666],{"href":1954}," step. Ops should measure time-to-approved-write and reject rate, and should refuse tenant-wide write toggles.",[190,12008,12009,12011],{},[193,12010,3325],{}," Friction versus incident. GTM should get fast quotes on low-radius fields first, not a weekend cleanup agent. Slack is not “safe chat”; it is still a write to a system people treat as official.",[190,12013,12014,12016,12017,12019,12020,12023,12024,12026],{},[193,12015,3331],{}," Jailbreaks should not execute. Combine least-privilege identity ",[193,12018,9165],{}," application-level quotes. The ",[200,12021,977],{"href":375,"rel":12022},[204]," is about requesting bad actions; the gate is about refusing to run them. ",[200,12025,758],{"href":757}," will happily pass a write; it will not implement fail-closed.",[809,12028,3340],{"id":3339},[190,12030,12031,12034],{},[193,12032,12033],{},"Salesforce permissions as sufficient."," Necessary. If the connected user can edit all objects, the agent inherits that blast radius even with quotes.",[190,12036,12037,12040],{},[193,12038,12039],{},"Prompt manners."," “Please ask first” is not fail-closed.",[190,12042,12043,12046],{},[193,12044,12045],{},"Admin allow-writes."," No payload, no named signer, no record of rejects.",[190,12048,12049,12052],{},[193,12050,12051],{},"Bulk one-click."," A rubber stamp with radius.",[190,12054,12055,12058],{},[193,12056,12057],{},"Rollback as the control."," You cannot unsend.",[190,12060,12061,12064],{},[193,12062,12063],{},"Fully autonomous production writes."," Only for low-radius, reversible actions you would otherwise schedule, with logging.",[190,12066,12067,12070,12071,230],{},[193,12068,12069],{},"Theatre HITL."," A checkbox is not a quote. See ",[200,12072,3923],{"href":1896},[190,12074,12075,12076,12078],{},"Good looks like: connectors default to read-only; writes opt-in per job and object; field-level quotes; named identity; model cannot waive; rejects stored on the ",[200,12077,3186],{"href":711},"; least-privilege tokens. Failure looks like a production key issued after a read-only demo.",[190,12080,12081,12082,12084],{},"This is a core job of an ",[200,12083,4372],{"href":4371},": I/O control, not chat with an API on the side.",[255,12086,793],{"id":792},[190,12088,12089,12090,12092],{},"Connectors default to ",[193,12091,4384],{},". Write-back is opt-in. The UI quotes the intended change at the field or line level. Humans sign with their identity. The model cannot waive the gate. Missing approval is fail-closed: nothing happens.",[190,12094,3395,12095,12097],{},[200,12096,23],{"href":711}," stores the quote, the decision, the execution, and errors. Agent teams can draft. They cannot release.",[190,12099,3411,12100,876,12102,230],{},[200,12101,39],{"href":40},[200,12103,50],{"href":51},[255,12105,807],{"id":806},[809,12107,12109],{"id":12108},"isnt-this-just-permissions-on-the-salesforce-user","Isn’t this just permissions on the Salesforce user?",[190,12111,12112,12113,12115],{},"Permissions are necessary. If the connected user can edit all objects, the agent inherits that blast radius even with quotes. Combine least-privilege identity ",[193,12114,9165],{}," application-level quotes.",[809,12117,12119],{"id":12118},"can-we-write-back-to-slack-but-not-crm","Can we write back to Slack but not CRM?",[190,12121,12122],{},"Yes. Different systems, different bars. Do not treat “chat” as inherently safe.",[809,12124,12126],{"id":12125},"does-this-slow-revenue-teams","Does this slow revenue teams?",[190,12128,12129],{},"It slows unreviewed mutation and speeds reviewed mutation relative to email-and-hope. Measure time-to-approved-write, not time-to-first-answer.",[809,12131,12133],{"id":12132},"can-a-jailbreak-bypass-the-gate","Can a jailbreak bypass the gate?",[190,12135,12136,12137,9202,12139,12141],{},"It can trick the model into ",[270,12138,5847],{},[270,12140,9205],{}," without a quote and a signer.",[809,12143,12145],{"id":12144},"what-about-fully-autonomous-writes","What about fully autonomous writes?",[190,12147,12148],{},"Only for low-radius, reversible actions you would otherwise schedule, with logging. If you would not let a scheduled job do it, do not let an agent do it unattended.",[809,12150,12152],{"id":12151},"how-is-copy-paste-different-from-a-connector-write","How is copy-paste different from a connector write?",[190,12154,12155,12156,12159],{},"The live system still changes. Copy-paste is usually unlogged as AI and often happens on ",[200,12157,12158],{"href":3249},"shadow"," tools. It is write-back without governance.",[809,12161,12163],{"id":12162},"why-start-read-only","Why start read-only?",[190,12165,12166],{},"Because you need a baseline of draft quality and a count of how often a human would have written. Turning on writes on day one teaches the organisation to skip the quote.",[809,12168,12170],{"id":12169},"is-a-next-step-note-the-same-as-amount","Is a next-step note the same as Amount?",[190,12172,12173],{},"No. Enable writes per object type. Low-radius, reversible fields can come first. Money and contractual language should not piggy-back on a note permission.",[809,12175,12177],{"id":12176},"do-rejects-matter","Do rejects matter?",[190,12179,12180],{},"Yes. A stored rejection proves the gate and teaches the playbook. A six-month zero reject rate is a finding.",[809,12182,12184],{"id":12183},"can-mcp-or-a-plugin-be-the-write-path","Can MCP or a plugin be the write path?",[190,12186,12187,12188,230],{},"They can be wiring. They are not the gate. See ",[200,12189,1141],{"href":757},[809,12191,12193],{"id":12192},"how-does-this-relate-to-the-eu-ai-act","How does this relate to the EU AI Act?",[190,12195,12196,12197,230],{},"Effective oversight, for higher-risk systems, includes the ability to interrupt. A fail-closed named signer is that instinct for operational writes. It is not, by itself, “Act compliant.” See ",[200,12198,601],{"href":599,"rel":12199},[204],[809,12201,12203],{"id":12202},"what-should-auditors-see","What should auditors see?",[190,12205,12206],{},"The payload the signer saw, the identity, the timestamp, the playbook version cited, and whether the live system accepted or rejected the change — on a chain, not in a screenshot.",[255,12208,869],{"id":868},[190,12210,12211,876,12213,230],{},[200,12212,3250],{"href":3249},[200,12214,4519],{"href":4371},[255,12216,883],{"id":882},[260,12218,12219,12224,12229,12234,12239,12244],{},[263,12220,12221],{},[200,12222,9429],{"href":1904,"rel":12223},[204],[263,12225,12226],{},[200,12227,11884],{"href":11882,"rel":12228},[204],[263,12230,12231],{},[200,12232,2228],{"href":2041,"rel":12233},[204],[263,12235,12236],{},[200,12237,9417],{"href":9048,"rel":12238},[204],[263,12240,12241],{},[200,12242,977],{"href":375,"rel":12243},[204],[263,12245,12246],{},[200,12247,4560],{"href":599,"rel":12248},[204],{"title":170,"searchDepth":171,"depth":171,"links":12250},[12251,12252,12257,12258,12272,12273],{"id":257,"depth":171,"text":258},{"id":1159,"depth":171,"text":1160,"children":12253},[12254,12255,12256],{"id":11957,"depth":1001,"text":11958},{"id":3289,"depth":1001,"text":3290},{"id":3339,"depth":1001,"text":3340},{"id":792,"depth":171,"text":793},{"id":806,"depth":171,"text":807,"children":12259},[12260,12261,12262,12263,12264,12265,12266,12267,12268,12269,12270,12271],{"id":12108,"depth":1001,"text":12109},{"id":12118,"depth":1001,"text":12119},{"id":12125,"depth":1001,"text":12126},{"id":12132,"depth":1001,"text":12133},{"id":12144,"depth":1001,"text":12145},{"id":12151,"depth":1001,"text":12152},{"id":12162,"depth":1001,"text":12163},{"id":12169,"depth":1001,"text":12170},{"id":12176,"depth":1001,"text":12177},{"id":12183,"depth":1001,"text":12184},{"id":12192,"depth":1001,"text":12193},{"id":12202,"depth":1001,"text":12203},{"id":868,"depth":171,"text":869},{"id":882,"depth":171,"text":883},"Write-back governance is the rule that an AI may not change a live business system until a named person has seen the exact change and signed it — and if they have not, nothing happens.","/blog/what-is-write-back-governance",{"title":11788,"description":12274},"blog/what-is-write-back-governance",[1606,229,340,225],"B0qAttLqSYzLKPVKMowRKG4dO8WM1fUVHa7Q77dA6yA",{"fold":12281,"id":12286,"title":12287,"archived":164,"authors":165,"badge":165,"body":12288,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":12292,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":12296,"relatedHeading":165,"seo":12297,"series":165,"sitemap":164,"status":165,"stem":12298,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":12299},{"headline":12282,"description":12283,"primaryLabel":8,"primaryTo":12284,"secondaryLabel":12285,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Free 7-day trial.","/checkout","Explore the platform","content/shared/cta.md","Site CTAs",{"type":167,"value":12289,"toc":12290},[],{"title":170,"searchDepth":171,"depth":171,"links":12291},[],{"headline":12293,"description":12294,"primaryLabel":8,"primaryTo":12284,"secondaryLabel":12295,"secondaryTo":99},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger - free for 7 days.","Talk to our team","/shared/cta",{"title":12287,"description":170},"shared/cta","wz4AdRHnaYH021WMdWcHnZvHmkZJNKaNG4XGZfnFBtw",{"enabled":164,"message":12301,"linkLabel":93,"linkHref":94,"id":12302,"title":12303,"archived":164,"authors":165,"badge":165,"body":12304,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":12308,"relatedHeading":165,"seo":12309,"series":165,"sitemap":164,"status":165,"stem":12310,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":12311},"We're hiring! Join the team building the Sentient Enterprise.","content/shared/hiring.md","Hiring banner",{"type":167,"value":12305,"toc":12306},[],{"title":170,"searchDepth":171,"depth":171,"links":12307},[],"/shared/hiring",{"title":12303,"description":170},"shared/hiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",1788985846770]