[{"data":1,"prerenderedAt":2298},["ShallowReactive",2],{"site-nav-content":3,"hub:evaluate:posts":178,"site-cta-content":2266,"hiring-banner-content":2286},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":131,"id":162,"title":163,"archived":164,"authors":165,"badge":165,"body":166,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":174,"relatedHeading":165,"seo":175,"series":165,"sitemap":164,"status":165,"stem":176,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":177},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","/overview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","/product/conflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","/product/agent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","/product/lifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","/product/wiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","/product/workstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","/product/perception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","/product/governance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","/models",{"label":47,"to":48},"Pricing","/pricing",{"label":50,"to":51},"Integrations","/integrations",{"label":53,"to":54},"Security","/security",{"label":56,"to":57},"Partners","/partners",{"label":59,"to":60},"Insights","/blog",{"productHeading":5,"companyHeading":62,"legalHeading":63,"docsLabel":64,"docsUrl":65,"statementLines":66,"copyright":69,"companyLinks":70,"legalLinks":100,"socialLinks":110,"bottomLinks":120},"Company","Legal","Docs","https://docs.gonimbus.ai",[67,68],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[71,72,73,74,75,78,81,84,87,90,92,95,98],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":76,"to":77},"Glossary","/glossary",{"label":79,"to":80},"Compare","/compare",{"label":82,"to":83},"Evaluate","/evaluate",{"label":85,"to":86},"Problems","/problems",{"label":88,"to":89},"Use cases","/use-cases",{"label":91,"to":57},"Partner Program",{"label":93,"to":94},"Careers","/careers",{"label":96,"to":97},"System status","/status",{"label":7,"to":99},"/contact",[101,104,107],{"label":102,"to":103},"Terms of Service","/terms",{"label":105,"to":106},"Privacy Policy","/privacy",{"label":108,"to":109},"Compliance","/compliance",[111,114,117],{"label":112,"href":113},"LinkedIn","https://www.linkedin.com/company/gonimbusai/",{"label":115,"href":116},"X","https://x.com/gonimbusai",{"label":118,"href":119},"Instagram","https://www.instagram.com/gonimbus_ai/",[121,123,125,126,127],{"label":122,"to":103},"Terms",{"label":124,"to":106},"Privacy",{"label":108,"to":109},{"label":96,"to":97},{"label":128,"to":129,"external":130},"LLMs.txt","/llms.txt",true,{"text":132,"prompt":133},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":134,"platforms":136},{" Summarize the highlights from Nimbus's website":135},"https://gonimbus.ai",[137,142,147,152,157],{"name":138,"label":139,"icon":140,"hrefPrefix":141},"chatgpt","ChatGPT","simple-icons:openai","https://chatgpt.com/?prompt=",{"name":143,"label":144,"icon":145,"hrefPrefix":146},"perplexity","Perplexity","mdi:magnify","https://www.perplexity.ai/search/new?q=",{"name":148,"label":149,"icon":150,"hrefPrefix":151},"grok","Grok","simple-icons:x","https://x.com/i/grok?text=",{"name":153,"label":154,"icon":155,"hrefPrefix":156},"claude","Claude","simple-icons:anthropic","https://claude.ai/new?q=",{"name":158,"label":159,"icon":160,"hrefPrefix":161},"google-ai","Google AI","simple-icons:google","https://www.google.com/search?udm=50&aep=11&q=","content/shared/nav.md","Site navigation",false,null,{"type":167,"value":168,"toc":169},"minimark",[],{"title":170,"searchDepth":171,"depth":171,"links":172},"",2,[],"md","/shared/nav",{"title":163,"description":170},"shared/nav","rDEv5cVG6P2l9ATcdQv2n6VOQiSL5ioOutyfvGevcD0",[179,947,1609,1863],{"id":180,"title":181,"archived":164,"authors":182,"badge":185,"body":187,"date":936,"definedTerm":165,"department":165,"description":937,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":938,"relatedHeading":165,"seo":939,"series":940,"sitemap":130,"status":165,"stem":941,"subhead":165,"tags":942,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":946},"content/blog/how-to-choose-between-a-coding-harness-and-an-enterprise-harness.md","How to Choose Between a Coding Harness and an Enterprise Harness",[183],{"name":184,"to":135},"Nimbus Research",{"label":186},"Evaluation",{"type":167,"value":188,"toc":918},[189,210,228,248,256,261,345,362,366,369,381,402,408,414,428,432,438,444,454,466,480,494,498,504,510,520,530,536,543,547,599,613,627,631,642,651,667,679,688,691,699,716,723,726,746,750,755,758,762,773,777,780,784,787,791,803,807,817,821],[190,191,192,193,197,198,201,202,209],"p",{},"Choosing between a ",[194,195,196],"strong",{},"coding harness"," and an ",[194,199,200],{},"enterprise harness"," is choosing the workspace. A coding harness (Claude Code, Cursor, Codex, open shells) wraps a model for a developer and a repository. An enterprise harness wraps a model for operators and systems of record. Same equation — ",[203,204,208],"a",{"href":205,"rel":206},"https://docs.langchain.com/oss/python/langchain/agents",[207],"nofollow","Agent = Model + Harness"," — different loop.",[190,211,212,213,217,218,222,223,227],{},"This is the buying companion to ",[203,214,216],{"href":215},"inner-vs-outer-agent-harness","inner vs outer agent harness",". It sits beside ",[203,219,221],{"href":220},"how-to-choose-between-a-copilot-and-a-work-os","how to choose between a copilot and a work OS",": copilots are personal assistants; coding harnesses are ",[224,225,226],"em",{},"agentic"," inner loops with tools and tests; enterprise harnesses are outer loops with grants and signers. Do not collapse all three into “we need ChatGPT.”",[190,229,230,235,236,241,242,247],{},[203,231,234],{"href":232,"rel":233},"https://martinfowler.com/articles/harness-engineering.html",[207],"Böckeler"," documents how coding-agent users add guides and sensors. ",[203,237,240],{"href":238,"rel":239},"https://addyosmani.com/blog/own-the-outer-loop/",[207],"Osmani"," tells engineers to own verify-and-release. ",[203,243,246],{"href":244,"rel":245},"https://www.thoughtworks.com/insights/articles/operating-system-enterprise-ai",[207],"Thoughtworks"," argues the organisational layer is still the gap. The purchase mistake is using one budget line for all three layers.",[190,249,250,255],{},[203,251,254],{"href":252,"rel":253},"https://www.mckinsey.com/capabilities/quantumblack/our-insights/the-state-of-ai",[207],"McKinsey’s 2025 State of AI"," is the organisational backdrop: usage is easy; scale is redesign. A Cursor rollout can scale pull requests. It will not, by itself, scale governed CRM writes. An OS-class rollout can scale those writes. It will annoy engineers if you force “rewrite this function” through a Critical gate.",[257,258,260],"h2",{"id":259},"words-youll-hear","Words you’ll hear",[262,263,264,292,303,325,335],"ul",{},[265,266,267,270,271,275,276,279,280,285,286,291],"li",{},[194,268,269],{},"Coding / inner harness."," Repo workspace, sandbox, ",[272,273,274],"code",{},"AGENTS.md"," / ",[272,277,278],{},"CLAUDE.md",", hooks, CI. Eval: ",[203,281,284],{"href":282,"rel":283},"https://www.swebench.com/",[207],"SWE-bench",", ",[203,287,290],{"href":288,"rel":289},"https://arxiv.org/abs/2601.11868",[207],"Terminal-Bench",", your tests.",[265,293,294,297,298,302],{},[194,295,296],{},"Enterprise / outer harness."," Job workspace, connectors, roster, write quotes, ledger. Eval: signed payload vs SoR. ",[203,299,301],{"href":300},"what-is-an-enterprise-agent-harness","What is an enterprise agent harness",".",[265,304,305,308,309,285,314,285,319,324],{},[194,306,307],{},"Copilot."," Personal completion surface. Often no repo loop. ",[203,310,313],{"href":311,"rel":312},"https://openai.com/business/chatgpt-enterprise/",[207],"ChatGPT Enterprise",[203,315,318],{"href":316,"rel":317},"https://www.microsoft.com/en-us/microsoft-365/copilot",[207],"Microsoft 365 Copilot",[203,320,323],{"href":321,"rel":322},"https://www.anthropic.com/news/claude-for-work",[207],"Claude for Work",". Keep for mail. Do not hand it the NetSuite token.",[265,326,327,330,331,302],{},[194,328,329],{},"Framework."," How you assemble a loop in code. Not a purchase of a company workspace. ",[203,332,334],{"href":333},"agent-harness-vs-agent-framework","Harness vs framework",[265,336,337,340,341,302],{},[194,338,339],{},"MCP."," Plug into either. Dangerous when both share a production write server. ",[203,342,344],{"href":343},"mcp-for-enterprise-integrations","MCP for enterprise",[190,346,347,348,285,351,285,354,357,358,302],{},"Nimbus is an enterprise / outer option: ",[203,349,350],{"href":32},"workstreams",[203,352,353],{"href":20},"teams",[203,355,356],{"href":40},"governance",". Claude Code is a coding / inner option. The rational stack is both, with a hard rule: no unsigned SoR writes from the inner harness. ",[203,359,361],{"href":360},"how-to-solve-unapproved-crm-writes-from-ai","How to solve unapproved CRM writes",[257,363,365],{"id":364},"why-the-choice-is-usually-both","Why the choice is usually “both”",[190,367,368],{},"The tools look similar in a first meeting. Both stream tokens. Both call tools. Both have “agents” on the website. The evaluation is what happens after the answer.",[190,370,371,374,375,380],{},[194,372,373],{},"Buy a coding harness when"," the artefact is code in a repo you already trust with CI: features, refactors, tests, developer docs, infra-as-code that merges through the same gates humans use. ",[203,376,379],{"href":377,"rel":378},"https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents",[207],"Anthropic’s long-running harness"," is this world: git, progress files, end-to-end checks.",[190,382,383,386,387,391,392,391,396,401],{},[194,384,385],{},"Buy an enterprise harness when"," the artefact is a change to Salesforce, NetSuite, a policy commitment, or a cross-department decision that must be replayed. ",[203,388,390],{"href":389},"what-is-write-back-governance","Write-back",". ",[203,393,395],{"href":394},"what-is-human-in-the-loop-ai","HITL",[203,397,400],{"href":398,"rel":399},"https://www.nist.gov/itl/ai-risk-management-framework",[207],"NIST RMF"," context of use is operations, not a checkout.",[190,403,404,407],{},[194,405,406],{},"Keep a copilot when"," the job is a paragraph in a mailbox. Do not scale it into an approval architecture.",[190,409,410,413],{},[194,411,412],{},"Build on a framework when"," engineers own a unique loop and will maintain grants. That is a programme, not a seat.",[190,415,416,421,422,427],{},[203,417,420],{"href":418,"rel":419},"https://hai.stanford.edu/ai-index/2025-ai-index-report",[207],"Stanford HAI’s 2025 AI Index"," charts the explosion of coding-agent tooling. Procurement that only reads that chart will under-buy the outer layer. Procurement that only reads ",[203,423,426],{"href":424,"rel":425},"https://www.iso.org/standard/42001",[207],"ISO 42001"," will over-process inner loops and lose developers.",[257,429,431],{"id":430},"decision-tests","Decision tests",[190,433,434,437],{},[194,435,436],{},"1. What is the system of record for the outcome?"," Git: inner. CRM/ERP/customer commitment: outer. Both: two harnesses, one write plane (the outer quotes).",[190,439,440,443],{},[194,441,442],{},"2. Who is the signer?"," The author of the PR (inner, plus CODEOWNERS). A named RevOps/Finance/Legal role (outer). If you cannot name the role, you are not ready to buy the outer write path — buy read-only first.",[190,445,446,449,450,302],{},[194,447,448],{},"3. What is the independent sensor?"," Pytest / tsc / CI (inner). Payload schema + SoR read-back (outer). “The model said it was fine” is neither. ",[203,451,453],{"href":452},"eval-loops-for-enterprise-agent-harnesses","Eval loops",[190,455,456,459,460,465],{},[194,457,458],{},"4. What identity should the tools use?"," Developer sandbox and repo token (inner). Workstream-scoped OAuth (outer). A shared MCP god account fails both ",[203,461,464],{"href":462,"rel":463},"https://genai.owasp.org/llm-top-10/",[207],"OWASP"," and SoD.",[190,467,468,471,472,474,475,479],{},[194,469,470],{},"5. How will you ratchet failures?"," Inner: ",[272,473,274],{}," + hooks + tests (",[203,476,478],{"href":477},"what-is-harness-engineering","harness engineering","). Outer: wiki revision + gate tier + graph. If your plan is “we’ll prompt better,” you have not chosen a harness. You have chosen hope.",[190,481,482,485,486,391,490,302],{},[194,483,484],{},"6. Time-to-value and staffing."," Cursor can be a week for a team that already has CI. AIP can be a programme. Nimbus-style self-service claims a product week for a standard write — verify with a ",[203,487,489],{"href":488},"how-to-run-an-enterprise-ai-proof-of-value","PoV",[203,491,493],{"href":492},"self-service-vs-forward-deployed-ai-platforms","Self-service vs FDE",[257,495,497],{"id":496},"anti-patterns","Anti-patterns",[190,499,500,503],{},[194,501,502],{},"Cursor for Salesforce."," MCP connected to production. Tests on fixtures. Amount changes. No signer in the ledger. Inner loop on an outer record.",[190,505,506,509],{},[194,507,508],{},"Work OS for a one-line refactor."," Critical gate, three departments. Engineers route around. Outer loop on an inner job.",[190,511,512,515,516,302],{},[194,513,514],{},"One mesh to rule them."," IDE, chatbot, and OS all write through the same server. Two writers. ",[203,517,519],{"href":518},"multi-agent-ai-architecture","Multi-agent architecture",[190,521,522,525,526,302],{},[194,523,524],{},"Benchmark shopping."," Buying Agentforce because of a coding leaderboard, or buying Claude Code because of a governance white paper. Wrong evidence. ",[203,527,529],{"href":528},"how-to-evaluate-an-agent-harness","How to evaluate an agent harness",[190,531,532,535],{},[194,533,534],{},"Banning inner harnesses until the OS ships."," Usually slows software and does not stop paste-into-CRM. Ban the write path; allow the compile path.",[190,537,538,539,542],{},"Nimbus should lose the inner job on purpose. If a vendor tries to replace Claude Code for application engineering, ask for sandbox, hooks, and merge sensors — ",[203,540,541],{"href":528},"evaluate the harness"," — and expect to keep a coding tool anyway. If a coding-tool vendor tries to replace the OS for NetSuite journals, ask for quoted GL lines and a Finance signer.",[257,544,546],{"id":545},"a-simple-portfolio","A simple portfolio",[548,549,550,563],"table",{},[551,552,553],"thead",{},[554,555,556,560],"tr",{},[557,558,559],"th",{},"Job",[557,561,562],{},"Buy",[564,565,566,575,583,591],"tbody",{},[554,567,568,572],{},[569,570,571],"td",{},"Mail, slides, one-off Q&A",[569,573,574],{},"Copilot",[554,576,577,580],{},[569,578,579],{},"Application and infra repos",[569,581,582],{},"Coding harness",[554,584,585,588],{},[569,586,587],{},"Cross-department SoR writes",[569,589,590],{},"Enterprise harness",[554,592,593,596],{},[569,594,595],{},"Unique simulation / exotic tools",[569,597,598],{},"Framework + your grants",[190,600,601,602,605,606,608,609,612],{},"Most enterprises tick all four rows. Budget them separately. Share policy ",[224,603,604],{},"intent"," (discount cap) via wiki and via ",[272,607,274],{}," where relevant; share ",[224,610,611],{},"enforcement"," only on the plane that can execute the write.",[190,614,615,616,618,619,622,623,626],{},"See ",[203,617,11],{"href":12}," for how Nimbus maps to the third row, ",[203,620,621],{"href":45},"models"," for routing, ",[203,624,625],{"href":51},"integrations"," for connectors. See Claude Code / Cursor docs for the second. Do not let a single SOW blur the rows.",[257,628,630],{"id":629},"procurement-sequence-that-does-not-waste-a-quarter","Procurement sequence that does not waste a quarter",[190,632,633,636,637,641],{},[194,634,635],{},"Week 1 — inventory loops, not vendors."," List jobs that already have a finish line. Tag each: git artefact, SoR artefact, mailbox artefact, unique research. You now have four shopping lists. ",[203,638,640],{"href":252,"rel":639},[207],"McKinsey"," programmes that skip this step buy one platform and force every row into it.",[190,643,644,647,648,302],{},[194,645,646],{},"Week 2 — freeze the write rule."," Unsigned SoR writes are impossible from copilots, coding agents, frameworks, and the OS. That rule is cheaper than any bake-off. It also tells Security what to revoke this month (god MCP servers). ",[203,649,650],{"href":360},"Unapproved CRM writes",[190,652,653,656,657,662,663,666],{},[194,654,655],{},"Week 3 — inner bake-off only if you lack a coding harness."," Hooks, sandbox, CI independence, model swap on the same tools. Terminal-Bench and SWE-bench as vendor quality, not as Legal’s control. ",[203,658,661],{"href":659,"rel":660},"https://code.claude.com/docs/en/hooks",[207],"Anthropic hooks"," vs Cursor rules vs Codex — pick for ",[224,664,665],{},"your"," repos.",[190,668,669,672,673,676,677,302],{},[194,670,671],{},"Week 4 — outer bake-off only for SoR jobs."," Run the refuse/replay script from ",[203,674,675],{"href":528},"how to evaluate an agent harness",". Include Nimbus, AIP, Agentforce, or a LangGraph programme as fits the staffing model. ",[203,678,493],{"href":492},[190,680,681,684,685,687],{},[194,682,683],{},"Do not"," hold week 3 until week 4 ships. Engineers will adopt inner tools anyway; you will only lose the chance to standardise hooks. ",[194,686,683],{}," skip week 4 because week 3’s coding agent “can also call Salesforce.” That is the anti-pattern.",[190,689,690],{},"Budget: copilot seats (predictable, personal); coding harness seats or usage (developer count); enterprise harness by work, not by mailbox count if you care about routing. Mixing all three into one “AI budget” is how flagship models burn on classify and how CRM writes go unquoted to save a line item.",[190,692,693,694,698],{},"Thoughtworks’ ",[203,695,697],{"href":244,"rel":696},[207],"organisational harness"," is the steering cadence after purchase: incidents become controls across both inner and outer. Buy tools that allow that ratchet. A coding harness that forbids custom hooks, or an OS that forbids adding a gate without FDE, will stall week 5.",[190,700,701,702,705,706,711,712,715],{},"Expect political arguments that are actually workspace arguments. Engineering will say the OS is slow. They are right for a one-line refactor. RevOps will say Cursor is unsafe. They are right for a production Opportunity. The CISO will say “one approved agent.” Translate: one ",[224,703,704],{},"write rule",", many loops. ",[203,707,710],{"href":708,"rel":709},"https://eur-lex.europa.eu/eli/reg/2024/1689/oj",[207],"EU AI Act"," oversight can be satisfied per system of use, not per brand. ",[203,713,400],{"href":398,"rel":714},[207]," Map is the same advice.",[190,717,718,719,722],{},"If budget forces a single purchase this half, buy the loop that matches the ",[224,720,721],{},"highest-harm"," unfinished job. Ungoverned CRM writes usually outrank “we could use a better coding agent” — paste already exists; unsigned APIs are new blast radius. If the highest-harm job is shipping software and SoR writes are still human, buy the coding harness and freeze the write rule until the outer product lands. Either way, write the rule down before the PO.",[190,724,725],{},"Nimbus should win the outer row on self-service quoting and graph export, and should lose the inner row on purpose. If a bake-off ranks us against Claude Code on SWE-bench, the scorecard is wrong. If it ranks us against a copilot on mail quality, also wrong. Rank us against AIP and Agentforce on the refuse/replay script, and against “we’ll build LangGraph” on time-to-first-governed-write.",[190,727,728,729,732,733,736,737,740,741,745],{},"The copilot row still matters. People will keep ",[203,730,313],{"href":311,"rel":731},[207]," for drafts. That is healthy if the write path is the easy official one. Banning unofficial ",[224,734,735],{},"drafts"," usually fails; making unofficial ",[224,738,739],{},"writes"," fail-closed usually works. ",[203,742,744],{"href":743},"what-is-shadow-ai","Shadow AI"," is often a write-path problem wearing a chat-policy costume.",[257,747,749],{"id":748},"questions-people-actually-ask","Questions people actually ask",[751,752,754],"h3",{"id":753},"we-already-paid-for-github-copilot","We already paid for GitHub Copilot.",[190,756,757],{},"That is often a completion copilot, not a full coding harness. You may still want Claude Code or Cursor for agentic repo work. Evaluate hooks and tests, not the seat.",[751,759,761],{"id":760},"can-the-enterprise-harness-include-a-coding-specialist","Can the enterprise harness include a coding specialist?",[190,763,764,765,768,769,302],{},"Yes, as a ",[224,766,767],{},"bounded tool"," that opens a draft PR. The SoR write still quotes in the outer harness. Specialists are hands. ",[203,770,772],{"href":771},"agent-team-architecture","Agent teams",[751,774,776],{"id":775},"what-if-legal-wants-one-vendor","What if Legal wants one vendor?",[190,778,779],{},"One vendor for identity and logging is reasonable. One vendor for repo loop and CRM loop is how you get a mediocre both. Prefer two harnesses and one interceptor rule: unsigned SoR writes are impossible everywhere.",[751,781,783],{"id":782},"how-do-we-score-nimbus-vs-claude-code-in-a-bake-off","How do we score Nimbus vs Claude Code in a bake-off?",[190,785,786],{},"Different jobs. Run inner tests on a repo. Run outer tests on a quoted CRM write. A combined “winner” is a category error unless you only have one job.",[751,788,790],{"id":789},"what-should-i-read-next","What should I read next?",[190,792,793,796,797,799,800,802],{},[203,794,795],{"href":215},"Inner vs outer"," for architecture. ",[203,798,529],{"href":528}," for the live tests. ",[203,801,301],{"href":300}," for the outer object.",[257,804,806],{"id":805},"related-reading","Related reading",[190,808,809,812,813,302],{},[203,810,811],{"href":220},"How to choose between a copilot and a work OS"," and ",[203,814,816],{"href":815},"build-vs-buy-an-enterprise-ai-os","Build vs buy an enterprise AI OS",[257,818,820],{"id":819},"sources","Sources",[262,822,823,829,835,841,847,853,859,865,870,875,881,887,893,899,905,911],{},[265,824,825],{},[203,826,828],{"href":205,"rel":827},[207],"LangChain, Agents",[265,830,831],{},[203,832,834],{"href":232,"rel":833},[207],"Böckeler, Harness engineering for coding agent users",[265,836,837],{},[203,838,840],{"href":238,"rel":839},[207],"Addy Osmani, Own the outer loop",[265,842,843],{},[203,844,846],{"href":244,"rel":845},[207],"Thoughtworks, The operating system for enterprise AI",[265,848,849],{},[203,850,852],{"href":377,"rel":851},[207],"Anthropic, Effective harnesses for long-running agents",[265,854,855],{},[203,856,858],{"href":321,"rel":857},[207],"Anthropic, Claude for Work",[265,860,861],{},[203,862,864],{"href":311,"rel":863},[207],"OpenAI, ChatGPT Enterprise",[265,866,867],{},[203,868,318],{"href":316,"rel":869},[207],[265,871,872],{},[203,873,284],{"href":282,"rel":874},[207],[265,876,877],{},[203,878,880],{"href":288,"rel":879},[207],"Terminal-Bench (arXiv:2601.11868)",[265,882,883],{},[203,884,886],{"href":252,"rel":885},[207],"McKinsey, The state of AI in 2025",[265,888,889],{},[203,890,892],{"href":418,"rel":891},[207],"Stanford HAI, 2025 AI Index",[265,894,895],{},[203,896,898],{"href":398,"rel":897},[207],"NIST AI RMF",[265,900,901],{},[203,902,904],{"href":424,"rel":903},[207],"ISO/IEC 42001",[265,906,907],{},[203,908,910],{"href":462,"rel":909},[207],"OWASP Top 10 for LLM applications",[265,912,913],{},[203,914,917],{"href":915,"rel":916},"https://modelcontextprotocol.io/specification/2025-11-25/index",[207],"Model Context Protocol specification",{"title":170,"searchDepth":171,"depth":171,"links":919},[920,921,922,923,924,925,926,934,935],{"id":259,"depth":171,"text":260},{"id":364,"depth":171,"text":365},{"id":430,"depth":171,"text":431},{"id":496,"depth":171,"text":497},{"id":545,"depth":171,"text":546},{"id":629,"depth":171,"text":630},{"id":748,"depth":171,"text":749,"children":927},[928,930,931,932,933],{"id":753,"depth":929,"text":754},3,{"id":760,"depth":929,"text":761},{"id":775,"depth":929,"text":776},{"id":782,"depth":929,"text":783},{"id":789,"depth":929,"text":790},{"id":805,"depth":171,"text":806},{"id":819,"depth":171,"text":820},"2026-08-24","A coding harness runs a repository — Claude Code, Cursor, Codex. An enterprise harness runs company jobs with connectors and signers. Most organisations need both; they are not substitutes.","/blog/how-to-choose-between-a-coding-harness-and-an-enterprise-harness",{"title":181,"description":937},"evaluation","blog/how-to-choose-between-a-coding-harness-and-an-enterprise-harness",[940,943,944,945],"agent-harness","coding-agents","enterprise-ai","zypj0rFuSxQgNAtRx0gY8CyrewpGlXGHc6zrzz32V4g",{"id":948,"title":949,"archived":164,"authors":950,"badge":952,"body":953,"date":936,"definedTerm":165,"department":165,"description":1602,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1603,"relatedHeading":165,"seo":1604,"series":940,"sitemap":130,"status":165,"stem":1605,"subhead":165,"tags":1606,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1608},"content/blog/how-to-evaluate-an-agent-harness.md","How to Evaluate an Agent Harness",[951],{"name":184,"to":135},{"label":186},{"type":167,"value":954,"toc":1583},[955,963,982,998,1000,1065,1073,1077,1080,1110,1121,1124,1128,1141,1160,1177,1188,1201,1212,1224,1235,1246,1250,1276,1286,1290,1296,1305,1308,1312,1315,1324,1334,1345,1355,1364,1372,1380,1393,1400,1403,1411,1415,1435,1437,1441,1444,1448,1453,1457,1460,1464,1467,1469,1485,1487,1497,1499],[190,956,957,958,962],{},"Evaluating an ",[203,959,961],{"href":960},"what-is-an-agent-harness","agent harness"," is checking whether the runtime around the model can finish a job under a stop you trust — not whether a demo answered a question.",[190,964,965,969,970,974,975,978,979,302],{},[203,966,968],{"href":205,"rel":967},[207],"LangChain"," defines the object: Agent = Model + Harness. The scoring sheet is therefore about the harness. If your RFP starts with context-window size and SWE-bench, you are scoring a model (and maybe an inner coding loop). You will miss whether an unsigned Salesforce PATCH is possible. ",[203,971,973],{"href":972},"how-to-evaluate-an-enterprise-ai-operating-system","How to evaluate an enterprise AI OS"," is the cousin sheet for wiki, workstreams, and routing as a ",[224,976,977],{},"product category",". This page is the runtime tests that apply to Claude Code, a LangGraph deployment, AIP, Agentforce, and Nimbus alike — then specialised by ",[203,980,981],{"href":215},"inner vs outer",[190,983,984,987,988,993,994,997],{},[203,985,254],{"href":252,"rel":986},[207]," already measured the trap: widespread use, limited scale. A fluent demo produces the first. A harness that can refuse, replay, and ratchet produces the second. ",[203,989,992],{"href":990,"rel":991},"https://www.nist.gov/itl/ai-risk-management-framework/nist-ai-rmf-playbook",[207],"NIST’s AI RMF Playbook"," is the measurement language. ",[203,995,904],{"href":424,"rel":996},[207]," is the management-system language. Neither is “the model seemed careful.”",[257,999,260],{"id":259},[262,1001,1002,1011,1026,1037,1045,1055],{},[265,1003,1004,1007,1008,1010],{},[194,1005,1006],{},"Harness vs framework."," Library versus running loop. ",[203,1009,334],{"href":333},". “We use LangChain” is not a passed test.",[265,1012,1013,1016,1017,1020,1021,1025],{},[194,1014,1015],{},"Sensor."," Independent check. ",[203,1018,234],{"href":232,"rel":1019},[207],"; ",[203,1022,246],{"href":1023,"rel":1024},"https://www.thoughtworks.com/en-us/insights/blog/generative-ai/harness-engineering-agent-feedback-exploring-ai-coding-sensors",[207],". Inner: tests. Outer: quote vs SoR.",[265,1027,1028,1031,1032,1036],{},[194,1029,1030],{},"Hook / interceptor."," Always runs. ",[203,1033,1035],{"href":659,"rel":1034},[207],"Claude Code hooks",". Outer: fail-closed adapter.",[265,1038,1039,1042,1043,302],{},[194,1040,1041],{},"Quote."," Structured payload, not a paragraph. ",[203,1044,390],{"href":389},[265,1046,1047,1050,1051,302],{},[194,1048,1049],{},"Replay."," Can you reconstruct signer, policy version, tool grants. ",[203,1052,1054],{"href":1053},"what-is-a-lifecycle-graph","Lifecycle graph",[265,1056,1057,1060,1061,302],{},[194,1058,1059],{},"Model portability."," Swap weights without rewriting tools. Not a logo on a slide. ",[203,1062,1064],{"href":1063},"what-is-model-routing","Model routing",[190,1066,1067,1068,812,1070,1072],{},"When you evaluate Nimbus, run these tests on ",[203,1069,350],{"href":32},[203,1071,356],{"href":40},", not on a homepage video. When you evaluate Claude Code, run them on a repo hook and CI, not on a blog SWE-bench screenshot. Same sheet, different workspace.",[257,1074,1076],{"id":1075},"why-evaluation-usually-fails","Why evaluation usually fails",[190,1078,1079],{},"People score agents like they score chat: quality of the paragraph, latency, brand of the model. That produces three false passes:",[1081,1082,1083,1092,1100],"ol",{},[265,1084,1085,1088,1089,302],{},[194,1086,1087],{},"The copilot pass."," SSO, a usage dashboard, a good answer. No loop ownership. ",[203,1090,1091],{"href":220},"Copilot vs work OS",[265,1093,1094,1097,1098,302],{},[194,1095,1096],{},"The benchmark pass."," SWE-bench or Terminal-Bench for an outer job. Inner eval, outer purchase. ",[203,1099,453],{"href":452},[265,1101,1102,1105,1106,1109],{},[194,1103,1104],{},"The framework pass."," A graph in a notebook with every production tool attached. ",[203,1107,464],{"href":462,"rel":1108},[207]," excessive agency with extra nodes.",[190,1111,1112,1117,1118,302],{},[203,1113,1116],{"href":1114,"rel":1115},"https://www.anthropic.com/engineering/building-effective-agents",[207],"Anthropic"," is blunt: encode the job, bound the tools, define done. Your proof of value should force those three. Written answers without a failed action are still a slide. ",[203,1119,1120],{"href":488},"How to run an enterprise AI proof of value",[190,1122,1123],{},"Red flags: chat as the entire proof; “we integrate” with no scoped grant; governance as PDF; memory as a long window; “model-agnostic” with a flagship default and seat pricing; MCP write tools that inherit a god service account; vendor database offered as the new system of record.",[257,1125,1127],{"id":1126},"checklist","Checklist",[190,1129,1130,471,1133,1136,1137,302],{},[194,1131,1132],{},"1. Can it stop an action the model wants?",[272,1134,1135],{},"PreToolUse"," denies a matched command; tests fail the merge. Outer: unsigned write does not execute; reject leaves SoR unchanged. If the only stop is max tokens, you have a fuse, not a control plane. ",[203,1138,1140],{"href":1139},"human-in-the-loop-approval-architecture","HITL architecture",[190,1142,1143,1144,1149,1150,1155,1156,1159],{},"Why this matters: ",[203,1145,1148],{"href":1146,"rel":1147},"https://www.cbc.ca/news/canada/british-columbia/air-canada-chatbot-lawsuit-1.7116416",[207],"Air Canada"," and the ",[203,1151,1154],{"href":1152,"rel":1153},"https://www.reuters.com/legal/new-york-lawyers-sanctioned-using-fake-chatgpt-cases-legal-brief-2023-06-22/",[207],"sanctioned ChatGPT brief"," are ungated generation reaching a record. Your demo must show a ",[224,1157,1158],{},"failed"," write.",[190,1161,1162,1165,1166,1168,1169,1173,1174,1176],{},[194,1163,1164],{},"2. Can you replay who signed and which harness version ran?"," Signer identity, wiki or ",[272,1167,274],{}," revision, tool grants, payload hash, model class. If the answer is Slack search or “the transcript,” you do not have a ledger. ",[203,1170,1172],{"href":1171},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",". Nimbus’s ",[203,1175,23],{"href":24}," is one implementation; demand the export without a vendor engineer.",[190,1178,1179,1182,1183,1187],{},[194,1180,1181],{},"3. Can you swap the model without rewriting tools?"," Change compact vs frontier on extract vs judgement. If tools are bound to one vendor’s function-calling dialect in application code with no adapter, portability is a hope. ",[203,1184,1186],{"href":205,"rel":1185},[207],"LangChain’s model interface"," exists for this; product harnesses must expose it as policy, not as a rewrite.",[190,1189,1190,1193,1194,1198,1199,302],{},[194,1191,1192],{},"4. Are tools grants or a belt?"," Least privilege per job. Missing Salesforce is a configuration error, not a hallucination. ",[203,1195,1197],{"href":1196},"connector-and-permissions-architecture","Connector architecture",". MCP servers inherit the same grant. ",[203,1200,344],{"href":343},[190,1202,1203,1206,1207,1211],{},[194,1204,1205],{},"5. Is verification outside the generator?"," Inner: CI the agent cannot mark skip without a hook. Outer: schema of the quote; SoR row matches. Anthropic’s ",[203,1208,1210],{"href":377,"rel":1209},[207],"long-running harness"," refuses “premature victory” by forcing artefacts and tests. Steal that instinct.",[190,1213,1214,1217,1218,1221,1222,302],{},[194,1215,1216],{},"6. Can an operator add a sensor without a six-month SOW?"," ",[203,1219,1220],{"href":477},"Harness engineering"," is a ratchet. If only vendor FDE can add a gate, you bought a programme. Fine for AIP-scale. Wrong for a standard CRM field this quarter. ",[203,1223,493],{"href":492},[190,1225,1226,1229,1230,1234],{},[194,1227,1228],{},"7. Is the workspace the job you are buying?"," Repo vs company. ",[203,1231,1233],{"href":1232},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose coding vs enterprise",". A single scoring sheet with no workspace column will buy the wrong loop.",[190,1236,1237,1240,1241,1245],{},[194,1238,1239],{},"8. Economics of the loop."," Max steps, spend cap, routing. Seat “unlimited” is often always-flagship. ",[203,1242,1244],{"href":1243},"ai-cost-control-architecture","AI cost control architecture",". Ask for a per-step model breakdown on a live run.",[751,1247,1249],{"id":1248},"rfp-questions","RFP questions",[1081,1251,1252,1255,1258,1261,1264,1267,1270,1273],{},[265,1253,1254],{},"Show an action the model attempted that the harness refused. What fired?",[265,1256,1257],{},"After a successful write (or merge), show the signer, policy version, and payload (or diff) without Slack.",[265,1259,1260],{},"Change the model on extract this week. Which tools broke?",[265,1262,1263],{},"Attach a connector (or repo permission) as an operator, not as SE. Time?",[265,1265,1266],{},"Detach the grant mid-job. Does the write fail closed?",[265,1268,1269],{},"What is the independent sensor for “done”? Who can mark skip?",[265,1271,1272],{},"Two departments, different scopes, one job — or one god toolbox?",[265,1274,1275],{},"Price: seats, tokens, NTUs, or a services quote? What stops flagship on classify?",[190,1277,1278,1279,391,1281,1285],{},"Put these in the RFP, then run them in a ",[203,1280,489],{"href":488},[203,1282,1284],{"href":1283},"rfp-questions-for-enterprise-ai-agents","RFP questions for enterprise AI agents"," overlaps; keep both. Agents without a harness test are a persona list.",[751,1287,1289],{"id":1288},"proof-of-value-short","Proof of value (short)",[190,1291,1292,1295],{},[194,1293,1294],{},"Inner job:"," real repo, required hook, red test the agent must fix, no production SoR token.",[190,1297,1298,1301,1302,1304],{},[194,1299,1300],{},"Outer job:"," real cross-department write, quoted payload, reject path, export. Nimbus should pass the same live sequence as anyone else: OAuth attach, blocked unsigned write, graph export. ",[203,1303,11],{"href":12}," is not the proof.",[190,1306,1307],{},"Skip any refuse/replay/swap and you evaluated a chat product, a benchmark, or a framework notebook.",[257,1309,1311],{"id":1310},"score-inner-and-outer-without-mixing-oracles","Score inner and outer without mixing oracles",[190,1313,1314],{},"Run two short scripts. Do not average them into one “AI score.”",[190,1316,1317,1320,1321,1323],{},[194,1318,1319],{},"Inner script (repo)."," Fresh checkout of a service you own. Required hook: deny a dangerous bash pattern. Agent must add a failing test then make it pass. CI is the merge sensor. No production CRM token in the environment. Record: did the hook fire, did CI stay independent, can you show the ",[272,1322,274],{}," revision. SWE-bench plots from the vendor are background, not this script.",[190,1325,1326,1329,1330,1333],{},[194,1327,1328],{},"Outer script (SoR)."," Sandbox Salesforce or equivalent. Operator (not SE) attaches OAuth. Model proposes a write. Unsigned path must fail. Reject path must leave records unchanged. Approve path: read-back matches hash. Export signer and wiki revision. Detach the connector and retry the write — must fail closed. ",[203,1331,1332],{"href":488},"Proof of value"," is this script with two departments on the canvas.",[190,1335,1336,1337,812,1339,1341,1342,302],{},"If a vendor refuses to run the outer script because “we are a coding tool,” believe them and buy them for inner only. If a vendor refuses the inner script because “we are an OS,” believe them and do not replace Cursor. If a vendor claims both and fails one script, you have a category error in their marketing. Nimbus should pass the outer script on ",[203,1338,350],{"href":32},[203,1340,356],{"href":40},". Claude Code should pass the inner script. ",[203,1343,1344],{"href":1232},"How to choose",[190,1346,1347,1350,1351,1354],{},[194,1348,1349],{},"Thoughtworks’ layer check."," After the scripts, ask where layer 4 lives: who owns the policy when the agent did what it was allowed to do and harm still happened. If the answer is a steering committee with no interceptor, you evaluated theatre. ",[203,1352,426],{"href":424,"rel":1353},[207]," will not save a missing refuse.",[190,1356,1357,1360,1361,302],{},[194,1358,1359],{},"Economics check."," Pull one live run’s step list: model class per step, tokens or NTUs, which sensor fired. Always-flagship with no cap is a failed harness eval even if the paragraph was good. ",[203,1362,1363],{"href":1243},"Cost control",[190,1365,1366,1369,1370,302],{},[194,1367,1368],{},"MCP check."," One write-capable server. Which workspaces may use it. If the answer is “any host that can see the URL,” fail. ",[203,1371,344],{"href":343},[190,1373,1374,1375,1379],{},"Weight the eight checklist items; do not add a ninth called “brand.” ",[203,1376,1378],{"href":418,"rel":1377},[207],"Stanford AI Index"," is useful context for how fast coding tools moved. It is not a substitute for the outer script.",[190,1381,1382,1383,1388,1389,1392],{},"Score vendors as systems, not as essays. A beautiful ",[203,1384,1387],{"href":1385,"rel":1386},"https://www.langchain.com/blog/the-anatomy-of-an-agent-harness",[207],"anatomy post"," does not pass the refuse test. A messy UI that blocks the unsigned PATCH does. Watch for “evaluation theatre”: the SE runs the happy path, the fail path is “we’ll configure that in phase two,” the ledger is a screenshot of LangSmith. Phase two is where ",[203,1390,640],{"href":252,"rel":1391},[207]," pilots go to die.",[190,1394,1395,1396,1399],{},"Bring your own oracle. For inner: a test the agent did not write. For outer: a sandbox row you control. If the vendor must supply the only success criterion, you are scoring their demo fixtures. Terminal-Bench’s strength is that the ",[224,1397,1398],{},"environment"," is the grader. Copy that.",[190,1401,1402],{},"People on the bake-off: an operator who will live in the product, someone who owns the SoR, someone who can say no for Legal, an engineer who will keep the inner harness. If only the vendor and an innovation lead attend, you will buy a narrative. Nimbus, AIP, Cursor, and a LangGraph SOW should all survive that room or be narrowed to the job they actually do.",[190,1404,1405,1406,1410],{},"Write the pass/fail before the demo so the SE cannot redefine success live. “Blocked unsigned write” is a boolean. “Felt enterprise-ready” is not. Record the session. If they cannot fail on camera, assume they cannot fail in production. ",[203,1407,1409],{"href":990,"rel":1408},[207],"NIST Playbook"," language helps here: you are Measuring a control, not a vibe.",[257,1412,1414],{"id":1413},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[190,1416,1417,1418,1421,1422,1425,1426,1428,1429,1431,1432,1434],{},"Nimbus is an ",[203,1419,1420],{"href":300},"enterprise / outer harness",": ",[203,1423,1424],{"href":28},"wiki"," as guides, connectors as grants, ",[203,1427,353],{"href":20}," as the hiring object, ",[203,1430,356],{"href":40}," as the interceptor, graph as replay, ",[203,1433,621],{"href":45}," as routing. Score those surfaces against the eight tests. Do not accept “we are a harness” as a substitute for a failed write. AIP and Agentforce deserve the same eight.",[257,1436,749],{"id":748},[751,1438,1440],{"id":1439},"can-we-score-claude-code-and-nimbus-on-one-spreadsheet","Can we score Claude Code and Nimbus on one spreadsheet?",[190,1442,1443],{},"Yes, with a workspace column. Shared rows: refuse, replay, swap, sensors, operator change, economics. Inner-only rows: tests, sandbox, PR. Outer-only rows: SoR quote, roster signer, workstream isolation.",[751,1445,1447],{"id":1446},"the-vendor-sent-a-swe-bench-plot","The vendor sent a SWE-bench plot.",[190,1449,1450,1451,302],{},"File it under inner quality. If you are buying CRM writes, it is not sufficient. ",[203,1452,453],{"href":452},[751,1454,1456],{"id":1455},"we-already-completed-a-copilot-rfp","We already completed a copilot RFP.",[190,1458,1459],{},"Keep it for personal tools. This sheet is for loops that act. Different job.",[751,1461,1463],{"id":1462},"is-iso-42001-certification-the-eval","Is ISO 42001 certification the eval?",[190,1465,1466],{},"It is a management-system signal. Still watch a write fail. Certification without an interceptor is paperwork.",[751,1468,790],{"id":789},[190,1470,1471,1475,1476,1480,1481,1484],{},[203,1472,1474],{"href":1473},"agent-harness-architecture","Agent harness architecture"," to know the parts. ",[203,1477,1479],{"href":1478},"how-to-evaluate-write-back-governance","How to evaluate write-back governance"," for the outer stop in detail. ",[203,1482,1483],{"href":477},"What is harness engineering"," for the ratchet after you buy.",[257,1486,806],{"id":805},[190,1488,1489,812,1493,302],{},[203,1490,1492],{"href":1491},"how-to-evaluate-multi-agent-platforms","How to evaluate multi-agent platforms",[203,1494,1496],{"href":1495},"how-to-evaluate-ai-governance-platforms","How to evaluate AI governance platforms",[257,1498,820],{"id":819},[262,1500,1501,1506,1512,1517,1523,1529,1534,1540,1545,1551,1556,1561,1567,1573,1578],{},[265,1502,1503],{},[203,1504,828],{"href":205,"rel":1505},[207],[265,1507,1508],{},[203,1509,1511],{"href":1385,"rel":1510},[207],"LangChain, The anatomy of an agent harness",[265,1513,1514],{},[203,1515,834],{"href":232,"rel":1516},[207],[265,1518,1519],{},[203,1520,1522],{"href":1023,"rel":1521},[207],"Thoughtworks, Harness engineering and agent feedback",[265,1524,1525],{},[203,1526,1528],{"href":1114,"rel":1527},[207],"Anthropic, Building effective agents",[265,1530,1531],{},[203,1532,852],{"href":377,"rel":1533},[207],[265,1535,1536],{},[203,1537,1539],{"href":659,"rel":1538},[207],"Claude Code, Hooks",[265,1541,1542],{},[203,1543,886],{"href":252,"rel":1544},[207],[265,1546,1547],{},[203,1548,1550],{"href":990,"rel":1549},[207],"NIST AI RMF Playbook",[265,1552,1553],{},[203,1554,904],{"href":424,"rel":1555},[207],[265,1557,1558],{},[203,1559,910],{"href":462,"rel":1560},[207],[265,1562,1563],{},[203,1564,1566],{"href":1146,"rel":1565},[207],"CBC, Air Canada chatbot lawsuit",[265,1568,1569],{},[203,1570,1572],{"href":1152,"rel":1571},[207],"Reuters, ChatGPT legal brief sanctions",[265,1574,1575],{},[203,1576,892],{"href":418,"rel":1577},[207],[265,1579,1580],{},[203,1581,917],{"href":915,"rel":1582},[207],{"title":170,"searchDepth":171,"depth":171,"links":1584},[1585,1586,1587,1591,1592,1593,1600,1601],{"id":259,"depth":171,"text":260},{"id":1075,"depth":171,"text":1076},{"id":1126,"depth":171,"text":1127,"children":1588},[1589,1590],{"id":1248,"depth":929,"text":1249},{"id":1288,"depth":929,"text":1289},{"id":1310,"depth":171,"text":1311},{"id":1413,"depth":171,"text":1414},{"id":748,"depth":171,"text":749,"children":1594},[1595,1596,1597,1598,1599],{"id":1439,"depth":929,"text":1440},{"id":1446,"depth":929,"text":1447},{"id":1455,"depth":929,"text":1456},{"id":1462,"depth":929,"text":1463},{"id":789,"depth":929,"text":790},{"id":805,"depth":171,"text":806},{"id":819,"depth":171,"text":820},"Evaluating an agent harness means checking whether it can stop a write, replay who signed, swap the model without rewriting tools, and fail a real sensor — not whether the demo answered a question.","/blog/how-to-evaluate-an-agent-harness",{"title":949,"description":1602},"blog/how-to-evaluate-an-agent-harness",[940,943,356,1607],"rfp","Dqpi4SUIjqo12Fpx1yo1H3653ZFcjYcBCFJYPELZbxU",{"id":1610,"title":1611,"archived":164,"authors":1612,"badge":1614,"body":1615,"date":1839,"definedTerm":165,"department":165,"description":1840,"extension":173,"eyebrow":165,"faqHeader":1841,"faqs":1844,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1857,"relatedHeading":165,"seo":1858,"series":940,"sitemap":130,"status":165,"stem":1859,"subhead":165,"tags":1860,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1862},"content/blog/what-auditors-are-asking-for.md","What are auditors asking for around AI?",[1613],{"name":184,"to":135},{"label":186},{"type":167,"value":1616,"toc":1832},[1617,1620,1623,1640,1643,1646,1652,1658,1661,1664,1667,1674,1677,1681,1684,1692,1700,1707,1714,1722,1725,1729,1732,1735,1738,1755,1758,1761,1764,1772,1776,1779,1799,1806,1809,1812,1816,1819,1822,1825],[190,1618,1619],{},"Auditors asking about AI usually want to follow one change: who decided, whether software could write without a person, and which rulebook you claim to follow. They pick a journal, a credit, a customer email, or a model connection, and they walk it from prompt to record.",[190,1621,1622],{},"This is showing up now because models sit on live systems, and existing control texts already care how a number became the number. You do not need every framework on day one. You need artefacts you can produce without asking anyone to remember.",[190,1624,1625,1626,1630,1631,1635,1636,1639],{},"This guide is a first evidence pack you can start this quarter. ",[203,1627,1629],{"href":1628},"what-is-ai-governance","What is AI governance"," is the rest of the access picture. ",[203,1632,1634],{"href":1633},"rbac-for-enterprise-ai","RBAC for enterprise AI"," is who may see the job. ",[203,1637,1638],{"href":389},"Write-back governance"," is the write checklist.",[257,1641,1611],{"id":1642},"what-are-auditors-asking-for-around-ai",[190,1644,1645],{},"Two operational questions arrive first.",[190,1647,1648,1651],{},[194,1649,1650],{},"Can you show who decided?"," A named person, on a clock the company trusts, bound to a quote that matches the write. “The team aligned” is not an answer. “The channel approved” is not an answer. “The bot user posted” is not an answer.",[190,1653,1654,1657],{},[194,1655,1656],{},"Can you show the model did not write unchecked?"," Write-back means the AI changes a live system. Fail-closed means if nobody approves, nothing happens. A prompt that says “ask first” is not the gate. A weekly sampling of logs is not the gate if the write already landed.",[190,1659,1660],{},"Role-based access control (RBAC) means who is allowed to do what. It explains why that person, and not a guest, was offered the button. Auditors understand roles. They do not understand “the workspace.”",[190,1662,1663],{},"A payload is the exact change: fields, old and new values, target record — or the exact text and recipient for a message.",[190,1665,1666],{},"Then comes the mapping question: which framework applies to us? Not every company is under every text. Pretending otherwise produces a pile of mappings and no artefact.",[190,1668,1669,1673],{},[203,1670,1672],{"href":1671},"collaborative-ai-for-legal-and-compliance-review","Collaborative AI for legal and compliance review"," still needs a signer when the review becomes a filing. Several departments on one job is not a shared identity.",[190,1675,1676],{},"If the decision was “we will not write,” that is still a decision. Store it. A read-only connector with a date and an owner is evidence.",[257,1678,1680],{"id":1679},"why-is-this-showing-up-now","Why is this showing up now?",[190,1682,1683],{},"Models are in the path of records that already had auditors: financial reporting, customer commitments, legal filings, operational tickets.",[190,1685,1686,1691],{},[203,1687,1690],{"href":1688,"rel":1689},"https://www.govinfo.gov/content/pkg/PLAW-107publ204/html/PLAW-107publ204.htm",[207],"Sarbanes-Oxley"," (2002) is still the text many US-listed teams feel first. Internal control over financial reporting does not care that the proposer is a model. If AI can post, the control environment includes that path.",[190,1693,1694,1695,1699],{},"NIST’s ",[203,1696,1698],{"href":398,"rel":1697},[207],"AI Risk Management Framework"," (2023) is organised as govern, map, measure, and manage. Measure, here, is the stored outcome, including the no. Govern is the roles and the owners. The framework will not click the refuse button for you.",[190,1701,1702,1706],{},[203,1703,904],{"href":1704,"rel":1705},"https://www.iso.org/standard/81230.html",[207]," (2023) adds a management system for AI: policies, roles, risk assessment, documented processes, and evidence that those processes run. Useful if you will be asked for a certificate. Not a substitute for a payload screen.",[190,1708,1709,1710,1713],{},"The ",[203,1711,710],{"href":708,"rel":1712},[207]," (2024/1689) is from 2024. A deployer is the organisation that uses an AI system under its authority, as the Act defines that role. You may also be a provider if you place a system on the market. Map the role with counsel. Human oversight that cannot refuse a write is not oversight.",[190,1715,1716,1721],{},[203,1717,1720],{"href":1718,"rel":1719},"https://eur-lex.europa.eu/eli/reg/2022/2554/oj",[207],"DORA"," (2022) is about digital operational resilience for financial entities and their ICT third parties. If you are in that sector, the AI vendor is an ICT provider conversation, not only an innovation conversation.",[190,1723,1724],{},"Customers and boards ask for structure even when a text is voluntary. That is why the questions arrive before a regulator has written your company’s name.",[257,1726,1728],{"id":1727},"how-do-you-prepare-evidence-without-a-huge-project","How do you prepare evidence without a huge project?",[190,1730,1731],{},"Do the one-change walk before anyone external does.",[190,1733,1734],{},"Pick a change a model proposed. Follow it from prompt to record. See whether you can produce a named person, a frozen payload, and a stored outcome without anyone’s memory.",[190,1736,1737],{},"Show:",[262,1739,1740,1743,1746,1749,1752],{},[265,1741,1742],{},"A connector in read-only mode, and a failed write attempt.",[265,1744,1745],{},"One object class with a frozen payload and a named signer — or a dated decision that no class is enabled yet.",[265,1747,1748],{},"The live system’s own validation still firing, if a write ran.",[265,1750,1751],{},"A success and a rejection.",[265,1753,1754],{},"A person who was removed and could not sign the next day.",[190,1756,1757],{},"If you cannot show the failed attempt, assume an auditor will treat write as on.",[190,1759,1760],{},"Unchecked also includes send. A customer message is a write to the relationship. If mail can go out because the connector was on for retrieval, that is an unchecked write with no field names to screenshot.",[190,1762,1763],{},"Do not start with a coverage matrix against every clause. Breadth without a sample fails the first request. Depth on one change lets you map the same artefact twice if two texts apply.",[190,1765,1766,1771],{},[203,1767,1770],{"href":1768,"rel":1769},"https://www.law.cornell.edu/rules/frcp/rule_37",[207],"Federal Rule of Civil Procedure 37(e)"," (2015) is about preserving electronically stored information you should have kept. Chat retention sliders are not that programme. Put approvals where a new manager can find them.",[257,1773,1775],{"id":1774},"what-is-a-reasonable-first-evidence-pack","What is a reasonable first evidence pack?",[190,1777,1778],{},"One page plus exports:",[262,1780,1781,1784,1787,1790,1793,1796],{},[265,1782,1783],{},"Job name, system, connector mode, date, owner.",[265,1785,1786],{},"Roster: guest, member, admin, signer — or “signer not yet named; write off.”",[265,1788,1789],{},"One stored refusal (sandbox is fine).",[265,1791,1792],{},"One stored success if you have enabled a class; otherwise omit.",[265,1794,1795],{},"Clock and retention note: where the artefact lives, how long, who can export it without a vendor ticket.",[265,1797,1798],{},"Which texts you claim: SOX ICFR if you file; NIST AI RMF as structure; ISO 42001 if you are on that path; EU AI Act role if in scope; DORA if you are a financial entity.",[190,1800,1801,1802,1805],{},"Your ",[203,1803,1804],{"href":109},"compliance"," programme should hold that page.",[190,1807,1808],{},"ISO 42001, if you take it seriously, adds an owner for AI, a statement of which systems models may connect to and in which mode, a way to handle incidents and model or prompt changes that alter write behaviour, and records that last longer than a chat default. It does not add object-level tokens. You can be certified and still have an admin token on a model. Ask the auditor of that management system to sample a stored rejection from a live job.",[190,1810,1811],{},"For deployers under the EU AI Act, the operational match is: know you are using AI, use it as intended, monitor, keep required records, and ensure human oversight where the Act requires it. “The vendor is the provider” does not move your ERP posting into their audit file. Your token, your records, your signer. High-risk classification is legal work. This guide will not guess it.",[257,1813,1815],{"id":1814},"how-do-you-start-this-quarter","How do you start this quarter?",[190,1817,1818],{},"This month: pick one real job. Run the one-change walk. Write the one-page pack. Fill blanks as findings, not as a reason to delay the page.",[190,1820,1821],{},"Next month: fix the first hole — usually the stored no, the read-only proof, or the named signer.",[190,1823,1824],{},"If you cannot complete the walk, keeping write off is the honest state of the control. Mapping will not replace it.",[190,1826,1827,812,1829,1831],{},[203,1828,1638],{"href":389},[203,1830,1634],{"href":1633}," are the two product habits that make the pack easier to gather later.",{"title":170,"searchDepth":171,"depth":171,"links":1833},[1834,1835,1836,1837,1838],{"id":1642,"depth":171,"text":1611},{"id":1679,"depth":171,"text":1680},{"id":1727,"depth":171,"text":1728},{"id":1774,"depth":171,"text":1775},{"id":1814,"depth":171,"text":1815},"2026-09-06","Who decided, did the model write unchecked, and which rulebook applies. How to prepare a first evidence pack this quarter without a huge project.",{"eyebrow":1842,"title":1843},"Short answers","One change you can walk",[1845,1848,1851,1854],{"question":1846,"answer":1847},"If we are not in the EU, can we ignore the AI Act?","You can ignore it as a legal duty only if you are not in its scope. You should still answer the same operational questions — who decided, and did the model write unchecked — because auditors and customers will ask them in other words.",{"question":1849,"answer":1850},"Does ISO 42001 certification mean our CRM writes are governed?","No. Certification speaks to a management system. It does not replace a named person on a payload, a stored rejection, or a connector that can be read-only. Ask to see those artefacts in your product, not only the certificate.",{"question":1852,"answer":1853},"What is the smallest evidence pack that still helps?","One change a model proposed: named person, frozen payload, stored outcome including a no, plus the connector mode and the roster on that job. Map that pack to whichever texts apply. Do not start with a matrix of empty controls.",{"question":1855,"answer":1856},"Do we need this if AI is still read-only?","A dated decision to stay read-only, with an owner and the connector name, is evidence. You need the full write pack before the first production write class.","/blog/what-auditors-are-asking-for",{"title":1611,"description":1840},"blog/what-auditors-are-asking-for",[940,1861,426,710,1804],"audit","EfFv_YZ08TZJm4MDym8B05aD4MGOwabpfOQ5EtWvDxw",{"id":1864,"title":1865,"archived":164,"authors":1866,"badge":1868,"body":1869,"date":2257,"definedTerm":165,"department":165,"description":2258,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":2259,"relatedHeading":165,"seo":2260,"series":940,"sitemap":130,"status":165,"stem":2261,"subhead":165,"tags":2262,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":2265},"content/blog/what-to-look-for-in-model-routing.md","What to Look for in Model Routing",[1867],{"name":184,"to":135},{"label":186},{"type":167,"value":1870,"toc":2240},[1871,1876,1879,1896,1898,1945,1956,1960,1963,1983,1990,2011,2017,2020,2024,2062,2072,2075,2079,2082,2110,2113,2115,2118,2125,2133,2135,2139,2145,2149,2152,2156,2159,2163,2170,2174,2181,2183,2195,2197],[190,1872,1873,1875],{},[194,1874,1064],{}," is the policy that maps a task class to a model class before inference runs. It is not a brand preference. It is not a dropdown labelled “best.”",[190,1877,1878],{},"Someone using the flagship model to label a ticket is how you pay frontier prices for work a compact model could have finished in a second. That is not a moral failing. It is a missing policy. The product either chooses before the call, or a person chooses in a menu, or the default is the largest model “for quality.” Only the first is routing.",[190,1880,1881,812,1886,1891,1892,1895],{},[203,1882,1885],{"href":1883,"rel":1884},"https://openai.com/api/pricing/",[207],"OpenAI’s API pricing",[203,1887,1890],{"href":1888,"rel":1889},"https://www.anthropic.com/pricing",[207],"Anthropic’s pricing"," make the same point in public: compact and frontier models are not the same invoice line. ",[203,1893,420],{"href":418,"rel":1894},[207]," has tracked how fast inference cost and capability moved — which is exactly why “best model” is not a routing policy. Best for a memo is not best for a classify step. Best last quarter is not best this quarter.",[257,1897,260],{"id":259},[262,1899,1900,1906,1912,1918,1929,1939],{},[265,1901,1902,1905],{},[194,1903,1904],{},"Frontier / flagship model."," The strongest (and usually most expensive) model a lab currently sells. Reserved for synthesis, hard reasoning, and novel language. Not for labelling.",[265,1907,1908,1911],{},[194,1909,1910],{},"Compact / small model."," Faster and cheaper. Often enough for extract, classify, and summarise. “Small” is a cost and latency class, not an insult.",[265,1913,1914,1917],{},[194,1915,1916],{},"Task class."," The kind of step: classify, retrieve, forecast, synthesise. If the platform cannot name the class, it cannot route. It can only default.",[265,1919,1920,1923,1924,1928],{},[194,1921,1922],{},"NTU."," A metered unit of useful work so you can quote and cap a loop. See ",[203,1925,1927],{"href":1926},"what-is-ai-token-economics","What is AI token economics",". Seats hide routing. NTU makes it visible.",[265,1930,1931,1934,1935,1938],{},[194,1932,1933],{},"Model-agnostic."," The platform can call more than one provider. That is a menu, not a policy, until it ",[224,1936,1937],{},"chooses"," by task class. Extra logos with a hidden flagship default is lock-in with branding.",[265,1940,1941,1944],{},[194,1942,1943],{},"Always-flagship."," Marketing for “we use the best model.” A classify job does not need a long-context reasoner. Quality theatre is a cost event.",[190,1946,1947,1948,1951,1952,1955],{},"Routing is also not ",[194,1949,1950],{},"fine-tuning"," (changing a model’s weights), and it is not ",[194,1953,1954],{},"orchestration"," (what steps exist). You can orchestrate a brilliant graph and still send every node to the flagship. You can fine-tune a compact model and still need a policy that sends classify there. Evaluate them separately.",[257,1957,1959],{"id":1958},"why-you-should-care","Why you should care",[190,1961,1962],{},"Teams do not wake up and choose waste. They inherit a default.",[262,1964,1965,1971,1977],{},[265,1966,1967,1970],{},[194,1968,1969],{},"Single-model shop."," Every label, every search, every memo calls the same flagship. Finance sees one invoice and cannot split labelling from reasoning. You cannot cap what you cannot see.",[265,1972,1973,1976],{},[194,1974,1975],{},"User-picked dropdown."," Power users pick the most expensive option “to be safe.” New hires copy that habit. Routing is now a training problem. Training problems do not survive quarter-end.",[265,1978,1979,1982],{},[194,1980,1981],{},"Always-flagship as quality theatre."," Best for whom? Best for a forecast interpolation is often a time-series path, not a frontier model inventing a number that looks fine in a short demo.",[190,1984,1985,1989],{},[203,1986,1988],{"href":252,"rel":1987},[207],"McKinsey’s 2025 State of AI survey"," keeps showing the operational gap: regular use, then a struggle to scale because cost and workflow were never treated as a system. Routing is that system for inference. Without it, scale is a token bill.",[190,1991,1992,1997,1998,2001,2002,812,2005,2010],{},[203,1993,1996],{"href":1994,"rel":1995},"https://www.gartner.com/en/articles/ai-governance-trism",[207],"Gartner’s AI TRiSM framing"," implies you can ",[224,1999,2000],{},"see"," which model ran, on which data, at what cost. A platform that cannot show that is not ready, regardless of its red-team slides. ",[203,2003,1720],{"href":1718,"rel":2004},[207],[203,2006,2009],{"href":2007,"rel":2008},"https://eur-lex.europa.eu/eli/dir/2022/2555/oj",[207],"NIS2"," change the evidence question: you should understand ICT dependencies. “We are not sure which model ran last Tuesday” is a dependency you cannot explain.",[190,2012,2013,2014,2016],{},"Choosing a model is choosing a brain. Choosing tools is choosing hands. Decide them separately. A compact model that extracts a refund still cannot write it without a quoted named signer if that is the workstream policy. Routing does not replace ",[203,2015,356],{"href":1628},". Governance does not replace routing. You need both.",[190,2018,2019],{},"Red flags: “we support many providers” with no task-class map; seat pricing that includes unlimited flagship; users pick the model in production; classify and memo share a model id in the demo; no NTU quote before a run expands; fallback is “switch the dropdown”; graph does not record model id per step; forecasting done by an LLM in the demo on a short series.",[257,2021,2023],{"id":2022},"what-to-look-for","What to look for",[262,2025,2026,2032,2038,2048,2054],{},[265,2027,2028,2031],{},[194,2029,2030],{},"They can refuse the flagship."," Run a classify-only job and show the model id. If classify used the same model as the memo, routing is a slide. Refusal is the proof. Support for many models is not.",[265,2033,2034,2037],{},[194,2035,2036],{},"Task-class map you can read."," Classify → compact. Retrieve → embeddings, not stuffing hundreds of tickets into a long window. Forecast → a time-series path, not a frontier model interpolating a spreadsheet. Reason → frontier. If they cannot name the classes, they cannot route them.",[265,2039,2040,2043,2044,302],{},[194,2041,2042],{},"NTU quotes before the run expands."," Operators see an estimate and can set a workstream cap. Seat licences hide routing. Unlimited flagship under a seat is always-flagship with a predictable opex line. See ",[203,2045,2047],{"href":2046},"total-cost-of-ownership-for-enterprise-ai","Total cost of ownership for enterprise AI",[265,2049,2050,2053],{},[194,2051,2052],{},"Fallback is a logged promotion",", not “users will switch the dropdown.” Compact models fail on novel schemas and policy-edge language. Temporarily raise that class, budget-aware, then revert. A promotion without a log is a silent cost change. A dropdown is a training problem.",[265,2055,2056,2059,2060,302],{},[194,2057,2058],{},"The graph records the model id per step."," Six months later you can answer “which model drafted this?” without grepping provider dashboards. That is audit as well as cost. See ",[203,2061,1172],{"href":1171},[190,2063,2064,2065,2068,2069,2071],{},"Ask for four artefacts from one workstream run: model id per step; NTU per task class; a classify job that did ",[224,2066,2067],{},"not"," use the flagship; a forecast that did ",[224,2070,2067],{}," use an LLM as the estimator. If the vendor can only show a chat transcript and a blended token total, routing is not in the product.",[190,2073,2074],{},"Why each artefact matters: model id is Measure in NIST language. NTU per class is how Finance splits labelling from reasoning. Classify-without-flagship is the refusal test. Forecast-without-LLM is whether they know the difference between narration and estimation. Demos are short series. Production is seasonality, holidays, and missing days.",[751,2076,2078],{"id":2077},"what-a-live-demo-should-prove","What a live demo should prove",[190,2080,2081],{},"Do not accept a provider logo wall.",[1081,2083,2084,2092,2095,2098,2101,2104,2107],{},[265,2085,2086,2087,2091],{},"Run one ",[203,2088,2090],{"href":2089},"what-is-an-ai-workstream","workstream"," with extract, classify, retrieve, and a memo.",[265,2093,2094],{},"Show model id per step. Classify and extract are compact. The memo may be frontier.",[265,2096,2097],{},"Show NTU (or tokens) per task class, quoted before the run grows, with a cap on the workstream.",[265,2099,2100],{},"Force a compact failure on a novel schema. Show a logged promotion, then a revert — not a user switching a dropdown.",[265,2102,2103],{},"Show a forecast path that is not an LLM interpolating a sheet. Narration can still be frontier.",[265,2105,2106],{},"Query the graph: which model drafted this payload? Answer from the ledger, not from a provider console.",[265,2108,2109],{},"Confirm a named-role gate still applies regardless of which model drafted. Routing chooses the brain. Governance still decides the write.",[190,2111,2112],{},"If they pass by opening a playground and picking “best,” you evaluated a dropdown.",[257,2114,1414],{"id":1413},[190,2116,2117],{},"Nimbus treats routing as an operating decision tied to workstream steps: task type, sensitivity, and cost — not “best everywhere.” Compact models handle extract. Frontier models are reserved for synthesis. Spend is NTU-metered, quoted per workstream, visible per step.",[190,2119,2120,2121,2124],{},"Release gates apply regardless of which model drafted the payload. Connectors stay read-only by default. Routing decides ",[224,2122,2123],{},"which brain"," reads them. Governance still decides whether anything writes.",[190,2126,615,2127,2129,2130,2132],{},[203,2128,44],{"href":45},". For the unit of account, ",[203,2131,1927],{"href":1926},". Score the four artefacts above. The product claim is the policy, not the catalogue.",[257,2134,749],{"id":748},[751,2136,2138],{"id":2137},"is-model-agnostic-the-same-as-routing","Is “model-agnostic” the same as routing?",[190,2140,2141,2142,2144],{},"No. Model-agnostic means more than one provider. Routing means it ",[194,2143,1937],{}," by task class, with a default that is cheap where cheap is correct. A hidden always-flagship default is lock-in with extra logos. Ask what happens if the operator never touches a dropdown. If the answer is flagship, you have your policy.",[751,2146,2148],{"id":2147},"should-operators-ever-pick-a-model","Should operators ever pick a model?",[190,2150,2151],{},"Rarely, and as an override. Production operators should brief outcomes. If quality depends on each user knowing which model is good at JSON, you have staffed a routing department by accident. Overrides should be logged, budget-aware, and exceptional. A dropdown on every run is how always-flagship returns through the side door.",[751,2153,2155],{"id":2154},"why-not-put-forecasting-in-the-llm-if-the-numbers-look-fine-in-the-demo","Why not put forecasting in the LLM if the numbers look fine in the demo?",[190,2157,2158],{},"Demos are short series. Production is seasonality, holidays, and missing days. Keep narration on the frontier model and estimation on a time-series path. A fluent number is not a control. The warehouse or the statistical path already owns the number. RAG plus a frontier model is for policy language, not for revenue by region.",[751,2160,2162],{"id":2161},"what-if-legal-requires-a-single-approved-model-vendor","What if legal requires a single approved model vendor?",[190,2164,2165,2166,2169],{},"Routing still applies ",[194,2167,2168],{},"inside"," that vendor’s catalogue: compact vs frontier vs embedding. Single-vendor is a contracting constraint, not an excuse to max tokens. Model-agnostic is nice. Task-class mapping inside one catalogue is the control. Do not skip routing because the RFP named one lab.",[751,2171,2173],{"id":2172},"does-nis2-or-dora-change-the-routing-question","Does NIS2 or DORA change the routing question?",[190,2175,2176,2177,2180],{},"They change the ",[194,2178,2179],{},"evidence"," question. DORA and NIS2 expect you to understand ICT dependencies. “We are not sure which model ran last Tuesday” is a dependency you cannot explain. Record model id per step on the graph. That is enough to start. You do not need a new product category. You need Measure.",[257,2182,806],{"id":805},[190,2184,2185,285,2189,2192,2193,302],{},[203,2186,2188],{"href":2187},"how-to-evaluate-ai-workstream-platforms","How to evaluate AI workstream platforms",[203,2190,2191],{"href":972},"How to evaluate an enterprise AI operating system",", and ",[203,2194,2047],{"href":2046},[257,2196,820],{"id":819},[262,2198,2199,2205,2211,2217,2222,2228,2234],{},[265,2200,2201],{},[203,2202,2204],{"href":1883,"rel":2203},[207],"OpenAI API pricing",[265,2206,2207],{},[203,2208,2210],{"href":1888,"rel":2209},[207],"Anthropic pricing",[265,2212,2213],{},[203,2214,2216],{"href":418,"rel":2215},[207],"Stanford HAI, 2025 AI Index Report",[265,2218,2219],{},[203,2220,886],{"href":252,"rel":2221},[207],[265,2223,2224],{},[203,2225,2227],{"href":1994,"rel":2226},[207],"Gartner, AI TRiSM / AI governance",[265,2229,2230],{},[203,2231,2233],{"href":1718,"rel":2232},[207],"DORA (Regulation 2022/2554)",[265,2235,2236],{},[203,2237,2239],{"href":2007,"rel":2238},[207],"NIS2 (Directive 2022/2555)",{"title":170,"searchDepth":171,"depth":171,"links":2241},[2242,2243,2244,2247,2248,2255,2256],{"id":259,"depth":171,"text":260},{"id":1958,"depth":171,"text":1959},{"id":2022,"depth":171,"text":2023,"children":2245},[2246],{"id":2077,"depth":929,"text":2078},{"id":1413,"depth":171,"text":1414},{"id":748,"depth":171,"text":749,"children":2249},[2250,2251,2252,2253,2254],{"id":2137,"depth":929,"text":2138},{"id":2147,"depth":929,"text":2148},{"id":2154,"depth":929,"text":2155},{"id":2161,"depth":929,"text":2162},{"id":2172,"depth":929,"text":2173},{"id":805,"depth":171,"text":806},{"id":819,"depth":171,"text":820},"2026-08-17","Model routing is a policy that uses a cheaper model for simple steps and a stronger model only when the task needs it — not a dropdown labelled “best.”","/blog/what-to-look-for-in-model-routing",{"title":1865,"description":2258},"blog/what-to-look-for-in-model-routing",[940,621,2263,2264],"routing","cost","CHj_o8nPz97yUKukPMWLIFwyFs4cGj__ST6_gCIWkhg",{"fold":2267,"id":2272,"title":2273,"archived":164,"authors":165,"badge":165,"body":2274,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":2278,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":2282,"relatedHeading":165,"seo":2283,"series":165,"sitemap":164,"status":165,"stem":2284,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":2285},{"headline":2268,"description":2269,"primaryLabel":8,"primaryTo":2270,"secondaryLabel":2271,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Free 7-day trial.","/checkout","Explore the platform","content/shared/cta.md","Site CTAs",{"type":167,"value":2275,"toc":2276},[],{"title":170,"searchDepth":171,"depth":171,"links":2277},[],{"headline":2279,"description":2280,"primaryLabel":8,"primaryTo":2270,"secondaryLabel":2281,"secondaryTo":99},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger - free for 7 days.","Talk to our team","/shared/cta",{"title":2273,"description":170},"shared/cta","wz4AdRHnaYH021WMdWcHnZvHmkZJNKaNG4XGZfnFBtw",{"enabled":164,"message":2287,"linkLabel":93,"linkHref":94,"id":2288,"title":2289,"archived":164,"authors":165,"badge":165,"body":2290,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":2294,"relatedHeading":165,"seo":2295,"series":165,"sitemap":164,"status":165,"stem":2296,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":2297},"We're hiring! Join the team building the Sentient Enterprise.","content/shared/hiring.md","Hiring banner",{"type":167,"value":2291,"toc":2292},[],{"title":170,"searchDepth":171,"depth":171,"links":2293},[],"/shared/hiring",{"title":2289,"description":170},"shared/hiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",1788982427939]