[{"data":1,"prerenderedAt":1737},["ShallowReactive",2],{"site-nav-content":3,"blog:/blog/eval-loops-for-enterprise-agent-harnesses":178,"blog-index-copy":919,"blog:/blog/eval-loops-for-enterprise-agent-harnesses:surround":940,"hiring-banner-content":1706,"site-cta-content":1718},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":131,"id":162,"title":163,"archived":164,"authors":165,"badge":165,"body":166,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":174,"relatedHeading":165,"seo":175,"series":165,"sitemap":164,"status":165,"stem":176,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":177},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","/overview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","/product/conflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","/product/agent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","/product/lifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","/product/wiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","/product/workstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","/product/perception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","/product/governance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","/models",{"label":47,"to":48},"Pricing","/pricing",{"label":50,"to":51},"Integrations","/integrations",{"label":53,"to":54},"Security","/security",{"label":56,"to":57},"Partners","/partners",{"label":59,"to":60},"Insights","/blog",{"productHeading":5,"companyHeading":62,"legalHeading":63,"docsLabel":64,"docsUrl":65,"statementLines":66,"copyright":69,"companyLinks":70,"legalLinks":100,"socialLinks":110,"bottomLinks":120},"Company","Legal","Docs","https://docs.gonimbus.ai",[67,68],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[71,72,73,74,75,78,81,84,87,90,92,95,98],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":76,"to":77},"Glossary","/glossary",{"label":79,"to":80},"Compare","/compare",{"label":82,"to":83},"Evaluate","/evaluate",{"label":85,"to":86},"Problems","/problems",{"label":88,"to":89},"Use cases","/use-cases",{"label":91,"to":57},"Partner Program",{"label":93,"to":94},"Careers","/careers",{"label":96,"to":97},"System status","/status",{"label":7,"to":99},"/contact",[101,104,107],{"label":102,"to":103},"Terms of Service","/terms",{"label":105,"to":106},"Privacy Policy","/privacy",{"label":108,"to":109},"Compliance","/compliance",[111,114,117],{"label":112,"href":113},"LinkedIn","https://www.linkedin.com/company/gonimbusai/",{"label":115,"href":116},"X","https://x.com/gonimbusai",{"label":118,"href":119},"Instagram","https://www.instagram.com/gonimbus_ai/",[121,123,125,126,127],{"label":122,"to":103},"Terms",{"label":124,"to":106},"Privacy",{"label":108,"to":109},{"label":96,"to":97},{"label":128,"to":129,"external":130},"LLMs.txt","/llms.txt",true,{"text":132,"prompt":133},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":134,"platforms":136},{" Summarize the highlights from Nimbus's website":135},"https://gonimbus.ai",[137,142,147,152,157],{"name":138,"label":139,"icon":140,"hrefPrefix":141},"chatgpt","ChatGPT","simple-icons:openai","https://chatgpt.com/?prompt=",{"name":143,"label":144,"icon":145,"hrefPrefix":146},"perplexity","Perplexity","mdi:magnify","https://www.perplexity.ai/search/new?q=",{"name":148,"label":149,"icon":150,"hrefPrefix":151},"grok","Grok","simple-icons:x","https://x.com/i/grok?text=",{"name":153,"label":154,"icon":155,"hrefPrefix":156},"claude","Claude","simple-icons:anthropic","https://claude.ai/new?q=",{"name":158,"label":159,"icon":160,"hrefPrefix":161},"google-ai","Google AI","simple-icons:google","https://www.google.com/search?udm=50&aep=11&q=","content/shared/nav.md","Site navigation",false,null,{"type":167,"value":168,"toc":169},"minimark",[],{"title":170,"searchDepth":171,"depth":171,"links":172},"",2,[],"md","/shared/nav",{"title":163,"description":170},"shared/nav","rDEv5cVG6P2l9ATcdQv2n6VOQiSL5ioOutyfvGevcD0",{"id":179,"title":180,"archived":164,"authors":181,"badge":184,"body":186,"date":909,"definedTerm":165,"department":165,"description":910,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":911,"relatedHeading":165,"seo":912,"series":913,"sitemap":130,"status":165,"stem":914,"subhead":165,"tags":915,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":918},"content/blog/eval-loops-for-enterprise-agent-harnesses.md","Eval Loops for Enterprise Agent Harnesses",[182],{"name":183,"to":135},"Nimbus Research",{"label":185},"Architecture",{"type":167,"value":187,"toc":890},[188,197,229,242,247,315,325,329,341,344,382,406,418,422,433,444,454,459,490,500,506,516,525,529,532,568,576,588,594,598,601,607,616,633,639,645,656,669,674,677,692,702,706,712,716,721,724,728,731,735,741,745,748,752,759,763,776,780,788,792],[189,190,191,192,196],"p",{},"An ",[193,194,195],"strong",{},"eval loop"," for an agent harness is an independent check that the job is actually done — tests, schemas, read-backs, humans — that does not take the model’s word.",[189,198,199,200,207,208,213,214,218,219,222,223,228],{},"Coding harnesses already have a public language for this. ",[201,202,206],"a",{"href":203,"rel":204},"https://www.swebench.com/",[205],"nofollow","SWE-bench"," gives an agent a GitHub issue and grades a patch with the repo’s tests. ",[201,209,212],{"href":210,"rel":211},"https://arxiv.org/abs/2601.11868",[205],"Terminal-Bench"," (Stanford / Laude Institute) gives an agent a machine and grades the ",[215,216,217],"em",{},"end state"," of a container, not the transcript. Leaderboards even report ",[193,220,221],{},"agent + model"," as a pair, which is the right unit: ",[201,224,227],{"href":225,"rel":226},"https://docs.langchain.com/oss/python/langchain/agents",[205],"Agent = Model + Harness",". Steal that honesty. Do not steal the benchmark as your control for Salesforce.",[189,230,231,232,236,237,241],{},"Enterprise eval is: did the quoted CRM write match the signed payload, and can you replay who signed. A 40% Terminal-Bench score does not tell you whether Opportunity.Amount was authorised. ",[201,233,235],{"href":234},"how-to-evaluate-an-agent-harness","How to evaluate an agent harness"," is the buying sheet. This page is the architecture of the sensor loop ",[201,238,240],{"href":239},"what-is-harness-engineering","harness engineering"," keeps tightening.",[243,244,246],"h2",{"id":245},"words-youll-hear","Words you’ll hear",[248,249,250,262,274,292,303,309],"ul",{},[251,252,253,256,257,261],"li",{},[193,254,255],{},"Oracle / verifier."," The independent test. SWE-bench: ",[258,259,260],"code",{},"FAIL_TO_PASS"," tests. Terminal-Bench: pytest-style assertions on container state. Enterprise: SoR read-back and payload hash.",[251,263,264,267,268,273],{},[193,265,266],{},"Transcript eval."," Grading the chain-of-thought. Useful for debugging. Insufficient as a release gate. Models claim victory; Anthropic’s ",[201,269,272],{"href":270,"rel":271},"https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents",[205],"long-running harness"," names premature victory as a failure mode.",[251,275,276,279,280,285,286,291],{},[193,277,278],{},"Computational vs inferential sensors."," ",[201,281,284],{"href":282,"rel":283},"https://martinfowler.com/articles/harness-engineering.html",[205],"Böckeler"," / ",[201,287,290],{"href":288,"rel":289},"https://www.thoughtworks.com/en-us/insights/blog/generative-ai/harness-engineering-agent-feedback-exploring-ai-coding-sensors",[205],"Thoughtworks",". Compiler vs LLM-as-judge. Prefer computational for invariants (schema, identity, hash). Use inferential for taste (narrative quality), never as the only SoR gate.",[251,293,294,297,298,302],{},[193,295,296],{},"LLM-as-judge."," Another stochastic component. Fine as a critic specialist. Not the signer. ",[201,299,301],{"href":300},"human-in-the-loop-approval-architecture","HITL architecture",".",[251,304,305,308],{},[193,306,307],{},"Offline vs online eval."," Offline: golden jobs, replay. Online: shadow reads, canary writes, production sensors. You need both; most teams only have a demo recording.",[251,310,311,314],{},[193,312,313],{},"Harness eval vs model eval."," Changing Claude vs GPT on the same tools is model eval. Changing hooks, grants, or wiki and keeping the model is harness eval. Report them separately or you will buy a new model for a missing schema check.",[189,316,317,318,321,322,324],{},"Nimbus’s production sensor for writes is the quote-and-gate plus graph: ",[201,319,320],{"href":40},"governance"," and ",[201,323,23],{"href":24},". That is computational. Wiki playbooks are guides. Do not confuse a fluent Conflux draft with a passed eval.",[243,326,328],{"id":327},"why-coding-benchmarks-are-the-wrong-outer-score","Why coding benchmarks are the wrong outer score",[189,330,331,332,335,336,340],{},"They are the ",[215,333,334],{},"right"," inner score. ",[201,337,339],{"href":338},"inner-vs-outer-agent-harness","Inner vs outer",". Terminal-Bench’s design is even a lesson: grade the environment, not the story. The environment for RevOps is Salesforce, not a Docker VM with a hidden oracle.",[189,342,343],{},"Problems when you import SWE-bench into an enterprise RFP:",[248,345,346,352,358,364,376],{},[251,347,348,351],{},[193,349,350],{},"Wrong workspace."," Patch quality ≠ payload authorisation.",[251,353,354,357],{},[193,355,356],{},"Saturation and leakage."," Public coding benches get gamed; your CRM schema is not a public task.",[251,359,360,363],{},[193,361,362],{},"No identity."," Benchmarks do not have a Finance signer.",[251,365,366,369,370,375],{},[193,367,368],{},"No replay duty."," A leaderboard row is not ",[201,371,374],{"href":372,"rel":373},"https://www.iso.org/standard/42001",[205],"ISO 42001"," evidence.",[251,377,378,381],{},[193,379,380],{},"Wrong “done.”"," Tests pass on a fixture; production field still wrong.",[189,383,384,389,390,395,396,399,400,405],{},[201,385,388],{"href":386,"rel":387},"https://www.mckinsey.com/capabilities/quantumblack/our-insights/the-state-of-ai",[205],"McKinsey 2025"," is about scaling work, not about bash tasks. ",[201,391,394],{"href":392,"rel":393},"https://www.nist.gov/itl/ai-risk-management-framework",[205],"NIST AI RMF"," Measure is: did the control work in ",[215,397,398],{},"your"," context of use. ",[201,401,404],{"href":402,"rel":403},"https://eur-lex.europa.eu/eli/reg/2024/1689/oj",[205],"EU AI Act"," wants interrupt and record, not a percentile on Terminal-Bench 2.1.",[189,407,408,409,413,414,302],{},"Use coding benches to pick an inner harness for engineering. Use quote/replay to pick an ",[201,410,412],{"href":411},"what-is-an-enterprise-agent-harness","enterprise harness",". ",[201,415,417],{"href":416},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose",[243,419,421],{"id":420},"what-an-enterprise-eval-loop-actually-runs","What an enterprise eval loop actually runs",[189,423,424,425,428,429,432],{},"Design it like Terminal-Bench in spirit: ",[193,426,427],{},"end state of the systems that matter",", plus ",[193,430,431],{},"process constraints"," the company cannot waive.",[189,434,435,438,439,443],{},[193,436,437],{},"Precondition sensors (feed-forward that is checkable)."," Required connectors attached. Roster includes the signer role. Wiki revision pinned. Spend quote accepted. If any fail, the run does not start. That is a harness eval of configuration, not of eloquence. ",[201,440,442],{"href":441},"agent-team-architecture","Agent teams"," declaring required systems belong here.",[189,445,446,449,450,302],{},[193,447,448],{},"Step sensors."," Retrieval logged and in-scope (no confused-deputy dump). Tool errors do not silently retry a write. Routing used compact on extract if that is policy. ",[201,451,453],{"href":452},"connector-and-permissions-architecture","Connector architecture",[189,455,456],{},[193,457,458],{},"Release sensors (the outer oracle).",[460,461,462,465,477,480,483],"ol",{},[251,463,464],{},"Quote is structured: object, fields, values, cardinality, hash.",[251,466,467,468,471,472,476],{},"Named human with the right role signed ",[215,469,470],{},"that"," hash (",[201,473,475],{"href":474},"what-is-write-back-governance","write-back",").",[251,478,479],{},"Adapter executed only that payload.",[251,481,482],{},"SoR read-back equals quote (or a documented, signed delta).",[251,484,485,486,302],{},"Graph (or equivalent ledger) contains brief, team, policy version, signer, payload, result. Export works without the vendor. ",[201,487,489],{"href":488},"what-is-a-lifecycle-graph","Lifecycle graph",[189,491,492,495,496,302],{},[193,493,494],{},"Negative tests."," Reject path: SoR unchanged. Detached grant: write impossible. Wrong role: Hard/Critical cannot complete. These are the analogue of tests that must stay red. If your PoV never fails, you did not eval the harness. You evaluated a happy path. ",[201,497,499],{"href":498},"how-to-run-an-enterprise-ai-proof-of-value","Proof of value",[189,501,502,505],{},[193,503,504],{},"Inferential sensors (optional, never sole)."," A legal specialist flags language. A critic agent scores a narrative. Useful. If they can waive a Hard gate, you added a second stochastic writer.",[189,507,508,279,511,515],{},[193,509,510],{},"Human as sensor, not as folklore.",[201,512,514],{"href":513},"what-is-human-in-the-loop-ai","HITL"," is a step with identity. A Slack emoji is transport. A six-month zero-reject rate is a finding: either perfect or unread.",[189,517,518,519,524],{},"Air Canada and the ",[201,520,523],{"href":521,"rel":522},"https://www.reuters.com/legal/new-york-lawyers-sanctioned-using-fake-chatgpt-cases-legal-brief-2023-06-22/",[205],"ChatGPT brief sanctions"," are eval-loop absences: no independent check before a system of record (policy page, court docket) changed.",[243,526,528],{"id":527},"offline-suites-you-can-actually-keep","Offline suites you can actually keep",[189,530,531],{},"You will not publish a public “CRM-bench.” You can keep a private suite:",[248,533,534,540,546,552,562],{},[251,535,536,539],{},[193,537,538],{},"Golden jobs."," Anonymised or sandbox SoR. Expected quote. Expected refuse.",[251,541,542,545],{},[193,543,544],{},"Replay."," Last month’s signed write: same hash, same graph nodes.",[251,547,548,551],{},[193,549,550],{},"Policy diffs."," Change wiki cap; next run must quote the new cap or refuse.",[251,553,554,557,558,302],{},[193,555,556],{},"Model swap."," Same harness, new weights: tools still dispatch; sensors still fire. That isolates model eval. ",[201,559,561],{"href":560},"what-is-model-routing","Model routing",[251,563,564,567],{},[193,565,566],{},"Chaos."," Kill the interceptor; writes must not fail open.",[189,569,570,571,575],{},"Version the suite with the harness. ",[201,572,574],{"href":573},"what-is-an-agentic-workflow","What is an agentic workflow",": the definition that ran is an input. A golden job that still “passes” after you removed the Hard gate is a broken eval, not a better model.",[189,577,578,579,582,583,587],{},"LangSmith, Phoenix, and similar tracing tools help ",[215,580,581],{},"observe"," inner and framework loops. They are not the SoR oracle. ",[201,584,586],{"href":585},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",". Tracing without a hash match is a nicer transcript.",[189,589,590,591,593],{},"Nimbus should be scored on whether you can automate those golden jobs on a sandbox org: attach, refuse, sign, read-back, export. ",[201,592,31],{"href":32}," are the fixture runner. If we cannot show a red refuse, we fail this architecture too.",[243,595,597],{"id":596},"building-a-private-suite-without-a-public-crm-bench","Building a private suite without a public CRM-bench",[189,599,600],{},"You do not need 2,294 GitHub issues. You need a dozen jobs that hurt when they are wrong.",[189,602,603,606],{},[193,604,605],{},"Pick three families."," (1) A write that must refuse (wrong role, missing field, detached grant). (2) A write that must match a fixture after sign-off. (3) A read-only job that must not call a write tool at all. Encode each as a workstream template or a scripted PoV. Run weekly. When a wiki cap changes, family (2) must fail until the quote updates — that is harness eval, not flaky CI.",[189,608,609,612,613,302],{},[193,610,611],{},"Grade environment state."," Terminal-Bench does not score the agent’s diary. Copy that. After the run, query the sandbox SoR. Compare to the signed hash. If you only grade the canvas prose, you are back to transcript eval. ",[201,614,615],{"href":474},"Write-back",[189,617,618,621,622,625,626,629,630,302],{},[193,619,620],{},"Keep model and harness scores apart."," Swap GPT vs Claude on the same golden job: if sensors still fire and hashes still match, the harness held. If a new model skips a field and the schema sensor catches it, that is a ",[215,623,624],{},"pass"," for the harness and a ",[215,627,628],{},"note"," for the model. If the sensor does not catch it, you do not need a larger model. You need a sensor. ",[201,631,632],{"href":239},"Harness engineering",[189,634,635,638],{},[193,636,637],{},"Report agent + model."," SWE-bench leaderboards already do this. Your internal dashboard should too: “Nimbus + routed compact/frontier” or “LangGraph + GPT + our interceptor.” Hiding the harness is how you buy a new model for a missing hook.",[189,640,641,644],{},[193,642,643],{},"Budget the eval itself."," Inferential judges on every step will cost more than the job. Thoughtworks’ advice: deterministic checks on every transaction; probabilistic judges on critical paths. Schema and identity are every-transaction. Narrative quality is not.",[189,646,647,650,651,655],{},[193,648,649],{},"What you can cite externally."," You can say you run refuse tests and read-backs. You cannot honestly say “we scored 83% on Terminal-Bench therefore Finance is safe.” ",[201,652,654],{"href":210,"rel":653},[205],"Stanford / Laude’s paper"," is a CLI benchmark. Use it for CLI harnesses.",[189,657,658,659,664,665,302],{},"Air Canada needed a sensor on “did we emit a policy commitment.” The court docket needed a sensor on “do these citations exist.” Your suite is that instinct with fixtures. ",[201,660,663],{"href":661,"rel":662},"https://www.cbc.ca/news/canada/british-columbia/air-canada-chatbot-lawsuit-1.7116416",[205],"CBC","; ",[201,666,668],{"href":521,"rel":667},[205],"Reuters",[189,670,671,672,302],{},"Online eval is the part teams skip. Offline goldens rot when the wiki moves. Shadow mode — agent quotes, human still writes, compare payloads — is an eval loop that does not need production write permission. Canary — one workstream, one object type, Hard gate, weekly refuse report — is how you learn whether operators rubber-stamp. A six-month zero-reject chart is not a quality medal. It is a sensor that may be dead. ",[201,673,514],{"href":513},[189,675,676],{},"Compare this to CI for software. You would not ship because the developer said the tests passed on their laptop. You would not replace CI with an LLM that reads the diff and scores “looks good.” You might add that LLM as a critic. Enterprise write eval is CI for mutations. Nimbus’s gate is the required check; your SoR read-back is the assertion file. If we only store the transcript, we are the laptop. Demand the assertion.",[189,678,679,684,685,688,689,302],{},[201,680,683],{"href":681,"rel":682},"https://docs.langchain.com/langsmith/observability",[205],"LangSmith"," and similar are the right place to ",[215,686,687],{},"debug"," traces for framework and inner loops. Export those traces into your golden runner; do not let the tracing UI become the only evidence for audit. Auditors will ask for the hash and the signer. ",[201,690,691],{"href":585},"How to evaluate AI audit",[189,693,694,695,697,698,701],{},"Do not wait for a consortium bench. Your suite is a competitive advantage if it encodes ",[215,696,398],{}," caps and objects. Share the ",[215,699,700],{},"method"," (refuse, read-back, replay) in the RFP. Keep the fixtures. Vendors who cannot run against your sandbox are not ready for your SoR, however they score on Terminal-Bench.",[243,703,705],{"id":704},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[189,707,708,709,711],{},"The product’s eval spine is: NTU quote before the run, scoped retrieval, canvas artefacts, write quotes, tiered gates, graph commit. Sensors you should still add: your own SoR read-back in the sandbox, your own golden files (the analogue of pytest). The platform cannot know your “correct Amount” without your oracle. Terminal-Bench ships oracles per task. You must ship oracles per job. That is ",[201,710,240],{"href":239},", not a missing model.",[243,713,715],{"id":714},"questions-people-actually-ask","Questions people actually ask",[717,718,720],"h3",{"id":719},"can-we-use-an-llm-as-judge-on-the-quote","Can we use an LLM-as-judge on the quote?",[189,722,723],{},"As a critic, yes. As the only signer, no. Computational match of fields is cheap and stable.",[717,725,727],{"id":726},"do-we-wait-for-an-industry-enterprise-swe-bench","Do we wait for an industry “enterprise SWE-bench”?",[189,729,730],{},"You would still need private oracles. Start this quarter with sandbox read-backs.",[717,732,734],{"id":733},"our-vendor-only-shares-swe-bench","Our vendor only shares SWE-bench.",[189,736,737,738,302],{},"File as inner evidence. Demand refuse/replay for outer. ",[201,739,740],{"href":234},"Evaluate the harness",[717,742,744],{"id":743},"is-tracing-enough-for-iso-42001","Is tracing enough for ISO 42001?",[189,746,747],{},"Traces help Measure. You still need Manage: a control that fired. A pretty trace of an unsigned write is a better incident report.",[717,749,751],{"id":750},"how-does-this-relate-to-agent-teams-vs-single-agents","How does this relate to agent teams vs single agents?",[189,753,754,755,302],{},"Teams add hand-off evals (typed artefacts). They do not replace the write oracle. ",[201,756,758],{"href":757},"how-to-evaluate-agent-teams-vs-single-agents","How to evaluate agent teams",[717,760,762],{"id":761},"what-should-i-read-next","What should I read next?",[189,764,765,413,769,413,773,302],{},[201,766,768],{"href":767},"agent-harness-architecture","Agent harness architecture",[201,770,772],{"href":771},"how-to-evaluate-write-back-governance","How to evaluate write-back governance",[201,774,775],{"href":239},"What is harness engineering",[243,777,779],{"id":778},"related-reading","Related reading",[189,781,782,321,784,302],{},[201,783,586],{"href":585},[201,785,787],{"href":786},"write-back-governance-for-systems-of-record","Write-back governance for systems of record",[243,789,791],{"id":790},"sources","Sources",[248,793,794,799,805,811,818,824,831,837,843,849,854,860,865,871,877,883],{},[251,795,796],{},[201,797,206],{"href":203,"rel":798},[205],[251,800,801],{},[201,802,804],{"href":210,"rel":803},[205],"Terminal-Bench (arXiv:2601.11868)",[251,806,807],{},[201,808,810],{"href":225,"rel":809},[205],"LangChain, Agents",[251,812,813],{},[201,814,817],{"href":815,"rel":816},"https://www.langchain.com/blog/the-anatomy-of-an-agent-harness",[205],"LangChain, The anatomy of an agent harness",[251,819,820],{},[201,821,823],{"href":270,"rel":822},[205],"Anthropic, Effective harnesses for long-running agents",[251,825,826],{},[201,827,830],{"href":828,"rel":829},"https://www.anthropic.com/engineering/building-effective-agents",[205],"Anthropic, Building effective agents",[251,832,833],{},[201,834,836],{"href":282,"rel":835},[205],"Böckeler, Harness engineering for coding agent users",[251,838,839],{},[201,840,842],{"href":288,"rel":841},[205],"Thoughtworks, Harness engineering and agent feedback",[251,844,845],{},[201,846,848],{"href":386,"rel":847},[205],"McKinsey, The state of AI in 2025",[251,850,851],{},[201,852,394],{"href":392,"rel":853},[205],[251,855,856],{},[201,857,859],{"href":372,"rel":858},[205],"ISO/IEC 42001",[251,861,862],{},[201,863,404],{"href":402,"rel":864},[205],[251,866,867],{},[201,868,870],{"href":661,"rel":869},[205],"CBC, Air Canada chatbot lawsuit",[251,872,873],{},[201,874,876],{"href":521,"rel":875},[205],"Reuters, ChatGPT legal brief sanctions",[251,878,879],{},[201,880,882],{"href":681,"rel":881},[205],"LangSmith observability",[251,884,885],{},[201,886,889],{"href":887,"rel":888},"https://modelcontextprotocol.io/specification/2025-11-25/index",[205],"Model Context Protocol specification",{"title":170,"searchDepth":171,"depth":171,"links":891},[892,893,894,895,896,897,898,907,908],{"id":245,"depth":171,"text":246},{"id":327,"depth":171,"text":328},{"id":420,"depth":171,"text":421},{"id":527,"depth":171,"text":528},{"id":596,"depth":171,"text":597},{"id":704,"depth":171,"text":705},{"id":714,"depth":171,"text":715,"children":899},[900,902,903,904,905,906],{"id":719,"depth":901,"text":720},3,{"id":726,"depth":901,"text":727},{"id":733,"depth":901,"text":734},{"id":743,"depth":901,"text":744},{"id":750,"depth":901,"text":751},{"id":761,"depth":901,"text":762},{"id":778,"depth":171,"text":779},{"id":790,"depth":171,"text":791},"2026-08-24","Coding agents can be scored on SWE-bench and Terminal-Bench. An enterprise harness is scored on whether the executed write matched the signed payload — independent sensors, not the model’s own claim that it was done.","/blog/eval-loops-for-enterprise-agent-harnesses",{"title":180,"description":910},"architecture","blog/eval-loops-for-enterprise-agent-harnesses",[913,916,917,320],"agent-harness","evaluation","CM1xXUF5XL1F03Rrcyw9I74LA1IpOJ_4H95nVjEY3r4",{"hero":920,"id":922,"title":923,"archived":164,"authors":165,"badge":165,"body":924,"date":165,"definedTerm":165,"department":165,"description":928,"extension":173,"eyebrow":929,"faqHeader":165,"faqs":165,"footerBand":930,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":60,"relatedHeading":936,"seo":937,"series":165,"sitemap":130,"status":165,"stem":938,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":939},{"filename":921},"u2221455217_Flat_design_of_a_futuristic_minimalist_landscape__5d589295-cdea-4ea9-a262-be766881accf_1.png","content/blog/index.md","Exploring the future of intelligence.",{"type":167,"value":925,"toc":926},[],{"title":170,"searchDepth":171,"depth":171,"links":927},[],"Deep dives into pre-cognitive intelligence, sentient enterprises, and the evolving landscape of AI-driven business transformation.","Latest Research",{"headline":931,"description":932,"primaryLabel":933,"primaryTo":934,"secondaryLabel":935,"secondaryTo":12},"Stay at the frontier.","Subscribe for product updates and new insights.","Subscribe","/newsletter","Explore the platform","More research",{"title":923,"description":928},"blog/index","BFSWGYO9bcTlaulivKYWyg08_DJHsdGg3OC6g_CG1Hw",[165,941],{"id":942,"title":943,"archived":164,"authors":944,"badge":946,"body":947,"date":909,"definedTerm":165,"department":165,"description":1699,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":164,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1700,"relatedHeading":165,"seo":1701,"series":913,"sitemap":130,"status":165,"stem":1702,"subhead":165,"tags":1703,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1705},"content/blog/agent-harness-architecture.md","Agent Harness Architecture",[945],{"name":183,"to":135},{"label":185},{"type":167,"value":948,"toc":1682},[949,954,980,1001,1003,1057,1084,1088,1108,1114,1117,1121,1124,1140,1150,1170,1176,1198,1207,1217,1228,1238,1249,1255,1261,1264,1268,1316,1322,1326,1333,1440,1443,1455,1461,1469,1476,1484,1500,1506,1508,1518,1520,1524,1527,1531,1538,1542,1550,1554,1561,1563,1572,1574,1583,1585],[189,950,951,953],{},[193,952,768],{}," is the design of the runtime around a model: who owns the loop, how tools run, what context is injected, which hooks can refuse, which identity the tools use, and how “done” is checked without taking the model’s word.",[189,955,956,960,961,966,967,971,972,971,975,971,978,302],{},[201,957,959],{"href":815,"rel":958},[205],"LangChain’s anatomy"," is the public parts list: prompts, tools and MCP, bundled infrastructure (filesystem, sandbox, browser), orchestration (subagents, routing), hooks and middleware (compaction, lint, continuation). ",[201,962,965],{"href":963,"rel":964},"https://www.databricks.com/blog/ai-harness",[205],"Databricks"," groups the same into tools, memory, workspace, guardrails. This article is that list as an architecture you can inspect — then the mapping onto company jobs: ",[201,968,970],{"href":969},"what-is-an-ai-workstream","workstreams",", ",[201,973,974],{"href":441},"agent teams",[201,976,977],{"href":452},"connectors",[201,979,475],{"href":474},[189,981,982,983,987,988,991,992,996,997,1000],{},"It is not a novel about kernels. It is not ",[201,984,986],{"href":985},"multi-agent-ai-architecture","multi-agent protocol"," (hand-offs between specialists) and not ",[201,989,990],{"href":300},"HITL state machines"," (quote → sign → execute), though a complete outer harness contains both. Start from ",[201,993,995],{"href":994},"what-is-an-agent-harness","what is an agent harness",". Use ",[201,998,999],{"href":234},"how to evaluate"," as the test of this diagram.",[243,1002,246],{"id":245},[248,1004,1005,1019,1027,1037,1047],{},[251,1006,1007,1010,1011,1014,1015,1018],{},[193,1008,1009],{},"Control plane vs data plane."," Control: grants, budgets, gates, routing policy — known independently of the model. Data: tokens, tool results, artefacts. If the orchestrator is only a system prompt, a jailbreak ",[215,1012,1013],{},"is"," a privilege escalation. ",[201,1016,1017],{"href":985},"Multi-agent architecture"," already said this; it is a harness invariant.",[251,1020,1021,1024,1025,302],{},[193,1022,1023],{},"Workspace."," Inner: checkout / sandbox. Outer: workstream. ",[201,1026,339],{"href":338},[251,1028,1029,1032,1033,302],{},[193,1030,1031],{},"Tool plane vs write plane."," Reads default on. Mutations fail-closed. MCP may implement both; architecture must split them. ",[201,1034,1036],{"href":887,"rel":1035},[205],"MCP spec",[251,1038,1039,1042,1043,1046],{},[193,1040,1041],{},"Compaction."," Harness-owned context management so the window does not become the only memory. Anthropic’s ",[201,1044,272],{"href":270,"rel":1045},[205]," offloads state to files and git.",[251,1048,1049,1052,1053,302],{},[193,1050,1051],{},"Routing."," Model class per step, not a user-picked mascot. ",[201,1054,1056],{"href":1055},"model-routing-architecture","Model routing architecture",[189,1058,1059,1060,1063,1064,1067,1068,1070,1071,1073,1074,1076,1077,1079,1080,1083],{},"Nimbus maps this architecture onto product objects rather than asking operators to draw LangGraph: ",[201,1061,1062],{"href":28},"wiki"," (guides), ",[201,1065,1066],{"href":51},"integrations"," (tool plane), ",[201,1069,974],{"href":20}," (orchestration contract), ",[201,1072,970],{"href":32}," (workspace), ",[201,1075,320],{"href":40}," (write plane), ",[201,1078,23],{"href":24}," (eval and memory), ",[201,1081,1082],{"href":45},"models"," (routing). Other vendors map the same boxes differently. Score the boxes.",[243,1085,1087],{"id":1086},"why-architecture-not-a-bigger-prompt","Why architecture (not a bigger prompt)",[189,1089,1090,1091,1093,1094,1097,1098,1101,1102,1107],{},"A prompt cannot own tool execution, identity, or a stop that survives a tired model. ",[201,1092,632],{"href":239}," is the practice; this page is the structure the practice edits. ",[201,1095,394],{"href":392,"rel":1096},[205]," Govern/Map need a system you can point to. ",[201,1099,374],{"href":372,"rel":1100},[205]," needs operational controls. ",[201,1103,1106],{"href":1104,"rel":1105},"https://genai.owasp.org/llm-top-10/",[205],"OWASP LLM Top 10"," excessive agency is what happens when the tool plane has no architecture.",[189,1109,1110,1113],{},[201,1111,388],{"href":386,"rel":1112},[205]," treats agentic value as organisational. Architecture is how you stop “every team’s unofficial loop” from becoming the estate.",[189,1115,1116],{},"It affects you if you are combining MCP servers, a coding agent, a copilot, and a CRM writer without a single grant and quote rule. Two writers to one object is an architecture bug, not a training issue.",[243,1118,1120],{"id":1119},"the-pieces","The pieces",[189,1122,1123],{},"Keep these as inspectable contracts.",[189,1125,1126,1129,1130,1134,1135,1139],{},[193,1127,1128],{},"1. Loop runtime."," Plan → act → observe, with max steps and a cost budget the model cannot waive. Frameworks (",[201,1131,1133],{"href":225,"rel":1132},[205],"create_agent",", LangGraph, CrewAI) implement this in process. Product harnesses implement it as a hosted run. ",[201,1136,1138],{"href":828,"rel":1137},[205],"Anthropic’s effective agents"," is still the best short note on bounding the loop. “The model says it is done” is an input to the runtime, not the runtime.",[189,1141,1142,1145,1146,1149],{},[193,1143,1144],{},"2. Workspace and filesystem."," Inner harnesses treat the directory as externalised memory — Manus-style and Anthropic-style artefacts. Outer harnesses treat the workstream as the directory analogue: artefacts on a canvas, not a hidden ",[258,1147,1148],{},"/tmp"," on a laptop. Do not store approved discounts only in a coding agent’s memory file.",[189,1151,1152,1155,1156,1159,1160,1164,1165,1169],{},[193,1153,1154],{},"3. Context assembly."," System prompt, skills, ",[258,1157,1158],{},"AGENTS.md"," / wiki slices, retrieved records, prior graph nodes. Guides in Böckeler’s sense. Compaction and retrieval belong here. ",[201,1161,1163],{"href":1162},"what-is-enterprise-rag","Enterprise RAG"," is a pattern inside this box, not the architecture. ",[201,1166,1168],{"href":1167},"what-is-a-company-wiki-for-ai-agents","Company wiki"," is asserted policy; do not collapse it into a private vector bucket per agent.",[189,1171,1172,1175],{},[193,1173,1174],{},"4. Tool dispatch."," Host executes; model proposes. Sandbox for shell. Adapters for SaaS. Timeouts, retries, structured errors back into the loop. Generic HTTP with a production token is not this box. It is a confused deputy.",[189,1177,1178,1181,1182,279,1187,285,1190,1193,1194,1197],{},[193,1179,1180],{},"5. Hooks / middleware."," Deterministic intercepts: ",[201,1183,1186],{"href":1184,"rel":1185},"https://code.claude.com/docs/en/hooks",[205],"Claude Code",[258,1188,1189],{},"PreToolUse",[258,1191,1192],{},"PostToolUse","; LangChain middleware; outer interceptor that never exposes the write API unsigned. ",[201,1195,1196],{"href":474},"Write-back governance",". Advice in markdown does not live in this box.",[189,1199,1200,1203,1204,302],{},[193,1201,1202],{},"6. Permissions and identity."," Who the harness authenticates as, per tool, per object, per job. Roster and workstream membership on the outer side. Repo and sandbox roles on the inner side. Teams declare required connectors; the workspace still grants. ",[201,1205,1206],{"href":441},"Agent team architecture",[189,1208,1209,1212,1213,302],{},[193,1210,1211],{},"7. Orchestration."," Subagents, specialist hand-offs, stop on gate. Optional until duties already split. Orchestrator in the product, not a manager persona with every login. ",[201,1214,1216],{"href":1215},"what-is-multi-agent-ai","What is multi-agent AI",[189,1218,1219,1222,1223,1227],{},[193,1220,1221],{},"8. Sensors and eval."," Compiler, tests, schema, quote-hash, SoR read-back, human review. Independent of the generator. ",[201,1224,1226],{"href":1225},"eval-loops-for-enterprise-agent-harnesses","Eval loops",". SWE-bench / Terminal-Bench measure inner coding harnesses; they do not close this box for GL posts.",[189,1229,1230,1233,1234,302],{},[193,1231,1232],{},"9. Durable memory of operations."," Files and git (inner). Wiki + Lifecycle Graph (outer). Session transcripts are a debug aid. They are not the ledger. ",[201,1235,1237],{"href":1236},"causal-memory-architecture-for-enterprise-ai","Causal memory",[189,1239,1240,1243,1244,1248],{},[193,1241,1242],{},"10. Routing and spend."," Step classes → model classes. Caps on the run. ",[201,1245,1247],{"href":1246},"ai-cost-control-architecture","AI cost control",". Seat-unlimited flagship is an architectural choice (always-frontier), not a missing feature.",[189,1250,1251,1254],{},[193,1252,1253],{},"Flow (outer)."," Brief on a workstream → satisfy connector contract → plan → retrieve (logged, scoped) → draft on canvas → quote if write in scope → gate → execute signed payload only → commit graph. If steps 5–7 live only in a prompt, jailbreaks and tired operators fall through the same hole.",[189,1256,1257,1260],{},[193,1258,1259],{},"Flow (inner)."," Session start loads guides → loop with shell/editor tools → hooks on tool events → tests as sensor → commit / PR → CI as outer-loop sensor in Osmani’s sense. Anthropic’s initializer vs coding agent is a two-role inner architecture for work that outlasts one window.",[189,1262,1263],{},"Nimbus’s hosted flow is the outer sequence. Perception and Conflux sit on retrieve/draft; they must not skip the quote. That is architecture, not brand.",[243,1265,1267],{"id":1266},"failure-modes-the-diagram-exists-to-prevent","Failure modes the diagram exists to prevent",[460,1269,1270,1276,1282,1288,1294,1300,1306],{},[251,1271,1272,1275],{},[193,1273,1274],{},"Orchestrator-in-the-model."," Jailbreak equals admin.",[251,1277,1278,1281],{},[193,1279,1280],{},"Shared toolbox."," Every specialist has every write.",[251,1283,1284,1287],{},[193,1285,1286],{},"Context as only memory."," Compaction deletes the approval.",[251,1289,1290,1293],{},[193,1291,1292],{},"MCP as control plane."," Plug without grants.",[251,1295,1296,1299],{},[193,1297,1298],{},"Eval = transcript."," The model graded itself.",[251,1301,1302,1305],{},[193,1303,1304],{},"Two harnesses, one SoR writer."," IDE MCP and OS both PATCH.",[251,1307,1308,1311,1312,302],{},[193,1309,1310],{},"Framework mistaken for architecture."," Nodes without identity. ",[201,1313,1315],{"href":1314},"agent-harness-vs-agent-framework","Harness vs framework",[189,1317,1318,1321],{},[201,1319,404],{"href":402,"rel":1320},[205]," oversight needs interrupt and record. Those are boxes 5, 6, and 9.",[243,1323,1325],{"id":1324},"mapping-langchains-anatomy-onto-company-objects","Mapping LangChain’s anatomy onto company objects",[189,1327,1328,1332],{},[201,1329,1331],{"href":815,"rel":1330},[205],"LangChain’s parts list"," is built from coding and general agents. Translate, do not copy:",[1334,1335,1336,1352],"table",{},[1337,1338,1339],"thead",{},[1340,1341,1342,1346,1349],"tr",{},[1343,1344,1345],"th",{},"Anatomy piece",[1343,1347,1348],{},"Inner binding",[1343,1350,1351],{},"Outer binding",[1353,1354,1355,1370,1381,1392,1405,1416,1429],"tbody",{},[1340,1356,1357,1361,1367],{},[1358,1359,1360],"td",{},"System prompts / skills",[1358,1362,1363,1366],{},[258,1364,1365],{},"CLAUDE.md",", skills",[1358,1368,1369],{},"Wiki playbooks, versioned with the run",[1340,1371,1372,1375,1378],{},[1358,1373,1374],{},"Tools + MCP",[1358,1376,1377],{},"Shell, apply_patch, browser",[1358,1379,1380],{},"Connectors; MCP behind the same grant",[1340,1382,1383,1386,1389],{},[1358,1384,1385],{},"Filesystem / sandbox",[1358,1387,1388],{},"Checkout, container",[1358,1390,1391],{},"Workstream canvas + isolated grants",[1340,1393,1394,1397,1400],{},[1358,1395,1396],{},"Orchestration",[1358,1398,1399],{},"Subagents in the IDE",[1358,1401,1402,1404],{},[201,1403,442],{"href":441}," on a roster",[1340,1406,1407,1410,1413],{},[1358,1408,1409],{},"Hooks / middleware",[1358,1411,1412],{},"PreToolUse, lint",[1358,1414,1415],{},"Write interceptor, spend cap",[1340,1417,1418,1421,1424],{},[1358,1419,1420],{},"Memory",[1358,1422,1423],{},"Files, git, memory md",[1358,1425,1426,1427],{},"Wiki + ",[201,1428,23],{"href":488},[1340,1430,1431,1434,1437],{},[1358,1432,1433],{},"Eval",[1358,1435,1436],{},"Tests, Terminal-Bench",[1358,1438,1439],{},"Quote hash, SoR read-back",[189,1441,1442],{},"If a vendor cannot fill the outer column, they are an inner (or framework) product. That is allowed. Do not invent the column in a slide.",[189,1444,1445,1448,1449,1452,1453,302],{},[193,1446,1447],{},"Control plane independence."," Whatever sits in the Orchestration row must know grants, budget, and gates ",[215,1450,1451],{},"without"," asking the model. LangGraph can do that if the nodes are code. A “manager agent” with every tool cannot. Nimbus’s orchestrator is product-hosted for that reason; you should still ask it to refuse when NetSuite is missing. ",[201,1454,82],{"href":234},[189,1456,1457,1460],{},[193,1458,1459],{},"Thoughtworks’ four combinations"," (deterministic/probabilistic × feed-forward/feedback) overlay this table. Whitelists and spend ceilings are box 5/6 deterministic feed-forward. Schema validation is box 8 deterministic feedback. Wiki retrieval is probabilistic feed-forward. LLM critic is probabilistic feedback — never the only item in box 8 for a GL post.",[189,1462,1463,1466,1467,302],{},[193,1464,1465],{},"Two harnesses, one SoR rule."," Draw both columns on one whiteboard. Draw one write plane. If two arrows reach Salesforce, you have an architecture incident waiting. ",[201,1468,339],{"href":338},[189,1470,1471,1472,302],{},"Version the diagram when you add a tool. A new MCP server is a change to boxes 4 and 6, not a chat plugin. ",[201,1473,1475],{"href":1474},"what-is-model-context-protocol","MCP",[189,1477,1478,1479,1483],{},"Implementation order for a company that has none of this: (1) split write plane from read plane — even if the “harness” is still a single agent; (2) pin policy version on the run; (3) add one deterministic sensor on the artefact you cannot get wrong; (4) host the orchestrator’s grants outside the prompt; (5) only then add specialists. Reversing that order is how shared-toolbox swarms ship. ",[201,1480,1482],{"href":828,"rel":1481},[205],"Anthropic"," starts with bounding tools and defining done for a reason.",[189,1485,1486,1487,971,1489,971,1491,971,1493,1496,1497,1499],{},"Framework teams should draw the ten boxes on the README of the graph repo and tick which are code, which are still prompts, which are missing. Product teams should map each box to a screen an operator can see. If box 8 is “the model reflects,” you do not have eval architecture. If box 6 is “the service account,” you do not have identity architecture. Nimbus’s screens are ",[201,1488,970],{"href":32},[201,1490,320],{"href":40},[201,1492,1062],{"href":28},[201,1494,1495],{"href":24},"graph"," — use them as a checklist, not as proof that the boxes exist in ",[215,1498,398],{}," configuration.",[189,1501,1502,1505],{},[201,1503,965],{"href":963,"rel":1504},[205]," calls the model the brain and the harness the body. Architecture is the anatomy of that body so Security can review it. If the diagram is only “LLM in the middle, tools around it,” you have a marketing poster. Add identity, the write split, the sensor that does not trust the brain, and the ledger. Then the poster is a design.",[243,1507,705],{"id":704},[189,1509,1510,1511,1513,1514,1517],{},"The product is a particular binding of the ten boxes for operators: hosted loop, workstream workspace, wiki context, connector dispatch, governance hooks, team orchestration, graph memory, NTU routing. ",[201,1512,11],{"href":12},". Inspect each box in a PoV the way you would inspect Claude Code’s hooks and sandbox for an inner buy. ",[201,1515,1516],{"href":234},"How to evaluate",". AIP and Agentforce bind the same boxes to Ontology or CRM; the architecture still applies.",[243,1519,715],{"id":714},[717,1521,1523],{"id":1522},"do-we-need-all-ten-boxes-on-day-one","Do we need all ten boxes on day one?",[189,1525,1526],{},"You need loop, tools, a stop, and a sensor for the job you are running. Add orchestration when duties split. Add graph when people leave. Do not add every MCP server first.",[717,1528,1530],{"id":1529},"is-this-the-same-as-an-enterprise-ai-os-architecture","Is this the same as an enterprise AI OS architecture?",[189,1532,1533,1537],{},[201,1534,1536],{"href":1535},"enterprise-ai-operating-system-architecture","OS architecture"," is the product category (collaboration, gates, ledger, routing). Harness architecture is the runtime idea that also covers Claude Code. Overlap on the outer side is expected.",[717,1539,1541],{"id":1540},"where-do-skills-fit","Where do skills fit?",[189,1543,1544,1545,302],{},"Reusable procedures in the context box. Not a substitute for hooks. ",[201,1546,1549],{"href":1547,"rel":1548},"https://claude.com/blog/steering-claude-code-skills-hooks-rules-subagents-and-more",[205],"Anthropic on steering",[717,1551,1553],{"id":1552},"can-langgraph-implement-this","Can LangGraph implement this?",[189,1555,1556,1557,302],{},"Yes. You will implement boxes 5, 6, and 9 yourself for enterprise writes. That is ",[201,1558,1560],{"href":1559},"build-vs-buy-an-enterprise-ai-os","build vs buy",[717,1562,762],{"id":761},[189,1564,1565,413,1567,413,1569,302],{},[201,1566,1226],{"href":1225},[201,1568,775],{"href":239},[201,1570,1571],{"href":411},"What is an enterprise agent harness",[243,1573,779],{"id":778},[189,1575,1576,321,1580,302],{},[201,1577,1579],{"href":1578},"workstream-architecture","Workstream architecture",[201,1581,1582],{"href":452},"Connector and permissions architecture",[243,1584,791],{"id":790},[248,1586,1587,1592,1597,1604,1610,1617,1622,1627,1633,1639,1644,1651,1656,1661,1666,1672,1677],{},[251,1588,1589],{},[201,1590,817],{"href":815,"rel":1591},[205],[251,1593,1594],{},[201,1595,810],{"href":225,"rel":1596},[205],[251,1598,1599],{},[201,1600,1603],{"href":1601,"rel":1602},"https://www.langchain.com/blog/how-to-build-a-custom-agent-harness",[205],"LangChain, How to build a custom agent harness",[251,1605,1606],{},[201,1607,1609],{"href":963,"rel":1608},[205],"Databricks, What is an AI agent harness?",[251,1611,1612],{},[201,1613,1616],{"href":1614,"rel":1615},"https://en.wikipedia.org/wiki/Agent_harness",[205],"Wikipedia, Agent harness",[251,1618,1619],{},[201,1620,830],{"href":828,"rel":1621},[205],[251,1623,1624],{},[201,1625,823],{"href":270,"rel":1626},[205],[251,1628,1629],{},[201,1630,1632],{"href":1547,"rel":1631},[205],"Anthropic, Steering Claude Code",[251,1634,1635],{},[201,1636,1638],{"href":1184,"rel":1637},[205],"Claude Code, Hooks",[251,1640,1641],{},[201,1642,836],{"href":282,"rel":1643},[205],[251,1645,1646],{},[201,1647,1650],{"href":1648,"rel":1649},"https://addyosmani.com/blog/agent-harness-engineering/",[205],"Addy Osmani, Agent harness engineering",[251,1652,1653],{},[201,1654,394],{"href":392,"rel":1655},[205],[251,1657,1658],{},[201,1659,859],{"href":372,"rel":1660},[205],[251,1662,1663],{},[201,1664,404],{"href":402,"rel":1665},[205],[251,1667,1668],{},[201,1669,1671],{"href":1104,"rel":1670},[205],"OWASP Top 10 for LLM applications",[251,1673,1674],{},[201,1675,848],{"href":386,"rel":1676},[205],[251,1678,1679],{},[201,1680,889],{"href":887,"rel":1681},[205],{"title":170,"searchDepth":171,"depth":171,"links":1683},[1684,1685,1686,1687,1688,1689,1690,1697,1698],{"id":245,"depth":171,"text":246},{"id":1086,"depth":171,"text":1087},{"id":1119,"depth":171,"text":1120},{"id":1266,"depth":171,"text":1267},{"id":1324,"depth":171,"text":1325},{"id":704,"depth":171,"text":705},{"id":714,"depth":171,"text":715,"children":1691},[1692,1693,1694,1695,1696],{"id":1522,"depth":901,"text":1523},{"id":1529,"depth":901,"text":1530},{"id":1540,"depth":901,"text":1541},{"id":1552,"depth":901,"text":1553},{"id":761,"depth":901,"text":762},{"id":778,"depth":171,"text":779},{"id":790,"depth":171,"text":791},"Agent harness architecture is the runtime around a model: loop, tools, context, hooks, permissions, and eval — mapped, for company jobs, onto workstreams, agent teams, connectors, and write gates.","/blog/agent-harness-architecture",{"title":943,"description":1699},"blog/agent-harness-architecture",[913,916,1704,320],"orchestration","NbF7f7oegzZJ7kbYT2wy7RoWBPoIrmmPE1R8WlUetFA",{"enabled":164,"message":1707,"linkLabel":93,"linkHref":94,"id":1708,"title":1709,"archived":164,"authors":165,"badge":165,"body":1710,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":165,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1714,"relatedHeading":165,"seo":1715,"series":165,"sitemap":164,"status":165,"stem":1716,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1717},"We're hiring! Join the team building the Sentient Enterprise.","content/shared/hiring.md","Hiring banner",{"type":167,"value":1711,"toc":1712},[],{"title":170,"searchDepth":171,"depth":171,"links":1713},[],"/shared/hiring",{"title":1709,"description":170},"shared/hiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",{"fold":1719,"id":1723,"title":1724,"archived":164,"authors":165,"badge":165,"body":1725,"date":165,"definedTerm":165,"department":165,"description":170,"extension":173,"eyebrow":165,"faqHeader":165,"faqs":165,"footerBand":1729,"headline":165,"image":165,"industry":165,"jobType":165,"listed":130,"location":165,"navigation":130,"openRoles":165,"pageLayout":165,"path":1733,"relatedHeading":165,"seo":1734,"series":165,"sitemap":164,"status":165,"stem":1735,"subhead":165,"tags":165,"video":165,"whyJoin":165,"workplaceType":165,"__hash__":1736},{"headline":1720,"description":1721,"primaryLabel":8,"primaryTo":1722,"secondaryLabel":935,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Free 7-day trial.","/checkout","content/shared/cta.md","Site CTAs",{"type":167,"value":1726,"toc":1727},[],{"title":170,"searchDepth":171,"depth":171,"links":1728},[],{"headline":1730,"description":1731,"primaryLabel":8,"primaryTo":1722,"secondaryLabel":1732,"secondaryTo":99},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger - free for 7 days.","Talk to our team","/shared/cta",{"title":1724,"description":170},"shared/cta","wz4AdRHnaYH021WMdWcHnZvHmkZJNKaNG4XGZfnFBtw",1788985847791]