[{"data":1,"prerenderedAt":1808},["ShallowReactive",2],{"site-nav-content":3,"blog:/blog/how-to-evaluate-collaborative-ai":179,"blog-index-copy":1050,"blog:/blog/how-to-evaluate-collaborative-ai:surround":1071,"hiring-banner-content":1777,"site-cta-content":1789},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":132,"id":163,"title":164,"archived":165,"authors":166,"badge":166,"body":167,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":175,"relatedHeading":166,"seo":176,"series":166,"sitemap":165,"status":166,"stem":177,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":178},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","/overview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","/product/conflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","/product/agent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","/product/lifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","/product/wiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","/product/workstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","/product/perception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","/product/governance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","/models",{"label":47,"to":48},"Pricing","/pricing",{"label":50,"to":51},"Integrations","/integrations",{"label":53,"to":54},"Security","/security",{"label":56,"to":57},"Partners","/partners",{"label":59,"to":60},"Insights","/blog",{"productHeading":5,"companyHeading":62,"resourcesHeading":63,"legalHeading":64,"docsLabel":65,"docsUrl":66,"statementLines":67,"copyright":70,"companyLinks":71,"resourcesLinks":86,"legalLinks":102,"socialLinks":109,"bottomLinks":119},"Company","Resources","Legal","Docs","https://docs.gonimbus.ai",[68,69],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[72,73,74,75,76,78,81,84],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":77,"to":57},"Partner Program",{"label":79,"to":80},"Careers","/careers",{"label":82,"to":83},"System status","/status",{"label":7,"to":85},"/contact",[87,90,93,96,99],{"label":88,"to":89},"Glossary","/glossary",{"label":91,"to":92},"Compare","/compare",{"label":94,"to":95},"Evaluate","/evaluate",{"label":97,"to":98},"Problems","/problems",{"label":100,"to":101},"Use cases","/use-cases",[103,106],{"label":104,"to":105},"Terms of Service","/terms",{"label":107,"to":108},"Privacy Policy","/privacy",[110,113,116],{"label":111,"href":112},"LinkedIn","https://www.linkedin.com/company/gonimbusai/",{"label":114,"href":115},"X","https://x.com/gonimbusai",{"label":117,"href":118},"Instagram","https://www.instagram.com/gonimbus_ai/",[120,122,124,127,128],{"label":121,"to":105},"Terms",{"label":123,"to":108},"Privacy",{"label":125,"to":126},"Compliance","/compliance",{"label":82,"to":83},{"label":129,"to":130,"external":131},"LLMs.txt","/llms.txt",true,{"text":133,"prompt":134},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":135,"platforms":137},{" Summarize the highlights from Nimbus's website":136},"https://gonimbus.ai",[138,143,148,153,158],{"name":139,"label":140,"icon":141,"hrefPrefix":142},"chatgpt","ChatGPT","simple-icons:openai","https://chatgpt.com/?prompt=",{"name":144,"label":145,"icon":146,"hrefPrefix":147},"perplexity","Perplexity","mdi:magnify","https://www.perplexity.ai/search/new?q=",{"name":149,"label":150,"icon":151,"hrefPrefix":152},"grok","Grok","simple-icons:x","https://x.com/i/grok?text=",{"name":154,"label":155,"icon":156,"hrefPrefix":157},"claude","Claude","simple-icons:anthropic","https://claude.ai/new?q=",{"name":159,"label":160,"icon":161,"hrefPrefix":162},"google-ai","Google AI","simple-icons:google","https://www.google.com/search?udm=50&aep=11&q=","content/shared/nav.md","Site navigation",false,null,{"type":168,"value":169,"toc":170},"minimark",[],{"title":171,"searchDepth":172,"depth":172,"links":173},"",2,[],"md","/shared/nav",{"title":164,"description":171},"shared/nav","1dD7ahDRl0SQ4hz53-kKo0tEFrGaLuaztZ3PPfp6a9k",{"id":180,"title":181,"archived":165,"authors":182,"badge":185,"body":187,"date":1023,"definedTerm":166,"department":166,"description":1024,"extension":174,"eyebrow":166,"faqHeader":1025,"faqs":1028,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1041,"relatedHeading":166,"seo":1042,"series":1043,"sitemap":131,"status":166,"stem":1044,"subhead":166,"tags":1045,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1049},"content/blog/how-to-evaluate-collaborative-ai.md","How to Evaluate Collaborative AI",[183],{"name":184,"to":136},"Nimbus Research",{"label":186},"Evaluation",{"type":168,"value":188,"toc":1005},[189,206,227,238,258,263,270,298,301,330,342,353,365,374,378,381,386,392,398,404,410,414,419,424,429,436,445,449,462,467,476,492,511,515,520,525,530,546,550,555,560,565,577,585,589,602,607,617,624,628,728,737,741,744,749,768,773,791,794,819,823,828,839,842,860,863,875,882,886,889,915,918,921,925,928,939,944,947,951,958,970,973,977,991,998,1001],[190,191,192,193,197,198,205],"p",{},"Evaluating collaborative AI means scoring whether several people can share ",[194,195,196],"strong",{},"one job"," — same files, same history, a named person who can stop a change — and whether that job still exists after the session ends. McKinsey’s ",[199,200,204],"a",{"href":201,"rel":202},"https://www.mckinsey.com/capabilities/quantumblack/our-insights/the-state-of-ai",[203],"nofollow","State of AI"," (2025) found that 88% of organisations use AI in at least one function while most remain in experiment or pilot. A common pilot is one person and one assistant. The next purchase mistake is rebranding that pilot “collaborative” because five people share a login. This checklist is written to catch that false pass early, on any vendor.",[190,207,208,209,213,214,218,219,222,223,226],{},"It is not the same as evaluating a copilot, an agent swarm, or a shared ChatGPT account. ",[199,210,212],{"href":211},"what-is-collaborative-ai","What is collaborative AI"," is the definition. ",[199,215,217],{"href":216},"collaborative-ai-and-personal-assistants","Collaborative AI and personal assistants"," is ",[194,220,221],{},"when"," to use a copilot versus a shared room — read that first if the purchase question is still “do we need both?” This page is the ",[194,224,225],{},"RFP sheet"," for the shared job: questions to put in procurement, demos, and proof-of-value scripts.",[190,228,229,233,234,237],{},[199,230,232],{"href":231},"how-to-evaluate-an-agent-harness","How to evaluate an agent harness"," is the runtime sheet — refused writes, replay, model swap. This page is the ",[194,235,236],{},"job"," sheet — roster, rejection, handover. Run both when the work crosses departments and may write to a live system. Do not run only the one the vendor prefers.",[190,239,240,245,246,249,250,253,254,257],{},[199,241,244],{"href":242,"rel":243},"https://www.nist.gov/itl/ai-risk-management-framework/nist-ai-rmf-playbook",[203],"NIST’s AI RMF Playbook"," is the measurement language. Translate it to demos: ",[194,247,248],{},"Map"," the job, ",[194,251,252],{},"Measure"," reopen time and signer completeness, ",[194,255,256],{},"Manage"," with stored rejections before write tokens. “The model seemed careful” is not a measure.",[259,260,262],"h2",{"id":261},"what-you-are-actually-buying","What you are actually buying",[190,264,265,266,269],{},"You are buying a ",[194,267,268],{},"container for shared work",", not a better paragraph generator.",[271,272,276],"pre",{"className":273,"code":274,"language":275,"meta":171,"style":171},"language-mermaid shiki shiki-themes github-light github-dark","flowchart LR\n  joinTest[\"Can a second team join?\"] --> stopTest[\"Can someone say no?\"]\n  stopTest --> mondayTest[\"Does Monday still have the files?\"]\n","mermaid",[277,278,279,287,292],"code",{"__ignoreMap":171},[280,281,284],"span",{"class":282,"line":283},"line",1,[280,285,286],{},"flowchart LR\n",[280,288,289],{"class":282,"line":172},[280,290,291],{},"  joinTest[\"Can a second team join?\"] --> stopTest[\"Can someone say no?\"]\n",[280,293,295],{"class":282,"line":294},3,[280,296,297],{},"  stopTest --> mondayTest[\"Does Monday still have the files?\"]\n",[190,299,300],{},"Minimum properties:",[302,303,304,312,318,324],"ul",{},[305,306,307,308,311],"li",{},"A ",[194,309,310],{},"named job",", not “the Slack channel.”",[305,313,314,317],{},[194,315,316],{},"Shared context"," — files and history visible to the roster.",[305,319,320,323],{},[194,321,322],{},"Tools with recorded steps"," — read, propose, sometimes write with a payload.",[305,325,307,326,329],{},[194,327,328],{},"named stop"," — someone other than the prompter can refuse a change.",[190,331,332,336,337,341],{},[199,333,335],{"href":334},"multiplayer-ai-and-multi-agent-ai","Multiplayer AI vs multi-agent AI"," separates people in the room from models in a loop. You can fail collaborative eval while passing multi-agent demos: agents pass tickets to each other; finance still cannot see the brief. ",[199,338,340],{"href":339},"what-is-multi-agent-ai","What is multi-agent AI"," is the cast. This sheet scores the stage.",[190,343,344,348,349,352],{},[199,345,347],{"href":346},"four-pillars-of-an-enterprise-ai-platform","Four pillars of an enterprise AI platform"," situates workstreams and governance in a full stack. This sheet scores whether the product you are viewing implements the ",[194,350,351],{},"collaborative"," pillar for one real job — Salesforce sidebar, Microsoft copilot, Slack bot, specialist agent platform, or anything else on the shortlist.",[190,354,355,360,361,364],{},[199,356,359],{"href":357,"rel":358},"https://www.anthropic.com/engineering/building-effective-agents",[203],"Anthropic’s guidance on building effective agents"," says encode the job, bound the tools, define done. Your proof of value should force those three ",[194,362,363],{},"in a room with two departments",", not in a solo sandbox.",[190,366,367,368,373],{},"Yang and colleagues, in ",[199,369,372],{"href":370,"rel":371},"https://www.nature.com/articles/s41562-021-01196-4",[203],"Nature Human Behaviour"," (2022), showed that firm-wide remote work made collaboration networks more static and siloed, with fewer bridges between groups. Shared jobs already fight that pull. A vendor that adds a fluent assistant to each silo will make the silo more confident, not more shared. Score the bridge.",[259,375,377],{"id":376},"rfp-questions-shared-job-vs-shared-login","RFP questions: shared job vs shared login",[190,379,380],{},"Put these verbatim in the RFP. Require live answers, not slides. If professional services will “configure that later,” note the time-to-value and price the configuration as part of the buy.",[382,383,385],"h3",{"id":384},"_1-is-this-a-shared-job-or-a-shared-login","1. Is this a shared job or a shared login?",[190,387,388,391],{},[194,389,390],{},"Pass:"," One work object with its own roster, files, and audit — independent of which user opened the UI today.",[190,393,394,397],{},[194,395,396],{},"Fail:"," Five people in one chatbot account, or five parallel threads that cannot see each other’s attachments.",[190,399,400,403],{},[194,401,402],{},"Proof:"," Show two users on the same job ID. Remove one user’s access. The job remains for the roster.",[190,405,406,407,409],{},"Why vendors fail this: shared seats are cheap to demo and expensive to govern. ",[199,408,217],{"href":216}," explains why a shared login is still a personal-assistant shape. Ask for the job ID in the URL or export. If the vendor cannot point at an object, they pointed at a session.",[382,411,413],{"id":412},"_2-can-a-second-department-join-mid-run","2. Can a second department join mid-run?",[190,415,416,418],{},[194,417,390],{}," Finance joins Thursday’s sales exception without a re-upload parade. They see the same CRM excerpt, the same draft payload, the same history.",[190,420,421,423],{},[194,422,396],{}," “Export and email the transcript.” “Start a new session and paste context.”",[190,425,426,428],{},[194,427,402],{}," Add a finance delegate mid-proof. They reject a proposal while sales watches. No side channel required.",[190,430,431,432,435],{},"This is the core ",[199,433,434],{"href":334},"multiplayer AI"," test dressed for procurement. Do not accept a pre-seeded “war room” that was built overnight by the vendor’s solutions team unless you can repeat the join on a job your people created.",[190,437,438,439,444],{},"Microsoft and LinkedIn’s ",[199,440,443],{"href":441,"rel":442},"https://www.microsoft.com/en-us/worklab/work-trend-index/ai-at-work-is-here-now-comes-the-hard-part",[203],"2024 Work Trend Index"," found that 78% of AI users bring their own tools. Mid-run join is how you find out whether the product can absorb that habit or whether finance will open a second private window.",[382,446,448],{"id":447},"_3-is-there-a-named-stop-not-human-review-in-the-abstract","3. Is there a named stop — not “human review” in the abstract?",[190,450,451,453,454,457,458,461],{},[194,452,390],{}," A person on ",[194,455,456],{},"this roster"," can halt ",[194,459,460],{},"this class of write"," while others on the job see the payload.",[190,463,464,466],{},[194,465,396],{}," A generic approval workflow outside the job, or a prompt that says “ask manager.”",[190,468,469,471,472,475],{},[194,470,402],{}," Attempt a live-system change. Show the signer field tied to a human identity. Show a ",[194,473,474],{},"stored rejection"," with name and timestamp on the job.",[190,477,478,482,483,487,488,491],{},[199,479,481],{"href":480},"what-is-human-in-the-loop-ai","What is human-in-the-loop AI"," and ",[199,484,486],{"href":485},"what-is-write-back-governance","write-back governance"," define the stop. This question tests whether they are ",[194,489,490],{},"on the job",". A ServiceNow ticket opened after the write is not a stop. A Slack reaction is not a signer.",[190,493,494,499,500,503,504,507,508,510],{},[199,495,498],{"href":496,"rel":497},"https://genai.owasp.org/llm-top-10/",[203],"OWASP’s LLM Top 10"," lists excessive agency and insecure output handling. Collaborative eval adds organisational agency: ",[194,501,502],{},"who"," could have stopped ",[194,505,506],{},"this"," change on ",[194,509,506],{}," job. If the only stop is max tokens, you have a fuse, not a control plane.",[382,512,514],{"id":513},"_4-are-files-in-one-place-not-five-inboxes","4. Are files in one place, not five inboxes?",[190,516,517,519],{},[194,518,390],{}," Attachments live on the job — WMS snapshot, ageing extract, partner PO — visible to the roster without re-forwarding.",[190,521,522,524],{},[194,523,396],{}," “Paste into the chat window.” “The model will fetch from SharePoint if you paste the link.”",[190,526,527,529],{},[194,528,402],{}," List attachments on the job object. Remove the original uploader from the roster. Files remain.",[190,531,532,536,537,541,542,545],{},[199,533,535],{"href":534},"search-is-not-memory","Search is not memory"," is why “we’ll find it in Slack later” fails eval. Do not let the vendor substitute a retrieval demo for an attachment that survives user removal. ",[199,538,540],{"href":539},"what-is-institutional-memory-in-enterprise-ai","Institutional memory in enterprise AI"," is the company-scale layering; this question only asks whether ",[194,543,544],{},"this job"," still has its two files on Monday.",[382,547,549],{"id":548},"_5-what-happens-after-the-session","5. What happens after the session?",[190,551,552,554],{},[194,553,390],{}," Monday reopen shows brief, files, last proposal, last rejection or signature — even if the original prompter is out and the model vendor changed.",[190,556,557,559],{},[194,558,396],{}," Session expiry deletes context. “Memory” is the user’s personal thread.",[190,561,562,564],{},[194,563,402],{}," Close the browser. Reopen with a different user. Continue the job without reconstruction.",[190,566,567,571,572,576],{},[199,568,570],{"href":569},"agents-should-be-disposable","Agents should be disposable"," is the design claim this question tests. ",[199,573,575],{"href":574},"what-is-an-ai-workstream","What an AI workstream is"," is the container name. If continuity requires the original model or the original person, you scored a session, not a job.",[190,578,579,584],{},[199,580,583],{"href":581,"rel":582},"https://www.iso.org/standard/42001",[203],"ISO/IEC 42001"," wants records and named actors. A session that evaporates is not a record. Ask for an export a later reader can use without vendor professional services.",[382,586,588],{"id":587},"_6-does-governance-live-in-the-room","6. Does governance live in the room?",[190,590,591,593,594,596,597,601],{},[194,592,390],{}," Roster, inherited authority, spend-cap pause, and notify-to-roster on ",[194,595,544],{}," — see ",[199,598,600],{"href":599},"governance-as-a-multiplayer-primitive","governance as a multiplayer primitive",".",[190,603,604,606],{},[194,605,396],{}," Governance PDF emailed quarterly while writes succeed unsigned.",[190,608,609,611,612,616],{},[194,610,402],{}," Show spend-cap alert to roster members only. Show agent grants bounded to the acting person’s authority. ",[199,613,615],{"href":614},"rbac-for-enterprise-ai","RBAC for enterprise AI"," is the access vocabulary; demand it scoped to the job.",[190,618,619,620,623],{},"A tenant-wide “AI policy acknowledged” checkbox is not this test. Neither is an SOC 2 report. Those are programme artefacts. This question is whether finance sees the same payload ops sees ",[194,621,622],{},"before"," execute.",[259,625,627],{"id":626},"scorecard-false-passes-to-reject","Scorecard: false passes to reject",[629,630,631,647],"table",{},[632,633,634],"thead",{},[635,636,637,641,644],"tr",{},[638,639,640],"th",{},"Demo looks like",[638,642,643],{},"Likely false pass",[638,645,646],{},"Ask instead",[648,649,650,662,673,684,695,706,717],"tbody",{},[635,651,652,656,659],{},[653,654,655],"td",{},"Fluent multi-user chat",[653,657,658],{},"Shared login",[653,660,661],{},"Job ID, roster, survive user removal",[635,663,664,667,670],{},[653,665,666],{},"Agent orchestra",[653,668,669],{},"Multi-agent without multiplayer",[653,671,672],{},"Second department join + named stop",[635,674,675,678,681],{},[653,676,677],{},"Copilot in CRM sidebar",[653,679,680],{},"Personal assistant",[653,682,683],{},"Cross-team exception with finance on job",[635,685,686,689,692],{},[653,687,688],{},"“We integrate Salesforce”",[653,690,691],{},"Connector without payload quote",[653,693,694],{},"Show field-level payload before write",[635,696,697,700,703],{},[653,698,699],{},"Long context window",[653,701,702],{},"Memory",[653,704,705],{},"Reopen Monday without original thread",[635,707,708,711,714],{},[653,709,710],{},"“Human review” node",[653,712,713],{},"Abstract HITL",[653,715,716],{},"Named signer + stored rejection on job",[635,718,719,722,725],{},[653,720,721],{},"Quarterly attestation",[653,723,724],{},"PDF after the write",[653,726,727],{},"Fail-closed unsigned attempt",[190,729,730,731,736],{},"NIST’s ",[199,732,735],{"href":733,"rel":734},"https://www.nist.gov/itl/ai-risk-management-framework",[203],"AI Risk Management Framework"," Measure function assumes you can observe outcomes. A demo that cannot produce a stored rejection has nothing to measure except fluency.",[259,738,740],{"id":739},"proof-of-value-script-two-weeks","Proof-of-value script (two weeks)",[190,742,743],{},"Do not let the vendor script a happy-path email draft. Use one exception you already run.",[190,745,746],{},[194,747,748],{},"Week one — read-only, multiplayer:",[750,751,752,755,758,761],"ol",{},[305,753,754],{},"Name one real exception: credit hold, discount outside grid, order hold in WMS.",[305,756,757],{},"Put ops and finance on the roster. Attach two files they already email.",[305,759,760],{},"Mid-week, add a second-department guest with a scoped view.",[305,762,763,764,767],{},"Model drafts release; finance ",[194,765,766],{},"rejects","; rejection must stay on the job.",[190,769,770],{},[194,771,772],{},"Week two — continuity and optional write:",[750,774,776,779,782,788],{"start":775},5,[305,777,778],{},"Swap the prompter. Reopen. A stranger continues without Slack archaeology.",[305,780,781],{},"Swap model tier or vendor if the product claims portability.",[305,783,784,785,601],{},"If write is in scope: enable fail-closed write with payload quote; show unsigned attempt ",[194,786,787],{},"failed",[305,789,790],{},"Independent reader reconstructs signer and payload without authors in the room.",[190,792,793],{},"If steps 4 or 5 fail, you do not have collaborative AI — you have a group chat with AI autocomplete. Stop the proof. Do not “save write for phase two” as a way to skip the rejection test. The rejection is the point.",[190,795,796,797,801,802,801,806,801,810,801,814,818],{},"Function-specific walkthroughs if you need a scenario library: ",[199,798,800],{"href":799},"collaborative-ai-for-operations","operations",", ",[199,803,805],{"href":804},"collaborative-ai-for-revenue-operations","revenue operations",[199,807,809],{"href":808},"collaborative-ai-for-customer-support","customer support",[199,811,813],{"href":812},"collaborative-ai-for-human-resources","human resources",[199,815,817],{"href":816},"collaborative-ai-for-finance-and-planning","finance and planning",". Use one. Do not run six proofs.",[259,820,822],{"id":821},"how-this-differs-from-harness-evaluation","How this differs from harness evaluation",[190,824,825,827],{},[199,826,232],{"href":231}," asks:",[302,829,830,833,836],{},[305,831,832],{},"Can the harness refuse a write?",[305,834,835],{},"Can you replay signer and policy version?",[305,837,838],{},"Can you swap the model without rewriting tools?",[190,840,841],{},"Collaborative eval asks:",[302,843,844,851,854],{},[305,845,846,847,850],{},"Can two departments ",[194,848,849],{},"see"," the same refusal?",[305,852,853],{},"Does the job survive people and agents leaving?",[305,855,856,857,859],{},"Are files and stops on the ",[194,858,236],{},", not in personal threads?",[190,861,862],{},"You need both when the purchase is “AI for cross-team exceptions that may touch CRM, ERP, or WMS.” Harness without collaborative passes produces a gated write finance never saw coming. Collaborative without harness passes produces a shared room where unsigned writes still slip through.",[190,864,865,869,870,874],{},[199,866,868],{"href":867},"agent-harness-vs-agent-framework","Agent harness vs agent framework"," is the build-versus-compose warning: a graph in a notebook is not a passed test. ",[199,871,873],{"href":872},"inner-vs-outer-agent-harness","Inner vs outer agent harness"," is why a coding-harness scorecard will mis-score an outer operations job. Use the right sibling sheet. Do not grade a warehouse hold with SWE-bench.",[190,876,877,881],{},[199,878,880],{"href":879},"what-is-ai-governance","What is AI governance"," is the programme within which both sheets fit. Collaborative AI is not a feature checkbox. It is whether shared work survives the people and models that staffed it this week.",[259,883,885],{"id":884},"vendor-questions-to-copy-into-procurement","Vendor questions to copy into procurement",[190,887,888],{},"Number them. Require a live show, a recording, or a written fail.",[750,890,891,894,897,900,903,906,909,912],{},[305,892,893],{},"Show one job ID with two departments and different tool grants on the same roster.",[305,895,896],{},"Show a rejection stored on the job with signer identity — not an email log.",[305,898,899],{},"Remove the original uploader; attachments remain.",[305,901,902],{},"Add a guest; guest cannot inherit write token.",[305,904,905],{},"Pause on spend cap; notify roster members; show delegate approval on the job.",[305,907,908],{},"Reopen after 72 hours with a different model; show continuity.",[305,910,911],{},"Attempt unsigned write to a system of record; show fail-closed.",[305,913,914],{},"Export replay — signer, payload, policy version — without vendor professional services.",[190,916,917],{},"If the vendor answers questions 1–6 with “our SI will configure that,” treat configuration time as part of the price. A product that needs six months of graph work before a second department can join is a framework purchase, not a collaborative-AI purchase. Say so in the scoring notes.",[190,919,920],{},"Ask incumbents the same questions you ask specialists. A CRM copilot that cannot put finance on the job fails this sheet even if it drafts beautiful emails. A multi-agent platform that cannot store a rejection fails this sheet even if the orchestra is elegant.",[259,922,924],{"id":923},"when-collaborative-eval-is-not-the-first-buy","When collaborative eval is not the first buy",[190,926,927],{},"Stay with personal assistants when:",[302,929,930,933,936],{},[305,931,932],{},"One owner drafts and nothing writes to a live system.",[305,934,935],{},"No second department must stand on the result this quarter.",[305,937,938],{},"The pain is blank-page speed, not lost outcomes.",[190,940,941,943],{},[199,942,217],{"href":216}," is the decision tree. Buy collaborative when the recurring meeting already exists and the pain is “we never keep the outcome.”",[190,945,946],{},"Do not force a roster onto solo research. Theatre rosters teach people that governance is ceremony. Save the sheet for jobs that already have a fight in chat.",[259,948,950],{"id":949},"how-to-run-the-bake-off","How to run the bake-off",[190,952,953,954,957],{},"Score every shortlisted product on a ",[194,955,956],{},"cross-team"," job, not the AE’s private email draft. Use the same exception, the same two files, the same two departments. Keep a shared scorecard with pass/fail per question above — not a 1–5 “wow” rating on fluency.",[190,959,960,961,482,965,969],{},"Include at least one incumbent copilot and at least one agent platform if both are on the table. The point of the sheet is comparison, not a single-vendor script. ",[199,962,964],{"href":963},"what-is-an-enterprise-agent-harness","What is an enterprise agent harness",[199,966,968],{"href":967},"what-is-an-enterprise-ai-operating-system","what is an enterprise AI operating system"," are category pages if you need language for the stack around the job. This sheet still scores the job.",[190,971,972],{},"Put this sheet in the RFP, then run it in a proof of value. A fluent demo without a stored rejection is still a slide.",[259,974,976],{"id":975},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[190,978,979,980,983,984,482,987,990],{},"When you include Nimbus in a bake-off, run ",[194,981,982],{},"this same sheet"," on ",[199,985,986],{"href":32},"workstreams",[199,988,989],{"href":40},"governance",". Do not substitute a homepage video or a pre-built demo room.",[190,992,993,994,601],{},"Ask for the stored rejection, the mid-run finance join, the reopen after a model swap, and the unsigned write that fails. Score incumbents on the same exception. Staffing and time-to-value for the bake-off itself are a separate frame — see ",[199,995,997],{"href":996},"/evaluate/","self-service vs forward-deployed",[190,999,1000],{},"If Nimbus cannot show the “no” on the job Monday, it fails this sheet the way any other vendor would. Collaborative eval is not a product tour.",[1002,1003,1004],"style",{},"html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":171,"searchDepth":172,"depth":172,"links":1006},[1007,1008,1016,1017,1018,1019,1020,1021,1022],{"id":261,"depth":172,"text":262},{"id":376,"depth":172,"text":377,"children":1009},[1010,1011,1012,1013,1014,1015],{"id":384,"depth":294,"text":385},{"id":412,"depth":294,"text":413},{"id":447,"depth":294,"text":448},{"id":513,"depth":294,"text":514},{"id":548,"depth":294,"text":549},{"id":587,"depth":294,"text":588},{"id":626,"depth":172,"text":627},{"id":739,"depth":172,"text":740},{"id":821,"depth":172,"text":822},{"id":884,"depth":172,"text":885},{"id":923,"depth":172,"text":924},{"id":949,"depth":172,"text":950},{"id":975,"depth":172,"text":976},"2026-09-12","An RFP sheet for shared jobs — not copilots. Questions on roster, stored rejection, second department, files in one place, and what survives after the session. Plain language for operators.",{"eyebrow":1026,"title":1027},"Common questions","Score the shared job",[1029,1032,1035,1038],{"question":1030,"answer":1031},"Is this the same as evaluating a copilot?","No. Copilot evals score solo drafting — latency, paragraph quality, seat SSO. This sheet scores whether two departments share one job, with a named stop and artefacts that survive the session. A product can pass every copilot test and still fail the moment finance joins Thursday’s exception. If your RFP only asks context-window size and brand of model, you will buy an assistant and call it collaboration. Read the personal-assistant comparison first if you are still deciding whether you need a shared room at all.",{"question":1033,"answer":1034},"Do we need multiplayer and collaborative AI in the RFP?","Ask both. Multiplayer is who is in the room now — can a second department join, see the same brief, and reject a proposal while the work is happening. Collaborative is whether Monday still has the files, the rejection, and the signer after everyone hangs up. You can fail one and pass the other. Five people in a call with nothing written down is multiplayer for an hour. A job two teams open on different days can be collaborative without a live huddle. Cross-department work that may write to a live system needs both.",{"question":1036,"answer":1037},"What is the fastest proof-of-value test?","Invite a second department mid-run, reject a proposal, swap the prompter, reopen Monday. If the job is empty, you evaluated a shared login. Do not let the vendor pick a solo email-draft scenario. Name a real exception you already fight in chat. Require a stored rejection with a name and a timestamp before anyone enables write. An independent reader who was not in the demo should reconstruct signer and payload without Slack. If they cannot, fail the proof even if the paragraph was fluent.",{"question":1039,"answer":1040},"Can we reuse our agent-harness RFP instead?","Reuse it as a sibling, not a substitute. The harness sheet scores refused writes, replay, and model swap on the runtime. This sheet scores roster, handover, and whether files live on the job. You need both when the purchase is AI for cross-team exceptions that may touch CRM, ERP, or WMS. Harness without collaborative passes produces a gated write finance never saw. Collaborative without harness passes produces a shared room where unsigned writes still slip through. Copy both question lists into procurement.","/blog/how-to-evaluate-collaborative-ai",{"title":181,"description":1024},"evaluation","blog/how-to-evaluate-collaborative-ai",[1043,1046,1047,1048],"collaborative-ai","RFP","multiplayer","HE9Si4bvYjh3cdtW3TifBd1idlhzsISzDNXFWiG50wY",{"hero":1051,"id":1053,"title":1054,"archived":165,"authors":166,"badge":166,"body":1055,"date":166,"definedTerm":166,"department":166,"description":1059,"extension":174,"eyebrow":1060,"faqHeader":166,"faqs":166,"footerBand":1061,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":60,"relatedHeading":1067,"seo":1068,"series":166,"sitemap":131,"status":166,"stem":1069,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1070},{"filename":1052},"u2221455217_Flat_design_of_a_futuristic_minimalist_landscape__5d589295-cdea-4ea9-a262-be766881accf_1.png","content/blog/index.md","Exploring the future of intelligence.",{"type":168,"value":1056,"toc":1057},[],{"title":171,"searchDepth":172,"depth":172,"links":1058},[],"Deep dives into pre-cognitive intelligence, sentient enterprises, and the evolving landscape of AI-driven business transformation.","Latest Research",{"headline":1062,"description":1063,"primaryLabel":1064,"primaryTo":1065,"secondaryLabel":1066,"secondaryTo":12},"Stay at the frontier.","Subscribe for product updates and new insights.","Subscribe","/newsletter","Explore the platform","More research",{"title":1054,"description":1059},"blog/index","BFSWGYO9bcTlaulivKYWyg08_DJHsdGg3OC6g_CG1Hw",[1072,166],{"id":1073,"title":1074,"archived":165,"authors":1075,"badge":1077,"body":1078,"date":1023,"definedTerm":166,"department":166,"description":1755,"extension":174,"eyebrow":166,"faqHeader":1756,"faqs":1758,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1770,"relatedHeading":166,"seo":1771,"series":1043,"sitemap":131,"status":166,"stem":1772,"subhead":166,"tags":1773,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1776},"content/blog/how-to-evaluate-loop-engineering.md","How to Evaluate Loop Engineering",[1076],{"name":184,"to":136},{"label":186},{"type":168,"value":1079,"toc":1737},[1080,1096,1099,1102,1151,1159,1163,1166,1186,1205,1208,1228,1236,1240,1243,1268,1272,1283,1289,1292,1295,1298,1302,1305,1308,1311,1314,1317,1321,1328,1339,1342,1345,1348,1352,1359,1365,1368,1371,1374,1378,1394,1397,1400,1403,1406,1410,1413,1425,1428,1431,1434,1438,1443,1446,1449,1452,1455,1459,1465,1478,1481,1484,1487,1491,1494,1524,1530,1534,1537,1543,1549,1555,1561,1564,1566,1653,1665,1672,1676,1679,1689,1699,1708,1720,1722,1732,1735],[190,1081,1082,1083,1086,1087,1090,1091,1095],{},"Evaluating ",[194,1084,1085],{},"loop engineering"," means testing whether a vendor can run ",[194,1088,1089],{},"standing orders"," the way operators actually work — skip when nothing changed, record quiet outcomes, reuse recipes, notify a named roster — not whether a demo chat looked fluent. ",[199,1092,1094],{"href":733,"rel":1093},[203],"NIST’s AI Risk Management Framework"," Measure and Manage functions assume you can observe outcomes and tighten controls after harm. A loop without a run page is not observable. It is email archaeology.",[190,1097,1098],{},"This sheet is vendor-agnostic. Paste it into an RFP. Run it on an incumbent automation suite, a new “agentic” platform, or a homegrown scheduler. The object under test is the standing order: a compiled recipe that starts on a signal and leaves a record. If the vendor cannot show that object, stop scoring adjectives.",[190,1100,1101],{},"Three nouns vendors conflate on slide one:",[302,1103,1104,1118,1128],{},[305,1105,1106,1109,1110,482,1114,601],{},[194,1107,1108],{},"Loop."," A compiled standing order with triggers and a run page. See ",[199,1111,1113],{"href":1112},"what-is-loop-engineering","what is loop engineering",[199,1115,1117],{"href":1116},"six-things-that-start-a-loop","six things that start a loop",[305,1119,1120,1123,1124,601],{},[194,1121,1122],{},"Eval loop."," Independent verification that a job finished — tests, read-back, signer — not a trigger. See ",[199,1125,1127],{"href":1126},"eval-loops-for-enterprise-agent-harnesses","eval loops for enterprise agent harnesses",[305,1129,1130,1133,1134,1138,1139,1142,1143,1146,1147,1150],{},[194,1131,1132],{},"Agentic workflow."," A designed interactive sequence with stops. See ",[199,1135,1137],{"href":1136},"what-is-an-agentic-workflow","what is an agentic workflow",". Score it with ",[199,1140,1141],{"href":231},"how to evaluate an agent harness",". ",[194,1144,1145],{},"That"," sheet scores the interactive harness. ",[194,1148,1149],{},"This"," sheet scores repeat automation.",[190,1152,1153,1154,1158],{},"If your RFP only asks context-window size, you will buy a model for a Monday cron job. ",[199,1155,583],{"href":1156,"rel":1157},"https://www.iso.org/standard/81230.html",[203]," wants named actors and records around AI systems. “The bot ran” is not a record. “The assistant said it was done” is not a record.",[259,1160,1162],{"id":1161},"why-loop-evaluation-fails","Why loop evaluation fails",[190,1164,1165],{},"Teams reuse copilot scorecards because those cards are already in the drawer. They fail in predictable ways:",[750,1167,1168,1174,1180],{},[305,1169,1170,1173],{},[194,1171,1172],{},"The demo pass."," A model summarises a file beautifully. No trigger, no skip, no version. You have scored reading, not standing orders.",[305,1175,1176,1179],{},[194,1177,1178],{},"The integration pass."," “We connect to Salesforce.” No run page, no quiet outcome, no roster. You have scored a connector, not an operating object.",[305,1181,1182,1185],{},[194,1183,1184],{},"The personal-automation pass."," A chain fires. Outcomes scatter across personal inboxes. No organisational recipe. You have scored glue, not loop engineering.",[190,1187,1188,1192,1193,1196,1197,1200,1201,1204],{},[199,1189,1191],{"href":201,"rel":1190},[203],"McKinsey’s 2025 State of AI"," reports wide experimentation and narrower scale. Loop engineering is how repeat work scales: ",[194,1194,1195],{},"compile"," what worked, ",[194,1198,1199],{},"trigger"," it reliably, ",[194,1202,1203],{},"record"," what happened. High-performing organisations in that survey are more likely to redesign workflows, not merely sprinkle assistants on the current process. This RFP is a redesign test.",[190,1206,1207],{},"Red flags you can mark in the room:",[302,1209,1210,1213,1216,1219,1222,1225],{},[305,1211,1212],{},"Outcomes live only in chat transcripts",[305,1214,1215],{},"“Success” when there was nothing to do — but Finance still got paged",[305,1217,1218],{},"Reuse means “find Sarah’s Slack thread from Q2”",[305,1220,1221],{},"Every write goes through a model — even fixed journal templates",[305,1223,1224],{},"No pause — cancel kills state without a resumable run",[305,1226,1227],{},"Notifications go to channels, not to a roster attached to the job",[190,1229,1230,1235],{},[199,1231,1234],{"href":1232,"rel":1233},"https://aiindex.stanford.edu/",[203],"Stanford HAI’s AI Index"," is a useful external reminder that capability is not the scarce input. You are not scoring whether the model can write a polite email. You are scoring whether Tuesday’s job exists when the author is on leave.",[259,1237,1239],{"id":1238},"the-rfp-sheet-eight-tests","The RFP sheet — eight tests",[190,1241,1242],{},"Each test has a why, a when, a thing to do, and a thing to refuse. Run them in order if you are short on time: skip semantics first, then run page, then reuse. A fluent demo that fails test 1 is not a standing order.",[271,1244,1246],{"className":273,"code":1245,"language":275,"meta":171,"style":171},"flowchart LR\n  skipTest[\"Can it skip quietly?\"] --> pageTest[\"Is there a run page?\"]\n  pageTest --> reuseTest[\"Can another team reuse it?\"]\n  reuseTest --> refuseTest[\"Can someone refuse a write?\"]\n",[277,1247,1248,1252,1257,1262],{"__ignoreMap":171},[280,1249,1250],{"class":282,"line":283},[280,1251,286],{},[280,1253,1254],{"class":282,"line":172},[280,1255,1256],{},"  skipTest[\"Can it skip quietly?\"] --> pageTest[\"Is there a run page?\"]\n",[280,1258,1259],{"class":282,"line":294},[280,1260,1261],{},"  pageTest --> reuseTest[\"Can another team reuse it?\"]\n",[280,1263,1265],{"class":282,"line":1264},4,[280,1266,1267],{},"  reuseTest --> refuseTest[\"Can someone refuse a write?\"]\n",[382,1269,1271],{"id":1270},"_1-can-it-skip-when-nothing-changed","1. Can it skip when nothing changed?",[190,1273,1274,1275,1278,1279,1282],{},"Run the loop on unchanged inputs. The run page should say ",[194,1276,1277],{},"skipped"," or ",[194,1280,1281],{},"no op"," — with timestamp and recipe version — not green-check spam.",[190,1284,1285,1286,1288],{},"Why it matters: month-end with zero exceptions is success. Bots that “succeed” on empty tables train operators to ignore alerts. See ",[199,1287,1117],{"href":1116}," for triggers that should dedupe.",[190,1290,1291],{},"When to insist: any scheduled or event-driven job that will run in unattended hours.",[190,1293,1294],{},"What to do: show ten consecutive skipped runs. Show alert volume — ideally zero.",[190,1296,1297],{},"What to refuse: a “success” email on an empty extract, or a skip that exists only as the absence of mail. Absence is not a record.",[382,1299,1301],{"id":1300},"_2-can-it-record-quiet-outcomes","2. Can it record quiet outcomes?",[190,1303,1304],{},"Quiet success is a first-class outcome: ran, nothing to write, roster optionally notified at digest frequency.",[190,1306,1307],{},"Why it matters: auditors and controllers ask what happened on the twelfth — including days nothing moved. Silence without a record is indistinguishable from failure. NIST’s Measure function is not optional because the week was quiet.",[190,1309,1310],{},"When to insist: regulated work, shared-service work, anything a second person will reconstruct.",[190,1312,1313],{},"What to do: export skipped runs for a month without vendor engineering.",[190,1315,1316],{},"What to refuse: a professional-services quote to produce last month’s quiet days. If export is a project, observation is not a product feature.",[382,1318,1320],{"id":1319},"_3-can-you-parameterise-a-write-without-a-model","3. Can you parameterise a write without a model?",[190,1322,1323,1324,1327],{},"Show a loop that posts or quotes a ",[194,1325,1326],{},"fixed-shape"," payload — accounts, amounts, CRM fields — from structured input, with no language model in the path.",[190,1329,1330,1331,482,1335,601],{},"Why it matters: many operational writes are templates, not essays. If the vendor routes everything through chat, you are paying inference tax on deterministic work and importing non-determinism into close. Compare ",[199,1332,1334],{"href":1333},"loop-vs-workflow-vs-agent","loop vs workflow vs agent",[199,1336,1338],{"href":1337},"a-loop-is-not-an-agent","a loop is not an agent",[190,1340,1341],{},"When to insist: journals, stage updates, status writes, any payload a controller could have typed from a spreadsheet.",[190,1343,1344],{},"What to do: disable the model. Does the loop still quote the write and wait on sign?",[190,1346,1347],{},"What to refuse: “the model is more flexible” as an answer to a fixed schema. Flexibility on a journal line is a defect.",[382,1349,1351],{"id":1350},"_4-can-it-pause-and-resume-cleanly","4. Can it pause and resume cleanly?",[190,1353,1354,1355,1358],{},"A loop waiting on a file, a signer, or an external system should ",[194,1356,1357],{},"pause"," with visible state — not vanish into a thread.",[190,1360,1361,1362,1364],{},"Why it matters: close week spans days. Operations must distinguish “waiting on a person” from “broken.” ",[199,1363,481],{"href":480}," applies to standing orders too.",[190,1366,1367],{},"When to insist: any recipe that crosses a night, a weekend, or a named approver.",[190,1369,1370],{},"What to do: pause mid-run. Attach the missing file. Resume without restarting from scratch unless you choose to. Keep the same run identifier.",[190,1372,1373],{},"What to refuse: cancel-as-pause. If state dies, you do not have a pause. You have a restart with extra steps.",[382,1375,1377],{"id":1376},"_5-can-it-notify-the-roster-not-a-copied-channel","5. Can it notify the roster — not a copied channel?",[190,1379,1380,1381,1384,1385,1388,1389,1393],{},"Notifications should target the people on the job — controller, RevOps, counsel — with a link to the ",[194,1382,1383],{},"run page",". See ",[199,1386,1387],{"href":574},"what is an AI workstream"," for the job-object idea, and ",[199,1390,1392],{"href":1391},"how-to-evaluate-collaborative-ai","how to evaluate collaborative AI"," for the roster questions.",[190,1395,1396],{},"Why it matters: Slack channels rot when people leave. Rosters follow the job. A copied channel is how a departed contractor keeps getting close packs, and how the new controller never does.",[190,1398,1399],{},"When to insist: any loop that another department will act on.",[190,1401,1402],{},"What to do: remove one person from the roster. Prove they stop receiving loop notifications without creating a new automation.",[190,1404,1405],{},"What to refuse: “we’ll update the webhook.” If membership is not data, notification is folklore.",[382,1407,1409],{"id":1408},"_6-can-you-reuse-a-recipe-without-copying-the-old-slack-channel","6. Can you reuse a recipe without copying the old Slack channel?",[190,1411,1412],{},"Clone the loop — triggers, steps, gates, notify rules — into a new team or region without re-prompting from memory.",[190,1414,1415,1416,1418,1419,1424],{},"Why it matters: ",[199,1417,1085],{"href":1112}," is an organisational capability, not hero prompts. If reuse requires export to JSON and a services quote, note the tax. ",[199,1420,1423],{"href":1421,"rel":1422},"https://www.thoughtworks.com/insights/articles/operating-system-enterprise-ai",[203],"Thoughtworks’ operating-system framing"," is useful here: ownership and durable state belong to the job, not to the person who first described it.",[190,1426,1427],{},"When to insist: any recipe you will need in a second business unit within a year.",[190,1429,1430],{},"What to do: stand up the same loop for a second team in under one hour — operator-led.",[190,1432,1433],{},"What to refuse: a clone that copies the prompt but drops the skip rules, the signer, or the run-page contract. That is a new folklore, not reuse.",[382,1435,1437],{"id":1436},"_7-does-every-trigger-land-on-the-same-run-page","7. Does every trigger land on the same run page?",[190,1439,1440,1441,601],{},"Schedule, file, data change, drop, ping, run now — several doors, one outcome surface. See ",[199,1442,1117],{"href":1116},[190,1444,1445],{},"Why it matters: operators should not learn six UIs. Audit should not merge six log formats. If the scheduled close and the emergency rerun do not look like the same object, you will get two classes of evidence.",[190,1447,1448],{},"When to insist: as soon as a team has more than one start condition for the same recipe.",[190,1450,1451],{},"What to do: fire two different triggers against the same recipe. Show both run pages side by side.",[190,1453,1454],{},"What to refuse: a “manual” path that writes with weaker gates than the scheduled path. Urgency is not a policy exception.",[382,1456,1458],{"id":1457},"_8-can-you-refuse-a-write-and-prove-the-system-of-record-unchanged","8. Can you refuse a write and prove the system of record unchanged?",[190,1460,1461,1462,601],{},"Even loops that only draft should demonstrate fail-closed behaviour when signers reject. Loops that write must show quote → sign → execute → read-back. See ",[199,1463,1464],{"href":485},"what is write-back governance",[190,1466,1467,1468,1470,1471,1474,1475,1477],{},"Why it matters: this is where loop evaluation meets harness evaluation. ",[199,1469,232],{"href":231}," test one — ",[194,1472,1473],{},"stop an action"," — applies to automated writes too. ",[199,1476,880],{"href":879}," is the category language.",[190,1479,1480],{},"When to insist: before any production write, including “just a status field.”",[190,1482,1483],{},"What to do: show a rejected payload. Show the CRM or ERP unchanged. Show the rejection on the run page, with who rejected and which policy version applied.",[190,1485,1486],{},"What to refuse: a write that cannot be shown in a quoted form. A paragraph the model later “applies” is not a payload.",[259,1488,1490],{"id":1489},"rfp-questions-paste-these","RFP questions — paste these",[190,1492,1493],{},"These are the short versions you can drop into a vendor questionnaire. They are not vendor-specific. They are not even AI-specific. They are standing-order tests.",[750,1495,1496,1499,1506,1509,1512,1515,1518,1521],{},[305,1497,1498],{},"Show ten skipped runs with timestamps and recipe version.",[305,1500,1501,1502,1505],{},"Show a write path with ",[194,1503,1504],{},"no model call"," — structured in, quoted out.",[305,1507,1508],{},"Pause a run for 48 hours; resume; show a continuous run identifier.",[305,1510,1511],{},"Clone a loop to a second team without re-entering prompts.",[305,1513,1514],{},"Trigger the same recipe via schedule and via file drop; compare run pages.",[305,1516,1517],{},"Remove a roster member; prove notifications stop.",[305,1519,1520],{},"Reject a quoted write; prove the system of record unchanged; show who rejected.",[305,1522,1523],{},"Where do outputs live if email is down — still on the run page?",[190,1525,1526,1527,1529],{},"Add the harness sheet when loops call agents or share connectors with interactive work. Add ",[199,1528,1392],{"href":1391}," when the output is a signed pack rather than a silent write.",[259,1531,1533],{"id":1532},"proof-of-value-one-week","Proof of value — one week",[190,1535,1536],{},"Do not spend the week watching a prepared demo. Spend it on one real job.",[190,1538,1539,1542],{},[194,1540,1541],{},"Day 1–2:"," Pick one repeat job — weekly pipeline summary, bank file intake, redline folder watch. Name the trigger your team already watches. Write the skip condition in a sentence a controller would accept.",[190,1544,1545,1548],{},[194,1546,1547],{},"Day 3:"," Run ten times with empty or stale inputs. Demand skipped run pages. If you cannot get them, the rest of the week is theatre.",[190,1550,1551,1554],{},[194,1552,1553],{},"Day 4:"," Run once with a real change. Confirm roster notification links to the run — not a pasted screenshot. Confirm an independent reader can open the run without the author.",[190,1556,1557,1560],{},[194,1558,1559],{},"Day 5:"," Clone to a second roster or region. Time it. Attempt a refused write. Confirm the system of record did not move.",[190,1562,1563],{},"Pass criteria: skip semantics, run page, roster notify, reuse, and one refused or unsigned write that did not land. Fail any one and you do not have loop engineering. You have a demo that will not survive the first quiet week.",[259,1565,822],{"id":821},[629,1567,1568,1585],{},[632,1569,1570],{},[635,1571,1572,1575,1578],{},[638,1573,1574],{},"Topic",[638,1576,1577],{},"Loop engineering (this page)",[638,1579,1580,1581,1584],{},"Agent harness (",[199,1582,1583],{"href":231},"sibling sheet",")",[648,1586,1587,1598,1609,1620,1631,1642],{},[635,1588,1589,1592,1595],{},[653,1590,1591],{},"Unit of buy",[653,1593,1594],{},"Standing order / recipe",[653,1596,1597],{},"Interactive runtime",[635,1599,1600,1603,1606],{},[653,1601,1602],{},"Hero metric",[653,1604,1605],{},"Skipped vs processed runs",[653,1607,1608],{},"Refused writes / replay",[635,1610,1611,1614,1617],{},[653,1612,1613],{},"Trigger surface",[653,1615,1616],{},"Several signal types",[653,1618,1619],{},"User goal / chat",[635,1621,1622,1625,1628],{},[653,1623,1624],{},"Quiet success",[653,1626,1627],{},"No op recorded",[653,1629,1630],{},"Waiting on signer",[635,1632,1633,1636,1639],{},[653,1634,1635],{},"Reuse",[653,1637,1638],{},"Recipe library",[653,1640,1641],{},"Versioned harness + tools",[635,1643,1644,1647,1650],{},[653,1645,1646],{},"Model role",[653,1648,1649],{},"Optional, bounded step",[653,1651,1652],{},"Often central",[190,1654,1655,1656,1659,1660,1664],{},"You need both sheets if ",[199,1657,1658],{"href":346},"the four pillars"," describe your stack — Automate (loops) plus Collaboration (agents on workstreams) under governance. ",[199,1661,1663],{"href":1662},"loop-engineering-vs-harness-engineering","Loop engineering vs harness engineering"," explains the crafts. This page and the harness page are how you buy them separately.",[190,1666,1667,1671],{},[199,1668,1670],{"href":1669},"loop-vs-rpa","Loop vs RPA"," is the adjacent comparison if incumbents sell bots. Do not let an RPA success-email farm satisfy test 1. “The script finished” is not a skipped run.",[259,1673,1675],{"id":1674},"department-lenses","Department lenses",[190,1677,1678],{},"Use the same eight tests. Change the job you bring to the POV.",[190,1680,1681,1684,1685,601],{},[194,1682,1683],{},"Finance"," — scheduled close packs, parameterised journals, controller on the roster. Skip on a week with no exceptions. See ",[199,1686,1688],{"href":1687},"loops-for-finance-and-planning","loops for finance and planning",[190,1690,1691,1694,1695,601],{},[194,1692,1693],{},"RevOps"," — stage-triggered hygiene, skip when fields are unchanged, notify the people who own the stage definition. See ",[199,1696,1698],{"href":1697},"loops-for-revenue-operations","loops for revenue operations",[190,1700,1701,1703,1704,601],{},[194,1702,64],{}," — inbound redlines, notify counsel, no send without a workflow gate. The loop starts the job; the workflow owns the stop. See ",[199,1705,1707],{"href":1706},"loops-for-legal-and-compliance","loops for legal and compliance",[190,1709,1710,1711,801,1714,1716,1717,1719],{},"Related reading: ",[199,1712,1713],{"href":211},"what is collaborative AI",[199,1715,1338],{"href":1337},", and ",[199,1718,968],{"href":967}," when you need the kernel metaphor rather than the RFP sheet.",[259,1721,976],{"id":975},[190,1723,1724,1725,1727,1728,1731],{},"Nimbus Loops are one implementation of the standing-order object this sheet scores: triggers, recipes, run pages, roster notify, and ",[199,1726,989],{"href":40}," gates on a ",[199,1729,1730],{"href":32},"workstream",". You can run the eight tests there. You should also run them on whoever else claims “unattended agents” or “intelligent automation.”",[190,1733,1734],{},"Nimbus is not loop engineering. Loop engineering is whether your organisation can compile repeat work, skip quietly, and leave a record a second person can open. The product should make those tests boring. If a Nimbus demo — or any demo — cannot show ten skipped runs and one refused write, treat it as a copilot evaluation and score it on the harness sheet instead.",[1002,1736,1004],{},{"title":171,"searchDepth":172,"depth":172,"links":1738},[1739,1740,1750,1751,1752,1753,1754],{"id":1161,"depth":172,"text":1162},{"id":1238,"depth":172,"text":1239,"children":1741},[1742,1743,1744,1745,1746,1747,1748,1749],{"id":1270,"depth":294,"text":1271},{"id":1300,"depth":294,"text":1301},{"id":1319,"depth":294,"text":1320},{"id":1350,"depth":294,"text":1351},{"id":1376,"depth":294,"text":1377},{"id":1408,"depth":294,"text":1409},{"id":1436,"depth":294,"text":1437},{"id":1457,"depth":294,"text":1458},{"id":1489,"depth":172,"text":1490},{"id":1532,"depth":172,"text":1533},{"id":821,"depth":172,"text":822},{"id":1674,"depth":172,"text":1675},{"id":975,"depth":172,"text":976},"Evaluating loop engineering means asking whether standing orders can skip quietly, record outcomes on a run page, parameterise a write without a model, pause, notify the roster, and reuse a recipe without copying the old Slack channel.",{"eyebrow":1026,"title":1757},"RFP tests for standing orders",[1759,1762,1764,1767],{"question":1760,"answer":1761},"Is this the same checklist as evaluating an agent harness?","Sibling, not duplicate, and scoring them on one sheet is how you buy a model for a Monday cron job. [How to evaluate an agent harness](how-to-evaluate-an-agent-harness) scores the interactive runtime — stops, replay, sensors around agents. This page scores standing orders — triggers, skip semantics, run pages, recipe reuse. Use both if you run agents and loops on the same roster. Use only this page if the job is unattended repeat work with a known path. Refuse a vendor that answers harness questions with a skipped-run demo, or loop questions with a fluent chat. They are different objects and they fail differently.",{"question":1036,"answer":1763},"Run the same standing order ten times with empty or unchanged inputs. You should get ten run pages marked skipped — and zero spurious writes or alert storms. Then run once with a real change and confirm the roster was notified with a link to the run, not a pasted screenshot. Do this on a job the team already watches, not on a synthetic demo the vendor prepared. Refuse a POV that only shows a beautiful summary of a file. That is a copilot test. It tells you nothing about whether Tuesday’s close exists as an object when the author is out.",{"question":1765,"answer":1766},"Do we need a model in every loop?","No, and a vendor that cannot show a loop without one is selling inference, not loop engineering. Many standing orders are deterministic — compare, route, notify, quote a write for sign. Ask for a parameterised write path that does not call a model. If everything routes through chat, you are scoring a copilot. Use a model step when the input is messy and the rest of the recipe is known. Refuse a design that puts a language model in the path of a fixed journal template. You will pay twice: tokens, and the day the model invents an account code.",{"question":1768,"answer":1769},"What should we refuse even if the demo is fluent?","Refuse outcomes that live only in transcripts, and refuse “success” on empty inputs that still pages Finance. Refuse reuse that means finding last quarter’s thread, and refuse a write that cannot be shown with the model disabled. Refuse a pause that kills state, and refuse notifications that go to a copied channel rather than a roster you can edit. NIST’s AI Risk Management Framework assumes you can observe outcomes and then manage them. If you cannot export a month of skipped runs without a professional-services ticket, you cannot Measure — and you should not buy.","/blog/how-to-evaluate-loop-engineering",{"title":1074,"description":1755},"blog/how-to-evaluate-loop-engineering",[1774,1043,1047,1775],"loops","automation","0qwQuy9uhlMrnUTMza--HZ1qmoRcv2zODPJJZ7ggXZE",{"enabled":165,"message":1778,"linkLabel":79,"linkHref":80,"id":1779,"title":1780,"archived":165,"authors":166,"badge":166,"body":1781,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1785,"relatedHeading":166,"seo":1786,"series":166,"sitemap":165,"status":166,"stem":1787,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1788},"We're hiring! Join the team building the Sentient Enterprise.","content/shared/hiring.md","Hiring banner",{"type":168,"value":1782,"toc":1783},[],{"title":171,"searchDepth":172,"depth":172,"links":1784},[],"/shared/hiring",{"title":1780,"description":171},"shared/hiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",{"fold":1790,"id":1794,"title":1795,"archived":165,"authors":166,"badge":166,"body":1796,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":1800,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1804,"relatedHeading":166,"seo":1805,"series":166,"sitemap":165,"status":166,"stem":1806,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1807},{"headline":1791,"description":1792,"primaryLabel":8,"primaryTo":1793,"secondaryLabel":1066,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Start on Free.","/signup?plan=free","content/shared/cta.md","Site CTAs",{"type":168,"value":1797,"toc":1798},[],{"title":171,"searchDepth":172,"depth":172,"links":1799},[],{"headline":1801,"description":1802,"primaryLabel":8,"primaryTo":1793,"secondaryLabel":1803,"secondaryTo":85},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger. Start on Free.","Talk to our team","/shared/cta",{"title":1795,"description":171},"shared/cta","PS2VPJsszmUpMBZT6nEp8cWXCdeiN6zDRl-p8d0uY2k",1789797894187]