[{"data":1,"prerenderedAt":3815},["ShallowReactive",2],{"site-nav-content":3,"hub:evaluate:posts":179,"site-cta-content":3783,"hiring-banner-content":3803},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":132,"id":163,"title":164,"archived":165,"authors":166,"badge":166,"body":167,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":175,"relatedHeading":166,"seo":176,"series":166,"sitemap":165,"status":166,"stem":177,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":178},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","\u002Foverview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","\u002Fproduct\u002Fconflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","\u002Fproduct\u002Fagent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","\u002Fproduct\u002Flifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","\u002Fproduct\u002Fwiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","\u002Fproduct\u002Fworkstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","\u002Fproduct\u002Fperception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","\u002Fproduct\u002Fgovernance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","\u002Fmodels",{"label":47,"to":48},"Pricing","\u002Fpricing",{"label":50,"to":51},"Integrations","\u002Fintegrations",{"label":53,"to":54},"Security","\u002Fsecurity",{"label":56,"to":57},"Partners","\u002Fpartners",{"label":59,"to":60},"Insights","\u002Fblog",{"productHeading":5,"companyHeading":62,"resourcesHeading":63,"legalHeading":64,"docsLabel":65,"docsUrl":66,"statementLines":67,"copyright":70,"companyLinks":71,"resourcesLinks":86,"legalLinks":102,"socialLinks":109,"bottomLinks":119},"Company","Resources","Legal","Docs","https:\u002F\u002Fdocs.gonimbus.ai",[68,69],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[72,73,74,75,76,78,81,84],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":77,"to":57},"Partner Program",{"label":79,"to":80},"Careers","\u002Fcareers",{"label":82,"to":83},"System status","\u002Fstatus",{"label":7,"to":85},"\u002Fcontact",[87,90,93,96,99],{"label":88,"to":89},"Glossary","\u002Fglossary",{"label":91,"to":92},"Compare","\u002Fcompare",{"label":94,"to":95},"Evaluate","\u002Fevaluate",{"label":97,"to":98},"Problems","\u002Fproblems",{"label":100,"to":101},"Use cases","\u002Fuse-cases",[103,106],{"label":104,"to":105},"Terms of Service","\u002Fterms",{"label":107,"to":108},"Privacy Policy","\u002Fprivacy",[110,113,116],{"label":111,"href":112},"LinkedIn","https:\u002F\u002Fwww.linkedin.com\u002Fcompany\u002Fgonimbusai\u002F",{"label":114,"href":115},"X","https:\u002F\u002Fx.com\u002Fgonimbusai",{"label":117,"href":118},"Instagram","https:\u002F\u002Fwww.instagram.com\u002Fgonimbus_ai\u002F",[120,122,124,127,128],{"label":121,"to":105},"Terms",{"label":123,"to":108},"Privacy",{"label":125,"to":126},"Compliance","\u002Fcompliance",{"label":82,"to":83},{"label":129,"to":130,"external":131},"LLMs.txt","\u002Fllms.txt",true,{"text":133,"prompt":134},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":135,"platforms":137},{" Summarize the highlights from Nimbus's website":136},"https:\u002F\u002Fgonimbus.ai",[138,143,148,153,158],{"name":139,"label":140,"icon":141,"hrefPrefix":142},"chatgpt","ChatGPT","simple-icons:openai","https:\u002F\u002Fchatgpt.com\u002F?prompt=",{"name":144,"label":145,"icon":146,"hrefPrefix":147},"perplexity","Perplexity","mdi:magnify","https:\u002F\u002Fwww.perplexity.ai\u002Fsearch\u002Fnew?q=",{"name":149,"label":150,"icon":151,"hrefPrefix":152},"grok","Grok","simple-icons:x","https:\u002F\u002Fx.com\u002Fi\u002Fgrok?text=",{"name":154,"label":155,"icon":156,"hrefPrefix":157},"claude","Claude","simple-icons:anthropic","https:\u002F\u002Fclaude.ai\u002Fnew?q=",{"name":159,"label":160,"icon":161,"hrefPrefix":162},"google-ai","Google AI","simple-icons:google","https:\u002F\u002Fwww.google.com\u002Fsearch?udm=50&aep=11&q=","content\u002Fshared\u002Fnav.md","Site navigation",false,null,{"type":168,"value":169,"toc":170},"minimark",[],{"title":171,"searchDepth":172,"depth":172,"links":173},"",2,[],"md","\u002Fshared\u002Fnav",{"title":164,"description":171},"shared\u002Fnav","1dD7ahDRl0SQ4hz53-kKo0tEFrGaLuaztZ3PPfp6a9k",[180,948,1610,2437,3137,3384],{"id":181,"title":182,"archived":165,"authors":183,"badge":186,"body":188,"date":937,"definedTerm":166,"department":166,"description":938,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":939,"relatedHeading":166,"seo":940,"series":941,"sitemap":131,"status":166,"stem":942,"subhead":166,"tags":943,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":947},"content\u002Fblog\u002Fhow-to-choose-between-a-coding-harness-and-an-enterprise-harness.md","How to Choose Between a Coding Harness and an Enterprise Harness",[184],{"name":185,"to":136},"Nimbus Research",{"label":187},"Evaluation",{"type":168,"value":189,"toc":919},[190,211,229,249,257,262,346,363,367,370,382,403,409,415,429,433,439,445,455,467,481,495,499,505,511,521,531,537,544,548,600,614,628,632,643,652,668,680,689,692,700,717,724,727,747,751,756,759,763,774,778,781,785,788,792,804,808,818,822],[191,192,193,194,198,199,202,203,210],"p",{},"Choosing between a ",[195,196,197],"strong",{},"coding harness"," and an ",[195,200,201],{},"enterprise harness"," is choosing the workspace. A coding harness (Claude Code, Cursor, Codex, open shells) wraps a model for a developer and a repository. An enterprise harness wraps a model for operators and systems of record. Same equation — ",[204,205,209],"a",{"href":206,"rel":207},"https:\u002F\u002Fdocs.langchain.com\u002Foss\u002Fpython\u002Flangchain\u002Fagents",[208],"nofollow","Agent = Model + Harness"," — different loop.",[191,212,213,214,218,219,223,224,228],{},"This is the buying companion to ",[204,215,217],{"href":216},"inner-vs-outer-agent-harness","inner vs outer agent harness",". It sits beside ",[204,220,222],{"href":221},"how-to-choose-between-a-copilot-and-a-work-os","how to choose between a copilot and a work OS",": copilots are personal assistants; coding harnesses are ",[225,226,227],"em",{},"agentic"," inner loops with tools and tests; enterprise harnesses are outer loops with grants and signers. Do not collapse all three into “we need ChatGPT.”",[191,230,231,236,237,242,243,248],{},[204,232,235],{"href":233,"rel":234},"https:\u002F\u002Fmartinfowler.com\u002Farticles\u002Fharness-engineering.html",[208],"Böckeler"," documents how coding-agent users add guides and sensors. ",[204,238,241],{"href":239,"rel":240},"https:\u002F\u002Faddyosmani.com\u002Fblog\u002Fown-the-outer-loop\u002F",[208],"Osmani"," tells engineers to own verify-and-release. ",[204,244,247],{"href":245,"rel":246},"https:\u002F\u002Fwww.thoughtworks.com\u002Finsights\u002Farticles\u002Foperating-system-enterprise-ai",[208],"Thoughtworks"," argues the organisational layer is still the gap. The purchase mistake is using one budget line for all three layers.",[191,250,251,256],{},[204,252,255],{"href":253,"rel":254},"https:\u002F\u002Fwww.mckinsey.com\u002Fcapabilities\u002Fquantumblack\u002Four-insights\u002Fthe-state-of-ai",[208],"McKinsey’s 2025 State of AI"," is the organisational backdrop: usage is easy; scale is redesign. A Cursor rollout can scale pull requests. It will not, by itself, scale governed CRM writes. An OS-class rollout can scale those writes. It will annoy engineers if you force “rewrite this function” through a Critical gate.",[258,259,261],"h2",{"id":260},"words-youll-hear","Words you’ll hear",[263,264,265,293,304,326,336],"ul",{},[266,267,268,271,272,276,277,280,281,286,287,292],"li",{},[195,269,270],{},"Coding \u002F inner harness."," Repo workspace, sandbox, ",[273,274,275],"code",{},"AGENTS.md"," \u002F ",[273,278,279],{},"CLAUDE.md",", hooks, CI. Eval: ",[204,282,285],{"href":283,"rel":284},"https:\u002F\u002Fwww.swebench.com\u002F",[208],"SWE-bench",", ",[204,288,291],{"href":289,"rel":290},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2601.11868",[208],"Terminal-Bench",", your tests.",[266,294,295,298,299,303],{},[195,296,297],{},"Enterprise \u002F outer harness."," Job workspace, connectors, roster, write quotes, ledger. Eval: signed payload vs SoR. ",[204,300,302],{"href":301},"what-is-an-enterprise-agent-harness","What is an enterprise agent harness",".",[266,305,306,309,310,286,315,286,320,325],{},[195,307,308],{},"Copilot."," Personal completion surface. Often no repo loop. ",[204,311,314],{"href":312,"rel":313},"https:\u002F\u002Fopenai.com\u002Fbusiness\u002Fchatgpt-enterprise\u002F",[208],"ChatGPT Enterprise",[204,316,319],{"href":317,"rel":318},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fmicrosoft-365\u002Fcopilot",[208],"Microsoft 365 Copilot",[204,321,324],{"href":322,"rel":323},"https:\u002F\u002Fwww.anthropic.com\u002Fnews\u002Fclaude-for-work",[208],"Claude for Work",". Keep for mail. Do not hand it the NetSuite token.",[266,327,328,331,332,303],{},[195,329,330],{},"Framework."," How you assemble a loop in code. Not a purchase of a company workspace. ",[204,333,335],{"href":334},"agent-harness-vs-agent-framework","Harness vs framework",[266,337,338,341,342,303],{},[195,339,340],{},"MCP."," Plug into either. Dangerous when both share a production write server. ",[204,343,345],{"href":344},"mcp-for-enterprise-integrations","MCP for enterprise",[191,347,348,349,286,352,286,355,358,359,303],{},"Nimbus is an enterprise \u002F outer option: ",[204,350,351],{"href":32},"workstreams",[204,353,354],{"href":20},"teams",[204,356,357],{"href":40},"governance",". Claude Code is a coding \u002F inner option. The rational stack is both, with a hard rule: no unsigned SoR writes from the inner harness. ",[204,360,362],{"href":361},"how-to-solve-unapproved-crm-writes-from-ai","How to solve unapproved CRM writes",[258,364,366],{"id":365},"why-the-choice-is-usually-both","Why the choice is usually “both”",[191,368,369],{},"The tools look similar in a first meeting. Both stream tokens. Both call tools. Both have “agents” on the website. The evaluation is what happens after the answer.",[191,371,372,375,376,381],{},[195,373,374],{},"Buy a coding harness when"," the artefact is code in a repo you already trust with CI: features, refactors, tests, developer docs, infra-as-code that merges through the same gates humans use. ",[204,377,380],{"href":378,"rel":379},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-harnesses-for-long-running-agents",[208],"Anthropic’s long-running harness"," is this world: git, progress files, end-to-end checks.",[191,383,384,387,388,392,393,392,397,402],{},[195,385,386],{},"Buy an enterprise harness when"," the artefact is a change to Salesforce, NetSuite, a policy commitment, or a cross-department decision that must be replayed. ",[204,389,391],{"href":390},"what-is-write-back-governance","Write-back",". ",[204,394,396],{"href":395},"what-is-human-in-the-loop-ai","HITL",[204,398,401],{"href":399,"rel":400},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework",[208],"NIST RMF"," context of use is operations, not a checkout.",[191,404,405,408],{},[195,406,407],{},"Keep a copilot when"," the job is a paragraph in a mailbox. Do not scale it into an approval architecture.",[191,410,411,414],{},[195,412,413],{},"Build on a framework when"," engineers own a unique loop and will maintain grants. That is a programme, not a seat.",[191,416,417,422,423,428],{},[204,418,421],{"href":419,"rel":420},"https:\u002F\u002Fhai.stanford.edu\u002Fai-index\u002F2025-ai-index-report",[208],"Stanford HAI’s 2025 AI Index"," charts the explosion of coding-agent tooling. Procurement that only reads that chart will under-buy the outer layer. Procurement that only reads ",[204,424,427],{"href":425,"rel":426},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F42001",[208],"ISO 42001"," will over-process inner loops and lose developers.",[258,430,432],{"id":431},"decision-tests","Decision tests",[191,434,435,438],{},[195,436,437],{},"1. What is the system of record for the outcome?"," Git: inner. CRM\u002FERP\u002Fcustomer commitment: outer. Both: two harnesses, one write plane (the outer quotes).",[191,440,441,444],{},[195,442,443],{},"2. Who is the signer?"," The author of the PR (inner, plus CODEOWNERS). A named RevOps\u002FFinance\u002FLegal role (outer). If you cannot name the role, you are not ready to buy the outer write path — buy read-only first.",[191,446,447,450,451,303],{},[195,448,449],{},"3. What is the independent sensor?"," Pytest \u002F tsc \u002F CI (inner). Payload schema + SoR read-back (outer). “The model said it was fine” is neither. ",[204,452,454],{"href":453},"eval-loops-for-enterprise-agent-harnesses","Eval loops",[191,456,457,460,461,466],{},[195,458,459],{},"4. What identity should the tools use?"," Developer sandbox and repo token (inner). Workstream-scoped OAuth (outer). A shared MCP god account fails both ",[204,462,465],{"href":463,"rel":464},"https:\u002F\u002Fgenai.owasp.org\u002Fllm-top-10\u002F",[208],"OWASP"," and SoD.",[191,468,469,472,473,475,476,480],{},[195,470,471],{},"5. How will you ratchet failures?"," Inner: ",[273,474,275],{}," + hooks + tests (",[204,477,479],{"href":478},"what-is-harness-engineering","harness engineering","). Outer: wiki revision + gate tier + graph. If your plan is “we’ll prompt better,” you have not chosen a harness. You have chosen hope.",[191,482,483,486,487,392,491,303],{},[195,484,485],{},"6. Time-to-value and staffing."," Cursor can be a week for a team that already has CI. AIP can be a programme. Nimbus-style self-service claims a product week for a standard write — verify with a ",[204,488,490],{"href":489},"how-to-run-an-enterprise-ai-proof-of-value","PoV",[204,492,494],{"href":493},"self-service-vs-forward-deployed-ai-platforms","Self-service vs FDE",[258,496,498],{"id":497},"anti-patterns","Anti-patterns",[191,500,501,504],{},[195,502,503],{},"Cursor for Salesforce."," MCP connected to production. Tests on fixtures. Amount changes. No signer in the ledger. Inner loop on an outer record.",[191,506,507,510],{},[195,508,509],{},"Work OS for a one-line refactor."," Critical gate, three departments. Engineers route around. Outer loop on an inner job.",[191,512,513,516,517,303],{},[195,514,515],{},"One mesh to rule them."," IDE, chatbot, and OS all write through the same server. Two writers. ",[204,518,520],{"href":519},"multi-agent-ai-architecture","Multi-agent architecture",[191,522,523,526,527,303],{},[195,524,525],{},"Benchmark shopping."," Buying Agentforce because of a coding leaderboard, or buying Claude Code because of a governance white paper. Wrong evidence. ",[204,528,530],{"href":529},"how-to-evaluate-an-agent-harness","How to evaluate an agent harness",[191,532,533,536],{},[195,534,535],{},"Banning inner harnesses until the OS ships."," Usually slows software and does not stop paste-into-CRM. Ban the write path; allow the compile path.",[191,538,539,540,543],{},"Nimbus should lose the inner job on purpose. If a vendor tries to replace Claude Code for application engineering, ask for sandbox, hooks, and merge sensors — ",[204,541,542],{"href":529},"evaluate the harness"," — and expect to keep a coding tool anyway. If a coding-tool vendor tries to replace the OS for NetSuite journals, ask for quoted GL lines and a Finance signer.",[258,545,547],{"id":546},"a-simple-portfolio","A simple portfolio",[549,550,551,564],"table",{},[552,553,554],"thead",{},[555,556,557,561],"tr",{},[558,559,560],"th",{},"Job",[558,562,563],{},"Buy",[565,566,567,576,584,592],"tbody",{},[555,568,569,573],{},[570,571,572],"td",{},"Mail, slides, one-off Q&A",[570,574,575],{},"Copilot",[555,577,578,581],{},[570,579,580],{},"Application and infra repos",[570,582,583],{},"Coding harness",[555,585,586,589],{},[570,587,588],{},"Cross-department SoR writes",[570,590,591],{},"Enterprise harness",[555,593,594,597],{},[570,595,596],{},"Unique simulation \u002F exotic tools",[570,598,599],{},"Framework + your grants",[191,601,602,603,606,607,609,610,613],{},"Most enterprises tick all four rows. Budget them separately. Share policy ",[225,604,605],{},"intent"," (discount cap) via wiki and via ",[273,608,275],{}," where relevant; share ",[225,611,612],{},"enforcement"," only on the plane that can execute the write.",[191,615,616,617,619,620,623,624,627],{},"See ",[204,618,11],{"href":12}," for how Nimbus maps to the third row, ",[204,621,622],{"href":45},"models"," for routing, ",[204,625,626],{"href":51},"integrations"," for connectors. See Claude Code \u002F Cursor docs for the second. Do not let a single SOW blur the rows.",[258,629,631],{"id":630},"procurement-sequence-that-does-not-waste-a-quarter","Procurement sequence that does not waste a quarter",[191,633,634,637,638,642],{},[195,635,636],{},"Week 1 — inventory loops, not vendors."," List jobs that already have a finish line. Tag each: git artefact, SoR artefact, mailbox artefact, unique research. You now have four shopping lists. ",[204,639,641],{"href":253,"rel":640},[208],"McKinsey"," programmes that skip this step buy one platform and force every row into it.",[191,644,645,648,649,303],{},[195,646,647],{},"Week 2 — freeze the write rule."," Unsigned SoR writes are impossible from copilots, coding agents, frameworks, and the OS. That rule is cheaper than any bake-off. It also tells Security what to revoke this month (god MCP servers). ",[204,650,651],{"href":361},"Unapproved CRM writes",[191,653,654,657,658,663,664,667],{},[195,655,656],{},"Week 3 — inner bake-off only if you lack a coding harness."," Hooks, sandbox, CI independence, model swap on the same tools. Terminal-Bench and SWE-bench as vendor quality, not as Legal’s control. ",[204,659,662],{"href":660,"rel":661},"https:\u002F\u002Fcode.claude.com\u002Fdocs\u002Fen\u002Fhooks",[208],"Anthropic hooks"," vs Cursor rules vs Codex — pick for ",[225,665,666],{},"your"," repos.",[191,669,670,673,674,677,678,303],{},[195,671,672],{},"Week 4 — outer bake-off only for SoR jobs."," Run the refuse\u002Freplay script from ",[204,675,676],{"href":529},"how to evaluate an agent harness",". Include Nimbus, AIP, Agentforce, or a LangGraph programme as fits the staffing model. ",[204,679,494],{"href":493},[191,681,682,685,686,688],{},[195,683,684],{},"Do not"," hold week 3 until week 4 ships. Engineers will adopt inner tools anyway; you will only lose the chance to standardise hooks. ",[195,687,684],{}," skip week 4 because week 3’s coding agent “can also call Salesforce.” That is the anti-pattern.",[191,690,691],{},"Budget: copilot seats (predictable, personal); coding harness seats or usage (developer count); enterprise harness by work, not by mailbox count if you care about routing. Mixing all three into one “AI budget” is how flagship models burn on classify and how CRM writes go unquoted to save a line item.",[191,693,694,695,699],{},"Thoughtworks’ ",[204,696,698],{"href":245,"rel":697},[208],"organisational harness"," is the steering cadence after purchase: incidents become controls across both inner and outer. Buy tools that allow that ratchet. A coding harness that forbids custom hooks, or an OS that forbids adding a gate without FDE, will stall week 5.",[191,701,702,703,706,707,712,713,716],{},"Expect political arguments that are actually workspace arguments. Engineering will say the OS is slow. They are right for a one-line refactor. RevOps will say Cursor is unsafe. They are right for a production Opportunity. The CISO will say “one approved agent.” Translate: one ",[225,704,705],{},"write rule",", many loops. ",[204,708,711],{"href":709,"rel":710},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2024\u002F1689\u002Foj",[208],"EU AI Act"," oversight can be satisfied per system of use, not per brand. ",[204,714,401],{"href":399,"rel":715},[208]," Map is the same advice.",[191,718,719,720,723],{},"If budget forces a single purchase this half, buy the loop that matches the ",[225,721,722],{},"highest-harm"," unfinished job. Ungoverned CRM writes usually outrank “we could use a better coding agent” — paste already exists; unsigned APIs are new blast radius. If the highest-harm job is shipping software and SoR writes are still human, buy the coding harness and freeze the write rule until the outer product lands. Either way, write the rule down before the PO.",[191,725,726],{},"Nimbus should win the outer row on self-service quoting and graph export, and should lose the inner row on purpose. If a bake-off ranks us against Claude Code on SWE-bench, the scorecard is wrong. If it ranks us against a copilot on mail quality, also wrong. Rank us against AIP and Agentforce on the refuse\u002Freplay script, and against “we’ll build LangGraph” on time-to-first-governed-write.",[191,728,729,730,733,734,737,738,741,742,746],{},"The copilot row still matters. People will keep ",[204,731,314],{"href":312,"rel":732},[208]," for drafts. That is healthy if the write path is the easy official one. Banning unofficial ",[225,735,736],{},"drafts"," usually fails; making unofficial ",[225,739,740],{},"writes"," fail-closed usually works. ",[204,743,745],{"href":744},"what-is-shadow-ai","Shadow AI"," is often a write-path problem wearing a chat-policy costume.",[258,748,750],{"id":749},"questions-people-actually-ask","Questions people actually ask",[752,753,755],"h3",{"id":754},"we-already-paid-for-github-copilot","We already paid for GitHub Copilot.",[191,757,758],{},"That is often a completion copilot, not a full coding harness. You may still want Claude Code or Cursor for agentic repo work. Evaluate hooks and tests, not the seat.",[752,760,762],{"id":761},"can-the-enterprise-harness-include-a-coding-specialist","Can the enterprise harness include a coding specialist?",[191,764,765,766,769,770,303],{},"Yes, as a ",[225,767,768],{},"bounded tool"," that opens a draft PR. The SoR write still quotes in the outer harness. Specialists are hands. ",[204,771,773],{"href":772},"agent-team-architecture","Agent teams",[752,775,777],{"id":776},"what-if-legal-wants-one-vendor","What if Legal wants one vendor?",[191,779,780],{},"One vendor for identity and logging is reasonable. One vendor for repo loop and CRM loop is how you get a mediocre both. Prefer two harnesses and one interceptor rule: unsigned SoR writes are impossible everywhere.",[752,782,784],{"id":783},"how-do-we-score-nimbus-vs-claude-code-in-a-bake-off","How do we score Nimbus vs Claude Code in a bake-off?",[191,786,787],{},"Different jobs. Run inner tests on a repo. Run outer tests on a quoted CRM write. A combined “winner” is a category error unless you only have one job.",[752,789,791],{"id":790},"what-should-i-read-next","What should I read next?",[191,793,794,797,798,800,801,803],{},[204,795,796],{"href":216},"Inner vs outer"," for architecture. ",[204,799,530],{"href":529}," for the live tests. ",[204,802,302],{"href":301}," for the outer object.",[258,805,807],{"id":806},"related-reading","Related reading",[191,809,810,813,814,303],{},[204,811,812],{"href":221},"How to choose between a copilot and a work OS"," and ",[204,815,817],{"href":816},"build-vs-buy-an-enterprise-ai-os","Build vs buy an enterprise AI OS",[258,819,821],{"id":820},"sources","Sources",[263,823,824,830,836,842,848,854,860,866,871,876,882,888,894,900,906,912],{},[266,825,826],{},[204,827,829],{"href":206,"rel":828},[208],"LangChain, Agents",[266,831,832],{},[204,833,835],{"href":233,"rel":834},[208],"Böckeler, Harness engineering for coding agent users",[266,837,838],{},[204,839,841],{"href":239,"rel":840},[208],"Addy Osmani, Own the outer loop",[266,843,844],{},[204,845,847],{"href":245,"rel":846},[208],"Thoughtworks, The operating system for enterprise AI",[266,849,850],{},[204,851,853],{"href":378,"rel":852},[208],"Anthropic, Effective harnesses for long-running agents",[266,855,856],{},[204,857,859],{"href":322,"rel":858},[208],"Anthropic, Claude for Work",[266,861,862],{},[204,863,865],{"href":312,"rel":864},[208],"OpenAI, ChatGPT Enterprise",[266,867,868],{},[204,869,319],{"href":317,"rel":870},[208],[266,872,873],{},[204,874,285],{"href":283,"rel":875},[208],[266,877,878],{},[204,879,881],{"href":289,"rel":880},[208],"Terminal-Bench (arXiv:2601.11868)",[266,883,884],{},[204,885,887],{"href":253,"rel":886},[208],"McKinsey, The state of AI in 2025",[266,889,890],{},[204,891,893],{"href":419,"rel":892},[208],"Stanford HAI, 2025 AI Index",[266,895,896],{},[204,897,899],{"href":399,"rel":898},[208],"NIST AI RMF",[266,901,902],{},[204,903,905],{"href":425,"rel":904},[208],"ISO\u002FIEC 42001",[266,907,908],{},[204,909,911],{"href":463,"rel":910},[208],"OWASP Top 10 for LLM applications",[266,913,914],{},[204,915,918],{"href":916,"rel":917},"https:\u002F\u002Fmodelcontextprotocol.io\u002Fspecification\u002F2025-11-25\u002Findex",[208],"Model Context Protocol specification",{"title":171,"searchDepth":172,"depth":172,"links":920},[921,922,923,924,925,926,927,935,936],{"id":260,"depth":172,"text":261},{"id":365,"depth":172,"text":366},{"id":431,"depth":172,"text":432},{"id":497,"depth":172,"text":498},{"id":546,"depth":172,"text":547},{"id":630,"depth":172,"text":631},{"id":749,"depth":172,"text":750,"children":928},[929,931,932,933,934],{"id":754,"depth":930,"text":755},3,{"id":761,"depth":930,"text":762},{"id":776,"depth":930,"text":777},{"id":783,"depth":930,"text":784},{"id":790,"depth":930,"text":791},{"id":806,"depth":172,"text":807},{"id":820,"depth":172,"text":821},"2026-08-24","A coding harness runs a repository — Claude Code, Cursor, Codex. An enterprise harness runs company jobs with connectors and signers. Most organisations need both; they are not substitutes.","\u002Fblog\u002Fhow-to-choose-between-a-coding-harness-and-an-enterprise-harness",{"title":182,"description":938},"evaluation","blog\u002Fhow-to-choose-between-a-coding-harness-and-an-enterprise-harness",[941,944,945,946],"agent-harness","coding-agents","enterprise-ai","zypj0rFuSxQgNAtRx0gY8CyrewpGlXGHc6zrzz32V4g",{"id":949,"title":950,"archived":165,"authors":951,"badge":953,"body":954,"date":937,"definedTerm":166,"department":166,"description":1603,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1604,"relatedHeading":166,"seo":1605,"series":941,"sitemap":131,"status":166,"stem":1606,"subhead":166,"tags":1607,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1609},"content\u002Fblog\u002Fhow-to-evaluate-an-agent-harness.md","How to Evaluate an Agent Harness",[952],{"name":185,"to":136},{"label":187},{"type":168,"value":955,"toc":1584},[956,964,983,999,1001,1066,1074,1078,1081,1111,1122,1125,1129,1142,1161,1178,1189,1202,1213,1225,1236,1247,1251,1277,1287,1291,1297,1306,1309,1313,1316,1325,1335,1346,1356,1365,1373,1381,1394,1401,1404,1412,1416,1436,1438,1442,1445,1449,1454,1458,1461,1465,1468,1470,1486,1488,1498,1500],[191,957,958,959,963],{},"Evaluating an ",[204,960,962],{"href":961},"what-is-an-agent-harness","agent harness"," is checking whether the runtime around the model can finish a job under a stop you trust — not whether a demo answered a question.",[191,965,966,970,971,975,976,979,980,303],{},[204,967,969],{"href":206,"rel":968},[208],"LangChain"," defines the object: Agent = Model + Harness. The scoring sheet is therefore about the harness. If your RFP starts with context-window size and SWE-bench, you are scoring a model (and maybe an inner coding loop). You will miss whether an unsigned Salesforce PATCH is possible. ",[204,972,974],{"href":973},"how-to-evaluate-an-enterprise-ai-operating-system","How to evaluate an enterprise AI OS"," is the cousin sheet for wiki, workstreams, and routing as a ",[225,977,978],{},"product category",". This page is the runtime tests that apply to Claude Code, a LangGraph deployment, AIP, Agentforce, and Nimbus alike — then specialised by ",[204,981,982],{"href":216},"inner vs outer",[191,984,985,988,989,994,995,998],{},[204,986,255],{"href":253,"rel":987},[208]," already measured the trap: widespread use, limited scale. A fluent demo produces the first. A harness that can refuse, replay, and ratchet produces the second. ",[204,990,993],{"href":991,"rel":992},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\u002Fnist-ai-rmf-playbook",[208],"NIST’s AI RMF Playbook"," is the measurement language. ",[204,996,905],{"href":425,"rel":997},[208]," is the management-system language. Neither is “the model seemed careful.”",[258,1000,261],{"id":260},[263,1002,1003,1012,1027,1038,1046,1056],{},[266,1004,1005,1008,1009,1011],{},[195,1006,1007],{},"Harness vs framework."," Library versus running loop. ",[204,1010,335],{"href":334},". “We use LangChain” is not a passed test.",[266,1013,1014,1017,1018,1021,1022,1026],{},[195,1015,1016],{},"Sensor."," Independent check. ",[204,1019,235],{"href":233,"rel":1020},[208],"; ",[204,1023,247],{"href":1024,"rel":1025},"https:\u002F\u002Fwww.thoughtworks.com\u002Fen-us\u002Finsights\u002Fblog\u002Fgenerative-ai\u002Fharness-engineering-agent-feedback-exploring-ai-coding-sensors",[208],". Inner: tests. Outer: quote vs SoR.",[266,1028,1029,1032,1033,1037],{},[195,1030,1031],{},"Hook \u002F interceptor."," Always runs. ",[204,1034,1036],{"href":660,"rel":1035},[208],"Claude Code hooks",". Outer: fail-closed adapter.",[266,1039,1040,1043,1044,303],{},[195,1041,1042],{},"Quote."," Structured payload, not a paragraph. ",[204,1045,391],{"href":390},[266,1047,1048,1051,1052,303],{},[195,1049,1050],{},"Replay."," Can you reconstruct signer, policy version, tool grants. ",[204,1053,1055],{"href":1054},"what-is-a-lifecycle-graph","Lifecycle graph",[266,1057,1058,1061,1062,303],{},[195,1059,1060],{},"Model portability."," Swap weights without rewriting tools. Not a logo on a slide. ",[204,1063,1065],{"href":1064},"what-is-model-routing","Model routing",[191,1067,1068,1069,813,1071,1073],{},"When you evaluate Nimbus, run these tests on ",[204,1070,351],{"href":32},[204,1072,357],{"href":40},", not on a homepage video. When you evaluate Claude Code, run them on a repo hook and CI, not on a blog SWE-bench screenshot. Same sheet, different workspace.",[258,1075,1077],{"id":1076},"why-evaluation-usually-fails","Why evaluation usually fails",[191,1079,1080],{},"People score agents like they score chat: quality of the paragraph, latency, brand of the model. That produces three false passes:",[1082,1083,1084,1093,1101],"ol",{},[266,1085,1086,1089,1090,303],{},[195,1087,1088],{},"The copilot pass."," SSO, a usage dashboard, a good answer. No loop ownership. ",[204,1091,1092],{"href":221},"Copilot vs work OS",[266,1094,1095,1098,1099,303],{},[195,1096,1097],{},"The benchmark pass."," SWE-bench or Terminal-Bench for an outer job. Inner eval, outer purchase. ",[204,1100,454],{"href":453},[266,1102,1103,1106,1107,1110],{},[195,1104,1105],{},"The framework pass."," A graph in a notebook with every production tool attached. ",[204,1108,465],{"href":463,"rel":1109},[208]," excessive agency with extra nodes.",[191,1112,1113,1118,1119,303],{},[204,1114,1117],{"href":1115,"rel":1116},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents",[208],"Anthropic"," is blunt: encode the job, bound the tools, define done. Your proof of value should force those three. Written answers without a failed action are still a slide. ",[204,1120,1121],{"href":489},"How to run an enterprise AI proof of value",[191,1123,1124],{},"Red flags: chathe entire proof; “we integrate” with no scoped grant; governance as PDF; memory as a long window; “model-agnostic” with a flagship default and seat pricing; MCP write tools that inherit a god service account; vendor database offered as the new system of record.",[258,1126,1128],{"id":1127},"checklist","Checklist",[191,1130,1131,472,1134,1137,1138,303],{},[195,1132,1133],{},"1. Can it stop an action the model wants?",[273,1135,1136],{},"PreToolUse"," denies a matched command; tests fail the merge. Outer: unsigned write does not execute; reject leaves SoR unchanged. If the only stop is max tokens, you have a fuse, not a control plane. ",[204,1139,1141],{"href":1140},"human-in-the-loop-approval-architecture","HITL architecture",[191,1143,1144,1145,1150,1151,1156,1157,1160],{},"Why this matters: ",[204,1146,1149],{"href":1147,"rel":1148},"https:\u002F\u002Fwww.cbc.ca\u002Fnews\u002Fcanada\u002Fbritish-columbia\u002Fair-canada-chatbot-lawsuit-1.7116416",[208],"Air Canada"," and the ",[204,1152,1155],{"href":1153,"rel":1154},"https:\u002F\u002Fwww.reuters.com\u002Flegal\u002Fnew-york-lawyers-sanctioned-using-fake-chatgpt-cases-legal-brief-2023-06-22\u002F",[208],"sanctioned ChatGPT brief"," are ungated generation reaching a record. Your demo must show a ",[225,1158,1159],{},"failed"," write.",[191,1162,1163,1166,1167,1169,1170,1174,1175,1177],{},[195,1164,1165],{},"2. Can you replay who signed and which harness version ran?"," Signer identity, wiki or ",[273,1168,275],{}," revision, tool grants, payload hash, model class. If the answer is Slack search or “the transcript,” you do not have a ledger. ",[204,1171,1173],{"href":1172},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",". Nimbus’s ",[204,1176,23],{"href":24}," is one implementation; demand the export without a vendor engineer.",[191,1179,1180,1183,1184,1188],{},[195,1181,1182],{},"3. Can you swap the model without rewriting tools?"," Change compact vs frontier on extract vs judgement. If tools are bound to one vendor’s function-calling dialect in application code with no adapter, portability is a hope. ",[204,1185,1187],{"href":206,"rel":1186},[208],"LangChain’s model interface"," exists for this; product harnesses must expose it as policy, not as a rewrite.",[191,1190,1191,1194,1195,1199,1200,303],{},[195,1192,1193],{},"4. Are tools grants or a belt?"," Least privilege per job. Missing Salesforce is a configuration error, not a hallucination. ",[204,1196,1198],{"href":1197},"connector-and-permissions-architecture","Connector architecture",". MCP servers inherit the same grant. ",[204,1201,345],{"href":344},[191,1203,1204,1207,1208,1212],{},[195,1205,1206],{},"5. Is verification outside the generator?"," Inner: CI the agent cannot mark skip without a hook. Outer: schema of the quote; SoR row matches. Anthropic’s ",[204,1209,1211],{"href":378,"rel":1210},[208],"long-running harness"," refuses “premature victory” by forcing artefacts and tests. Steal that instinct.",[191,1214,1215,1218,1219,1222,1223,303],{},[195,1216,1217],{},"6. Can an operator add a sensor without a six-month SOW?"," ",[204,1220,1221],{"href":478},"Harness engineering"," is a ratchet. If only vendor FDE can add a gate, you bought a programme. Fine for AIP-scale. Wrong for a standard CRM field this quarter. ",[204,1224,494],{"href":493},[191,1226,1227,1230,1231,1235],{},[195,1228,1229],{},"7. Is the workspace the job you are buying?"," Repo vs company. ",[204,1232,1234],{"href":1233},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose coding vs enterprise",". A single scoring sheet with no workspace column will buy the wrong loop.",[191,1237,1238,1241,1242,1246],{},[195,1239,1240],{},"8. Economics of the loop."," Max steps, spend cap, routing. Seat “unlimited” is often always-flagship. ",[204,1243,1245],{"href":1244},"ai-cost-control-architecture","AI cost control architecture",". Ask for a per-step model breakdown on a live run.",[752,1248,1250],{"id":1249},"rfp-questions","RFP questions",[1082,1252,1253,1256,1259,1262,1265,1268,1271,1274],{},[266,1254,1255],{},"Show an action the model attempted that the harness refused. What fired?",[266,1257,1258],{},"After a successful write (or merge), show the signer, policy version, and payload (or diff) without Slack.",[266,1260,1261],{},"Change the model on extract this week. Which tools broke?",[266,1263,1264],{},"Attach a connector (or repo permission) as an operator, not as SE. Time?",[266,1266,1267],{},"Detach the grant mid-job. Does the write fail closed?",[266,1269,1270],{},"What is the independent sensor for “done”? Who can mark skip?",[266,1272,1273],{},"Two departments, different scopes, one job — or one god toolbox?",[266,1275,1276],{},"Price: seats, tokens, NTUs, or a services quote? What stops flagship on classify?",[191,1278,1279,1280,392,1282,1286],{},"Put these in the RFP, then run them in a ",[204,1281,490],{"href":489},[204,1283,1285],{"href":1284},"rfp-questions-for-enterprise-ai-agents","RFP questions for enterprise AI agents"," overlaps; keep both. Agents without a harness test are a persona list.",[752,1288,1290],{"id":1289},"proof-of-value-short","Proof of value (short)",[191,1292,1293,1296],{},[195,1294,1295],{},"Inner job:"," real repo, required hook, red test the agent must fix, no production SoR token.",[191,1298,1299,1302,1303,1305],{},[195,1300,1301],{},"Outer job:"," real cross-department write, quoted payload, reject path, export. Nimbus should pass the same live sequence as anyone else: OAuth attach, blocked unsigned write, graph export. ",[204,1304,11],{"href":12}," is not the proof.",[191,1307,1308],{},"Skip any refuse\u002Freplay\u002Fswap and you evaluated a chat product, a benchmark, or a framework notebook.",[258,1310,1312],{"id":1311},"score-inner-and-outer-without-mixing-oracles","Score inner and outer without mixing oracles",[191,1314,1315],{},"Run two short scripts. Do not average them into one “AI score.”",[191,1317,1318,1321,1322,1324],{},[195,1319,1320],{},"Inner script (repo)."," Fresh checkout of a service you own. Required hook: deny a dangerous bash pattern. Agent must add a failing test then make it pass. CI is the merge sensor. No production CRM token in the environment. Record: did the hook fire, did CI stay independent, can you show the ",[273,1323,275],{}," revision. SWE-bench plots from the vendor are background, not this script.",[191,1326,1327,1330,1331,1334],{},[195,1328,1329],{},"Outer script (SoR)."," Sandbox Salesforce or equivalent. Operator (not SE) attaches OAuth. Model proposes a write. Unsigned path must fail. Reject path must leave records unchanged. Approve path: read-back matches hash. Export signer and wiki revision. Detach the connector and retry the write — must fail closed. ",[204,1332,1333],{"href":489},"Proof of value"," is this script with two departments on the canvas.",[191,1336,1337,1338,813,1340,1342,1343,303],{},"If a vendor refuses to run the outer script because “we are a coding tool,” believe them and buy them for inner only. If a vendor refuses the inner script because “we are an OS,” believe them and do not replace Cursor. If a vendor claims both and fails one script, you have a category error in their marketing. Nimbus should pass the outer script on ",[204,1339,351],{"href":32},[204,1341,357],{"href":40},". Claude Code should pass the inner script. ",[204,1344,1345],{"href":1233},"How to choose",[191,1347,1348,1351,1352,1355],{},[195,1349,1350],{},"Thoughtworks’ layer check."," After the scripts, ask where layer 4 lives: who owns the policy when the agent did what it was allowed to do and harm still happened. If the answer is a steering committee with no interceptor, you evaluated theatre. ",[204,1353,427],{"href":425,"rel":1354},[208]," will not save a missing refuse.",[191,1357,1358,1361,1362,303],{},[195,1359,1360],{},"Economics check."," Pull one live run’s step list: model class per step, tokens or NTUs, which sensor fired. Always-flagship with no cap is a failed harness eval even if the paragraph was good. ",[204,1363,1364],{"href":1244},"Cost control",[191,1366,1367,1370,1371,303],{},[195,1368,1369],{},"MCP check."," One write-capable server. Which workspaces may use it. If the answer is “any host that can see the URL,” fail. ",[204,1372,345],{"href":344},[191,1374,1375,1376,1380],{},"Weight the eight checklist items; do not add a ninth called “brand.” ",[204,1377,1379],{"href":419,"rel":1378},[208],"Stanford AI Index"," is useful context for how fast coding tools moved. It is not a substitute for the outer script.",[191,1382,1383,1384,1389,1390,1393],{},"Score vendorsystems, not as essays. A beautiful ",[204,1385,1388],{"href":1386,"rel":1387},"https:\u002F\u002Fwww.langchain.com\u002Fblog\u002Fthe-anatomy-of-an-agent-harness",[208],"anatomy post"," does not pass the refuse test. A messy UI that blocks the unsigned PATCH does. Watch for “evaluation theatre”: the SE runs the happy path, the fail path is “we’ll configure that in phase two,” the ledger is a screenshot of LangSmith. Phase two is where ",[204,1391,641],{"href":253,"rel":1392},[208]," pilots go to die.",[191,1395,1396,1397,1400],{},"Bring your own oracle. For inner: a test the agent did not write. For outer: a sandbox row you control. If the vendor must supply the only success criterion, you are scoring their demo fixtures. Terminal-Bench’s strength is that the ",[225,1398,1399],{},"environment"," is the grader. Copy that.",[191,1402,1403],{},"People on the bake-off: an operator who will live in the product, someone who owns the SoR, someone who can say no for Legal, an engineer who will keep the inner harness. If only the vendor and an innovation lead attend, you will buy a narrative. Nimbus, AIP, Cursor, and a LangGraph SOW should all survive that room or be narrowed to the job they actually do.",[191,1405,1406,1407,1411],{},"Write the pass\u002Ffail before the demo so the SE cannot redefine success live. “Blocked unsigned write” is a boolean. “Felt enterprise-ready” is not. Record the session. If they cannot fail on camera, assume they cannot fail in production. ",[204,1408,1410],{"href":991,"rel":1409},[208],"NIST Playbook"," language helps here: you are Measuring a control, not a vibe.",[258,1413,1415],{"id":1414},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[191,1417,1418,1419,1422,1423,1426,1427,1429,1430,1432,1433,1435],{},"Nimbus is an ",[204,1420,1421],{"href":301},"enterprise \u002F outer harness",": ",[204,1424,1425],{"href":28},"wiki"," as guides, connectors as grants, ",[204,1428,354],{"href":20}," as the hiring object, ",[204,1431,357],{"href":40}," as the interceptor, graph as replay, ",[204,1434,622],{"href":45}," as routing. Score those surfaces against the eight tests. Do not accept “we are a harness” as a substitute for a failed write. AIP and Agentforce deserve the same eight.",[258,1437,750],{"id":749},[752,1439,1441],{"id":1440},"can-we-score-claude-code-and-nimbus-on-one-spreadsheet","Can we score Claude Code and Nimbus on one spreadsheet?",[191,1443,1444],{},"Yes, with a workspace column. Shared rows: refuse, replay, swap, sensors, operator change, economics. Inner-only rows: tests, sandbox, PR. Outer-only rows: SoR quote, roster signer, workstream isolation.",[752,1446,1448],{"id":1447},"the-vendor-sent-a-swe-bench-plot","The vendor sent a SWE-bench plot.",[191,1450,1451,1452,303],{},"File it under inner quality. If you are buying CRM writes, it is not sufficient. ",[204,1453,454],{"href":453},[752,1455,1457],{"id":1456},"we-already-completed-a-copilot-rfp","We already completed a copilot RFP.",[191,1459,1460],{},"Keep it for personal tools. This sheet is for loops that act. Different job.",[752,1462,1464],{"id":1463},"is-iso-42001-certification-the-eval","Is ISO 42001 certification the eval?",[191,1466,1467],{},"It is a management-system signal. Still watch a write fail. Certification without an interceptor is paperwork.",[752,1469,791],{"id":790},[191,1471,1472,1476,1477,1481,1482,1485],{},[204,1473,1475],{"href":1474},"agent-harness-architecture","Agent harness architecture"," to know the parts. ",[204,1478,1480],{"href":1479},"how-to-evaluate-write-back-governance","How to evaluate write-back governance"," for the outer stop in detail. ",[204,1483,1484],{"href":478},"What is harness engineering"," for the ratchet after you buy.",[258,1487,807],{"id":806},[191,1489,1490,813,1494,303],{},[204,1491,1493],{"href":1492},"how-to-evaluate-multi-agent-platforms","How to evaluate multi-agent platforms",[204,1495,1497],{"href":1496},"how-to-evaluate-ai-governance-platforms","How to evaluate AI governance platforms",[258,1499,821],{"id":820},[263,1501,1502,1507,1513,1518,1524,1530,1535,1541,1546,1552,1557,1562,1568,1574,1579],{},[266,1503,1504],{},[204,1505,829],{"href":206,"rel":1506},[208],[266,1508,1509],{},[204,1510,1512],{"href":1386,"rel":1511},[208],"LangChain, The anatomy of an agent harness",[266,1514,1515],{},[204,1516,835],{"href":233,"rel":1517},[208],[266,1519,1520],{},[204,1521,1523],{"href":1024,"rel":1522},[208],"Thoughtworks, Harness engineering and agent feedback",[266,1525,1526],{},[204,1527,1529],{"href":1115,"rel":1528},[208],"Anthropic, Building effective agents",[266,1531,1532],{},[204,1533,853],{"href":378,"rel":1534},[208],[266,1536,1537],{},[204,1538,1540],{"href":660,"rel":1539},[208],"Claude Code, Hooks",[266,1542,1543],{},[204,1544,887],{"href":253,"rel":1545},[208],[266,1547,1548],{},[204,1549,1551],{"href":991,"rel":1550},[208],"NIST AI RMF Playbook",[266,1553,1554],{},[204,1555,905],{"href":425,"rel":1556},[208],[266,1558,1559],{},[204,1560,911],{"href":463,"rel":1561},[208],[266,1563,1564],{},[204,1565,1567],{"href":1147,"rel":1566},[208],"CBC, Air Canada chatbot lawsuit",[266,1569,1570],{},[204,1571,1573],{"href":1153,"rel":1572},[208],"Reuters, ChatGPT legal brief sanctions",[266,1575,1576],{},[204,1577,893],{"href":419,"rel":1578},[208],[266,1580,1581],{},[204,1582,918],{"href":916,"rel":1583},[208],{"title":171,"searchDepth":172,"depth":172,"links":1585},[1586,1587,1588,1592,1593,1594,1601,1602],{"id":260,"depth":172,"text":261},{"id":1076,"depth":172,"text":1077},{"id":1127,"depth":172,"text":1128,"children":1589},[1590,1591],{"id":1249,"depth":930,"text":1250},{"id":1289,"depth":930,"text":1290},{"id":1311,"depth":172,"text":1312},{"id":1414,"depth":172,"text":1415},{"id":749,"depth":172,"text":750,"children":1595},[1596,1597,1598,1599,1600],{"id":1440,"depth":930,"text":1441},{"id":1447,"depth":930,"text":1448},{"id":1456,"depth":930,"text":1457},{"id":1463,"depth":930,"text":1464},{"id":790,"depth":930,"text":791},{"id":806,"depth":172,"text":807},{"id":820,"depth":172,"text":821},"Evaluating an agent harness means checking whether it can stop a write, replay who signed, swap the model without rewriting tools, and fail a real sensor — not whether the demo answered a question.","\u002Fblog\u002Fhow-to-evaluate-an-agent-harness",{"title":950,"description":1603},"blog\u002Fhow-to-evaluate-an-agent-harness",[941,944,357,1608],"rfp","1CUE6bav0yO3UQKgSvj4IpjPNqJHOWrGl0WKWVQII7k",{"id":1611,"title":1612,"archived":165,"authors":1613,"badge":1615,"body":1616,"date":2411,"definedTerm":166,"department":166,"description":2412,"extension":174,"eyebrow":166,"faqHeader":2413,"faqs":2416,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":2429,"relatedHeading":166,"seo":2430,"series":941,"sitemap":131,"status":166,"stem":2431,"subhead":166,"tags":2432,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":2436},"content\u002Fblog\u002Fhow-to-evaluate-collaborative-ai.md","How to Evaluate Collaborative AI",[1614],{"name":185,"to":136},{"label":187},{"type":168,"value":1617,"toc":2393},[1618,1630,1651,1660,1678,1682,1689,1715,1718,1745,1757,1768,1779,1788,1792,1795,1799,1805,1811,1817,1823,1827,1832,1837,1842,1849,1858,1862,1875,1880,1889,1902,1920,1924,1929,1934,1939,1955,1959,1964,1969,1974,1986,1992,1996,2008,2013,2023,2030,2034,2128,2136,2140,2143,2148,2166,2171,2188,2191,2215,2219,2224,2235,2238,2256,2259,2269,2276,2280,2283,2309,2312,2315,2319,2322,2333,2338,2341,2345,2352,2362,2365,2367,2379,2386,2389],[191,1619,1620,1621,1624,1625,1629],{},"Evaluating collaborative AI means scoring whether several people can share ",[195,1622,1623],{},"one job"," — same files, same history, a named person who can stop a change — and whether that job still exists after the session ends. McKinsey’s ",[204,1626,1628],{"href":253,"rel":1627},[208],"State of AI"," (2025) found that 88% of organisations use AI in at least one function while most remain in experiment or pilot. A common pilot is one person and one assistant. The next purchase mistake is rebranding that pilot “collaborative” because five people share a login. This checklist is written to catch that false pass early, on any vendor.",[191,1631,1632,1633,1637,1638,1642,1643,1646,1647,1650],{},"It is not the same as evaluating a copilot, an agent swarm, or a shared ChatGPT account. ",[204,1634,1636],{"href":1635},"what-is-collaborative-ai","What is collaborative AI"," is the definition. ",[204,1639,1641],{"href":1640},"collaborative-ai-and-personal-assistants","Collaborative AI and personal assistants"," is ",[195,1644,1645],{},"when"," to use a copilot versus a shared room — read that first if the purchase question is still “do we need both?” This page is the ",[195,1648,1649],{},"RFP sheet"," for the shared job: questions to put in procurement, demos, and proof-of-value scripts.",[191,1652,1653,1655,1656,1659],{},[204,1654,530],{"href":529}," is the runtime sheet — refused writes, replay, model swap. This page is the ",[195,1657,1658],{},"job"," sheet — roster, rejection, handover. Run both when the work crosses departments and may write to a live system. Do not run only the one the vendor prefers.",[191,1661,1662,1665,1666,1669,1670,1673,1674,1677],{},[204,1663,993],{"href":991,"rel":1664},[208]," is the measurement language. Translate it to demos: ",[195,1667,1668],{},"Map"," the job, ",[195,1671,1672],{},"Measure"," reopen time and signer completeness, ",[195,1675,1676],{},"Manage"," with stored rejections before write tokens. “The model seemed careful” is not a measure.",[258,1679,1681],{"id":1680},"what-you-are-actually-buying","What you are actually buying",[191,1683,1684,1685,1688],{},"You are buying a ",[195,1686,1687],{},"container for shared work",", not a better paragraph generator.",[1690,1691,1695],"pre",{"className":1692,"code":1693,"language":1694,"meta":171,"style":171},"language-mermaid shiki shiki-themes github-light github-dark","flowchart LR\n  joinTest[\"Can a second team join?\"] --> stopTest[\"Can someone say no?\"]\n  stopTest --> mondayTest[\"Does Monday still have the files?\"]\n","mermaid",[273,1696,1697,1705,1710],{"__ignoreMap":171},[1698,1699,1702],"span",{"class":1700,"line":1701},"line",1,[1698,1703,1704],{},"flowchart LR\n",[1698,1706,1707],{"class":1700,"line":172},[1698,1708,1709],{},"  joinTest[\"Can a second team join?\"] --> stopTest[\"Can someone say no?\"]\n",[1698,1711,1712],{"class":1700,"line":930},[1698,1713,1714],{},"  stopTest --> mondayTest[\"Does Monday still have the files?\"]\n",[191,1716,1717],{},"Minimum properties:",[263,1719,1720,1727,1733,1739],{},[266,1721,1722,1723,1726],{},"A ",[195,1724,1725],{},"named job",", not “the Slack channel.”",[266,1728,1729,1732],{},[195,1730,1731],{},"Shared context"," — files and history visible to the roster.",[266,1734,1735,1738],{},[195,1736,1737],{},"Tools with recorded steps"," — read, propose, sometimes write with a payload.",[266,1740,1722,1741,1744],{},[195,1742,1743],{},"named stop"," — someone other than the prompter can refuse a change.",[191,1746,1747,1751,1752,1756],{},[204,1748,1750],{"href":1749},"multiplayer-ai-and-multi-agent-ai","Multiplayer AI vs multi-agent AI"," separates people in the room from models in a loop. You can fail collaborative eval while passing multi-agent demos: agents pass tickets to each other; finance still cannot see the brief. ",[204,1753,1755],{"href":1754},"what-is-multi-agent-ai","What is multi-agent AI"," is the cast. This sheet scores the stage.",[191,1758,1759,1763,1764,1767],{},[204,1760,1762],{"href":1761},"four-pillars-of-an-enterprise-ai-platform","Four pillars of an enterprise AI platform"," situates workstreams and governance in a full stack. This sheet scores whether the product you are viewing implements the ",[195,1765,1766],{},"collaborative"," pillar for one real job — Salesforce sidebar, Microsoft copilot, Slack bot, specialist agent platform, or anything else on the shortlist.",[191,1769,1770,1774,1775,1778],{},[204,1771,1773],{"href":1115,"rel":1772},[208],"Anthropic’s guidance on building effective agents"," says encode the job, bound the tools, define done. Your proof of value should force those three ",[195,1776,1777],{},"in a room with two departments",", not in a solo sandbox.",[191,1780,1781,1782,1787],{},"Yang and colleagues, in ",[204,1783,1786],{"href":1784,"rel":1785},"https:\u002F\u002Fwww.nature.com\u002Farticles\u002Fs41562-021-01196-4",[208],"Nature Human Behaviour"," (2022), showed that firm-wide remote work made collaboration networks more static and siloed, with fewer bridges between groups. Shared jobs already fight that pull. A vendor that adds a fluent assistant to each silo will make the silo more confident, not more shared. Score the bridge.",[258,1789,1791],{"id":1790},"rfp-questions-shared-job-vs-shared-login","RFP questions: shared job vs shared login",[191,1793,1794],{},"Put these verbatim in the RFP. Require live answers, not slides. If professional services will “configure that later,” note the time-to-value and price the configuration as part of the buy.",[752,1796,1798],{"id":1797},"_1-is-this-a-shared-job-or-a-shared-login","1. Is this a shared job or a shared login?",[191,1800,1801,1804],{},[195,1802,1803],{},"Pass:"," One work object with its own roster, files, and audit — independent of which user opened the UI today.",[191,1806,1807,1810],{},[195,1808,1809],{},"Fail:"," Five people in one chatbot account, or five parallel threads that cannot see each other’s attachments.",[191,1812,1813,1816],{},[195,1814,1815],{},"Proof:"," Show two users on the same job ID. Remove one user’s access. The job remains for the roster.",[191,1818,1819,1820,1822],{},"Why vendors fail this: shared seats are cheap to demo and expensive to govern. ",[204,1821,1641],{"href":1640}," explains why a shared login is still a personal-assistant shape. Ask for the job ID in the URL or export. If the vendor cannot point at an object, they pointed at a session.",[752,1824,1826],{"id":1825},"_2-can-a-second-department-join-mid-run","2. Can a second department join mid-run?",[191,1828,1829,1831],{},[195,1830,1803],{}," Finance joins Thursday’s sales exception without a re-upload parade. They see the same CRM excerpt, the same draft payload, the same history.",[191,1833,1834,1836],{},[195,1835,1809],{}," “Export and email the transcript.” “Start a new session and paste context.”",[191,1838,1839,1841],{},[195,1840,1815],{}," Add a finance delegate mid-proof. They reject a proposal while sales watches. No side channel required.",[191,1843,1844,1845,1848],{},"This is the core ",[204,1846,1847],{"href":1749},"multiplayer AI"," test dressed for procurement. Do not accept a pre-seeded “war room” that was built overnight by the vendor’s solutions team unless you can repeat the join on a job your people created.",[191,1850,1851,1852,1857],{},"Microsoft and LinkedIn’s ",[204,1853,1856],{"href":1854,"rel":1855},"https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fworklab\u002Fwork-trend-index\u002Fai-at-work-is-here-now-comes-the-hard-part",[208],"2024 Work Trend Index"," found that 78% of AI users bring their own tools. Mid-run join is how you find out whether the product can absorb that habit or whether finance will open a second private window.",[752,1859,1861],{"id":1860},"_3-is-there-a-named-stop-not-human-review-in-the-abstract","3. Is there a named stop — not “human review” in the abstract?",[191,1863,1864,1866,1867,1870,1871,1874],{},[195,1865,1803],{}," A person on ",[195,1868,1869],{},"this roster"," can halt ",[195,1872,1873],{},"this class of write"," while others on the job see the payload.",[191,1876,1877,1879],{},[195,1878,1809],{}," A generic approval workflow outside the job, or a prompt that says “ask manager.”",[191,1881,1882,1884,1885,1888],{},[195,1883,1815],{}," Attempt a live-system change. Show the signer field tied to a human identity. Show a ",[195,1886,1887],{},"stored rejection"," with name and timestamp on the job.",[191,1890,1891,813,1894,1897,1898,1901],{},[204,1892,1893],{"href":395},"What is human-in-the-loop AI",[204,1895,1896],{"href":390},"write-back governance"," define the stop. This question tests whether they are ",[195,1899,1900],{},"on the job",". A ServiceNow ticket opened after the write is not a stop. A Slack reaction is not a signer.",[191,1903,1904,1908,1909,1912,1913,1916,1917,1919],{},[204,1905,1907],{"href":463,"rel":1906},[208],"OWASP’s LLM Top 10"," lists excessive agency and insecure output handling. Collaborative eval adds organisational agency: ",[195,1910,1911],{},"who"," could have stopped ",[195,1914,1915],{},"this"," change on ",[195,1918,1915],{}," job. If the only stop is max tokens, you have a fuse, not a control plane.",[752,1921,1923],{"id":1922},"_4-are-files-in-one-place-not-five-inboxes","4. Are files in one place, not five inboxes?",[191,1925,1926,1928],{},[195,1927,1803],{}," Attachments live on the job — WMS snapshot, ageing extract, partner PO — visible to the roster without re-forwarding.",[191,1930,1931,1933],{},[195,1932,1809],{}," “Paste into the chat window.” “The model will fetch from SharePoint if you paste the link.”",[191,1935,1936,1938],{},[195,1937,1815],{}," List attachments on the job object. Remove the original uploader from the roster. Files remain.",[191,1940,1941,1945,1946,1950,1951,1954],{},[204,1942,1944],{"href":1943},"search-is-not-memory","Search is not memory"," is why “we’ll find it in Slack later” fails eval. Do not let the vendor substitute a retrieval demo for an attachment that survives user removal. ",[204,1947,1949],{"href":1948},"what-is-institutional-memory-in-enterprise-ai","Institutional memory in enterprise AI"," is the company-scale layering; this question only asks whether ",[195,1952,1953],{},"this job"," still has its two files on Monday.",[752,1956,1958],{"id":1957},"_5-what-happens-after-the-session","5. What happens after the session?",[191,1960,1961,1963],{},[195,1962,1803],{}," Monday reopen shows brief, files, last proposal, last rejection or signature — even if the original prompter is out and the model vendor changed.",[191,1965,1966,1968],{},[195,1967,1809],{}," Session expiry deletes context. “Memory” is the user’s personal thread.",[191,1970,1971,1973],{},[195,1972,1815],{}," Close the browser. Reopen with a different user. Continue the job without reconstruction.",[191,1975,1976,1980,1981,1985],{},[204,1977,1979],{"href":1978},"agents-should-be-disposable","Agents should be disposable"," is the design claim this question tests. ",[204,1982,1984],{"href":1983},"what-is-an-ai-workstream","What an AI workstream is"," is the container name. If continuity requires the original model or the original person, you scored a session, not a job.",[191,1987,1988,1991],{},[204,1989,905],{"href":425,"rel":1990},[208]," wants records and named actors. A session that evaporates is not a record. Ask for an export a later reader can use without vendor professional services.",[752,1993,1995],{"id":1994},"_6-does-governance-live-in-the-room","6. Does governance live in the room?",[191,1997,1998,2000,2001,2003,2004,303],{},[195,1999,1803],{}," Roster, inherited authority, spend-cap pause, and notify-to-roster on ",[195,2002,1953],{}," — see ",[204,2005,2007],{"href":2006},"governance-as-a-multiplayer-primitive","governance as a multiplayer primitive",[191,2009,2010,2012],{},[195,2011,1809],{}," Governance PDF emailed quarterly while writes succeed unsigned.",[191,2014,2015,2017,2018,2022],{},[195,2016,1815],{}," Show spend-cap alert to roster members only. Show agent grants bounded to the acting person’s authority. ",[204,2019,2021],{"href":2020},"rbac-for-enterprise-ai","RBAC for enterprise AI"," is the access vocabulary; demand it scoped to the job.",[191,2024,2025,2026,2029],{},"A tenant-wide “AI policy acknowledged” checkbox is not this test. Neither is an SOC 2 report. Those are programme artefacts. This question is whether finance sees the same payload ops sees ",[195,2027,2028],{},"before"," execute.",[258,2031,2033],{"id":2032},"scorecard-false-passes-to-reject","Scorecard: false passes to reject",[549,2035,2036,2049],{},[552,2037,2038],{},[555,2039,2040,2043,2046],{},[558,2041,2042],{},"Demo looks like",[558,2044,2045],{},"Likely false pass",[558,2047,2048],{},"Ask instead",[565,2050,2051,2062,2073,2084,2095,2106,2117],{},[555,2052,2053,2056,2059],{},[570,2054,2055],{},"Fluent multi-user chat",[570,2057,2058],{},"Shared login",[570,2060,2061],{},"Job ID, roster, survive user removal",[555,2063,2064,2067,2070],{},[570,2065,2066],{},"Agent orchestra",[570,2068,2069],{},"Multi-agent without multiplayer",[570,2071,2072],{},"Second department join + named stop",[555,2074,2075,2078,2081],{},[570,2076,2077],{},"Copilot in CRM sidebar",[570,2079,2080],{},"Personal assistant",[570,2082,2083],{},"Cross-team exception with finance on job",[555,2085,2086,2089,2092],{},[570,2087,2088],{},"“We integrate Salesforce”",[570,2090,2091],{},"Connector without payload quote",[570,2093,2094],{},"Show field-level payload before write",[555,2096,2097,2100,2103],{},[570,2098,2099],{},"Long context window",[570,2101,2102],{},"Memory",[570,2104,2105],{},"Reopen Monday without original thread",[555,2107,2108,2111,2114],{},[570,2109,2110],{},"“Human review” node",[570,2112,2113],{},"Abstract HITL",[570,2115,2116],{},"Named signer + stored rejection on job",[555,2118,2119,2122,2125],{},[570,2120,2121],{},"Quarterly attestation",[570,2123,2124],{},"PDF after the write",[570,2126,2127],{},"Fail-closed unsigned attempt",[191,2129,2130,2131,2135],{},"NIST’s ",[204,2132,2134],{"href":399,"rel":2133},[208],"AI Risk Management Framework"," Measure function assumes you can observe outcomes. A demo that cannot produce a stored rejection has nothing to measure except fluency.",[258,2137,2139],{"id":2138},"proof-of-value-script-two-weeks","Proof-of-value script (two weeks)",[191,2141,2142],{},"Do not let the vendor script a happy-path email draft. Use one exception you already run.",[191,2144,2145],{},[195,2146,2147],{},"Week one — read-only, multiplayer:",[1082,2149,2150,2153,2156,2159],{},[266,2151,2152],{},"Name one real exception: credit hold, discount outside grid, order hold in WMS.",[266,2154,2155],{},"Put ops and finance on the roster. Attach two files they already email.",[266,2157,2158],{},"Mid-week, add a second-department guest with a scoped view.",[266,2160,2161,2162,2165],{},"Model drafts release; finance ",[195,2163,2164],{},"rejects","; rejection must stay on the job.",[191,2167,2168],{},[195,2169,2170],{},"Week two — continuity and optional write:",[1082,2172,2174,2177,2180,2185],{"start":2173},5,[266,2175,2176],{},"Swap the prompter. Reopen. A stranger continues without Slack archaeology.",[266,2178,2179],{},"Swap model tier or vendor if the product claims portability.",[266,2181,2182,2183,303],{},"If write is in scope: enable fail-closed write with payload quote; show unsigned attempt ",[195,2184,1159],{},[266,2186,2187],{},"Independent reader reconstructs signer and payload without authors in the room.",[191,2189,2190],{},"If steps 4 or 5 fail, you do not have collaborative AI — you have a group chat with AI autocomplete. Stop the proof. Do not “save write for phase two” as a way to skip the rejection test. The rejection is the point.",[191,2192,2193,2194,286,2198,286,2202,286,2206,286,2210,2214],{},"Function-specific walkthroughs if you need a scenario library: ",[204,2195,2197],{"href":2196},"collaborative-ai-for-operations","operations",[204,2199,2201],{"href":2200},"collaborative-ai-for-revenue-operations","revenue operations",[204,2203,2205],{"href":2204},"collaborative-ai-for-customer-support","customer support",[204,2207,2209],{"href":2208},"collaborative-ai-for-human-resources","human resources",[204,2211,2213],{"href":2212},"collaborative-ai-for-finance-and-planning","finance and planning",". Use one. Do not run six proofs.",[258,2216,2218],{"id":2217},"how-this-differs-from-harness-evaluation","How this differs from harness evaluation",[191,2220,2221,2223],{},[204,2222,530],{"href":529}," asks:",[263,2225,2226,2229,2232],{},[266,2227,2228],{},"Can the harness refuse a write?",[266,2230,2231],{},"Can you replay signer and policy version?",[266,2233,2234],{},"Can you swap the model without rewriting tools?",[191,2236,2237],{},"Collaborative eval asks:",[263,2239,2240,2247,2250],{},[266,2241,2242,2243,2246],{},"Can two departments ",[195,2244,2245],{},"see"," the same refusal?",[266,2248,2249],{},"Does the job survive people and agents leaving?",[266,2251,2252,2253,2255],{},"Are files and stops on the ",[195,2254,1658],{},", not in personal threads?",[191,2257,2258],{},"You need both when the purchase is “AI for cross-team exceptions that may touch CRM, ERP, or WMS.” Harness without collaborative passes produces a gated write finance never saw coming. Collaborative without harness passes produces a shared room where unsigned writes still slip through.",[191,2260,2261,2264,2265,2268],{},[204,2262,2263],{"href":334},"Agent harness vs agent framework"," is the build-versus-compose warning: a graph in a notebook is not a passed test. ",[204,2266,2267],{"href":216},"Inner vs outer agent harness"," is why a coding-harness scorecard will mis-score an outer operations job. Use the right sibling sheet. Do not grade a warehouse hold with SWE-bench.",[191,2270,2271,2275],{},[204,2272,2274],{"href":2273},"what-is-ai-governance","What is AI governance"," is the programme within which both sheets fit. Collaborative AI is not a feature checkbox. It is whether shared work survives the people and models that staffed it this week.",[258,2277,2279],{"id":2278},"vendor-questions-to-copy-into-procurement","Vendor questions to copy into procurement",[191,2281,2282],{},"Number them. Require a live show, a recording, or a written fail.",[1082,2284,2285,2288,2291,2294,2297,2300,2303,2306],{},[266,2286,2287],{},"Show one job ID with two departments and different tool grants on the same roster.",[266,2289,2290],{},"Show a rejection stored on the job with signer identity — not an email log.",[266,2292,2293],{},"Remove the original uploader; attachments remain.",[266,2295,2296],{},"Add a guest; guest cannot inherit write token.",[266,2298,2299],{},"Pause on spend cap; notify roster members; show delegate approval on the job.",[266,2301,2302],{},"Reopen after 72 hours with a different model; show continuity.",[266,2304,2305],{},"Attempt unsigned write to a system of record; show fail-closed.",[266,2307,2308],{},"Export replay — signer, payload, policy version — without vendor professional services.",[191,2310,2311],{},"If the vendor answers questions 1–6 with “our SI will configure that,” treat configuration time as part of the price. A product that needs six months of graph work before a second department can join is a framework purchase, not a collaborative-AI purchase. Say so in the scoring notes.",[191,2313,2314],{},"Ask incumbents the same questions you ask specialists. A CRM copilot that cannot put finance on the job fails this sheet even if it drafts beautiful emails. A multi-agent platform that cannot store a rejection fails this sheet even if the orchestra is elegant.",[258,2316,2318],{"id":2317},"when-collaborative-eval-is-not-the-first-buy","When collaborative eval is not the first buy",[191,2320,2321],{},"Stay with personal assistants when:",[263,2323,2324,2327,2330],{},[266,2325,2326],{},"One owner drafts and nothing writes to a live system.",[266,2328,2329],{},"No second department must stand on the result this quarter.",[266,2331,2332],{},"The pain is blank-page speed, not lost outcomes.",[191,2334,2335,2337],{},[204,2336,1641],{"href":1640}," is the decision tree. Buy collaborative when the recurring meeting already exists and the pain is “we never keep the outcome.”",[191,2339,2340],{},"Do not force a roster onto solo research. Theatre rosters teach people that governance is ceremony. Save the sheet for jobs that already have a fight in chat.",[258,2342,2344],{"id":2343},"how-to-run-the-bake-off","How to run the bake-off",[191,2346,2347,2348,2351],{},"Score every shortlisted product on a ",[195,2349,2350],{},"cross-team"," job, not the AE’s private email draft. Use the same exception, the same two files, the same two departments. Keep a shared scorecard with pass\u002Ffail per question above — not a 1–5 “wow” rating on fluency.",[191,2353,2354,2355,813,2357,2361],{},"Include at least one incumbent copilot and at least one agent platform if both are on the table. The point of the sheet is comparison, not a single-vendor script. ",[204,2356,302],{"href":301},[204,2358,2360],{"href":2359},"what-is-an-enterprise-ai-operating-system","what is an enterprise AI operating system"," are category pages if you need language for the stack around the job. This sheet still scores the job.",[191,2363,2364],{},"Put this sheet in the RFP, then run it in a proof of value. A fluent demo without a stored rejection is still a slide.",[258,2366,1415],{"id":1414},[191,2368,2369,2370,2373,2374,813,2376,2378],{},"When you include Nimbus in a bake-off, run ",[195,2371,2372],{},"this same sheet"," on ",[204,2375,351],{"href":32},[204,2377,357],{"href":40},". Do not substitute a homepage video or a pre-built demo room.",[191,2380,2381,2382,303],{},"Ask for the stored rejection, the mid-run finance join, the reopen after a model swap, and the unsigned write that fails. Score incumbents on the same exception. Staffing and time-to-value for the bake-off itself are a separate frame — see ",[204,2383,2385],{"href":2384},"\u002Fevaluate\u002F","self-service vs forward-deployed",[191,2387,2388],{},"If Nimbus cannot show the “no” on the job Monday, it fails this sheet the way any other vendor would. Collaborative eval is not a product tour.",[2390,2391,2392],"style",{},"html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":171,"searchDepth":172,"depth":172,"links":2394},[2395,2396,2404,2405,2406,2407,2408,2409,2410],{"id":1680,"depth":172,"text":1681},{"id":1790,"depth":172,"text":1791,"children":2397},[2398,2399,2400,2401,2402,2403],{"id":1797,"depth":930,"text":1798},{"id":1825,"depth":930,"text":1826},{"id":1860,"depth":930,"text":1861},{"id":1922,"depth":930,"text":1923},{"id":1957,"depth":930,"text":1958},{"id":1994,"depth":930,"text":1995},{"id":2032,"depth":172,"text":2033},{"id":2138,"depth":172,"text":2139},{"id":2217,"depth":172,"text":2218},{"id":2278,"depth":172,"text":2279},{"id":2317,"depth":172,"text":2318},{"id":2343,"depth":172,"text":2344},{"id":1414,"depth":172,"text":1415},"2026-09-12","An RFP sheet for shared jobs — not copilots. Questions on roster, stored rejection, second department, files in one place, and what survives after the session. Plain language for operators.",{"eyebrow":2414,"title":2415},"Common questions","Score the shared job",[2417,2420,2423,2426],{"question":2418,"answer":2419},"Is this the same as evaluating a copilot?","No. Copilot evals score solo drafting — latency, paragraph quality, seat SSO. This sheet scores whether two departments share one job, with a named stop and artefacts that survive the session. A product can pass every copilot test and still fail the moment finance joins Thursday’s exception. If your RFP only asks context-window size and brand of model, you will buy an assistant and call it collaboration. Read the personal-assistant comparison first if you are still deciding whether you need a shared room at all.",{"question":2421,"answer":2422},"Do we need multiplayer and collaborative AI in the RFP?","Ask both. Multiplayer is who is in the room now — can a second department join, see the same brief, and reject a proposal while the work is happening. Collaborative is whether Monday still has the files, the rejection, and the signer after everyone hangs up. You can fail one and pass the other. Five people in a call with nothing written down is multiplayer for an hour. A job two teams open on different days can be collaborative without a live huddle. Cross-department work that may write to a live system needs both.",{"question":2424,"answer":2425},"What is the fastest proof-of-value test?","Invite a second department mid-run, reject a proposal, swap the prompter, reopen Monday. If the job is empty, you evaluated a shared login. Do not let the vendor pick a solo email-draft scenario. Name a real exception you already fight in chat. Require a stored rejection with a name and a timestamp before anyone enables write. An independent reader who was not in the demo should reconstruct signer and payload without Slack. If they cannot, fail the proof even if the paragraph was fluent.",{"question":2427,"answer":2428},"Can we reuse our agent-harness RFP instead?","Reuse it as a sibling, not a substitute. The harness sheet scores refused writes, replay, and model swap on the runtime. This sheet scores roster, handover, and whether files live on the job. You need both when the purchase is AI for cross-team exceptions that may touch CRM, ERP, or WMS. Harness without collaborative passes produces a gated write finance never saw. Collaborative without harness passes produces a shared room where unsigned writes still slip through. Copy both question lists into procurement.","\u002Fblog\u002Fhow-to-evaluate-collaborative-ai",{"title":1612,"description":2412},"blog\u002Fhow-to-evaluate-collaborative-ai",[941,2433,2434,2435],"collaborative-ai","RFP","multiplayer","HE9Si4bvYjh3cdtW3TifBd1idlhzsISzDNXFWiG50wY",{"id":2438,"title":2439,"archived":165,"authors":2440,"badge":2442,"body":2443,"date":2411,"definedTerm":166,"department":166,"description":3115,"extension":174,"eyebrow":166,"faqHeader":3116,"faqs":3118,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":3130,"relatedHeading":166,"seo":3131,"series":941,"sitemap":131,"status":166,"stem":3132,"subhead":166,"tags":3133,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":3136},"content\u002Fblog\u002Fhow-to-evaluate-loop-engineering.md","How to Evaluate Loop Engineering",[2441],{"name":185,"to":136},{"label":187},{"type":168,"value":2444,"toc":3097},[2445,2461,2464,2467,2513,2521,2525,2528,2548,2566,2569,2589,2597,2601,2604,2629,2633,2644,2650,2653,2656,2659,2663,2666,2669,2672,2675,2678,2682,2689,2700,2703,2706,2709,2713,2720,2726,2729,2732,2735,2739,2755,2758,2761,2764,2767,2771,2774,2785,2788,2791,2794,2798,2803,2806,2809,2812,2815,2819,2825,2838,2841,2844,2847,2851,2854,2884,2890,2894,2897,2903,2909,2915,2921,2924,2926,3013,3025,3032,3036,3039,3049,3059,3068,3080,3082,3092,3095],[191,2446,2447,2448,2451,2452,2455,2456,2460],{},"Evaluating ",[195,2449,2450],{},"loop engineering"," means testing whether a vendor can run ",[195,2453,2454],{},"standing orders"," the way operators actually work — skip when nothing changed, record quiet outcomes, reuse recipes, notify a named roster — not whether a demo chat looked fluent. ",[204,2457,2459],{"href":399,"rel":2458},[208],"NIST’s AI Risk Management Framework"," Measure and Manage functions assume you can observe outcomes and tighten controls after harm. A loop without a run page is not observable. It is email archaeology.",[191,2462,2463],{},"This sheet is vendor-agnostic. Paste it into an RFP. Run it on an incumbent automation suite, a new “agentic” platform, or a homegrown scheduler. The object under test is the standing order: a compiled recipe that starts on a signal and leaves a record. If the vendor cannot show that object, stop scoring adjectives.",[191,2465,2466],{},"Three nouns vendors conflate on slide one:",[263,2468,2469,2483,2492],{},[266,2470,2471,2474,2475,813,2479,303],{},[195,2472,2473],{},"Loop."," A compiled standing order with triggers and a run page. See ",[204,2476,2478],{"href":2477},"what-is-loop-engineering","what is loop engineering",[204,2480,2482],{"href":2481},"six-things-that-start-a-loop","six things that start a loop",[266,2484,2485,2488,2489,303],{},[195,2486,2487],{},"Eval loop."," Independent verification that a job finished — tests, read-back, signer — not a trigger. See ",[204,2490,2491],{"href":453},"eval loops for enterprise agent harnesses",[266,2493,2494,2497,2498,2502,2503,392,2505,2508,2509,2512],{},[195,2495,2496],{},"Agentic workflow."," A designed interactive sequence with stops. See ",[204,2499,2501],{"href":2500},"what-is-an-agentic-workflow","what is an agentic workflow",". Score it with ",[204,2504,676],{"href":529},[195,2506,2507],{},"That"," sheet scores the interactive harness. ",[195,2510,2511],{},"This"," sheet scores repeat automation.",[191,2514,2515,2516,2520],{},"If your RFP only asks context-window size, you will buy a model for a Monday cron job. ",[204,2517,905],{"href":2518,"rel":2519},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F81230.html",[208]," wants named actors and records around AI systems. “The bot ran” is not a record. “The assistant said it was done” is not a record.",[258,2522,2524],{"id":2523},"why-loop-evaluation-fails","Why loop evaluation fails",[191,2526,2527],{},"Teams reuse copilot scorecards because those cards are already in the drawer. They fail in predictable ways:",[1082,2529,2530,2536,2542],{},[266,2531,2532,2535],{},[195,2533,2534],{},"The demo pass."," A model summarises a file beautifully. No trigger, no skip, no version. You have scored reading, not standing orders.",[266,2537,2538,2541],{},[195,2539,2540],{},"The integration pass."," “We connect to Salesforce.” No run page, no quiet outcome, no roster. You have scored a connector, not an operating object.",[266,2543,2544,2547],{},[195,2545,2546],{},"The personal-automation pass."," A chain fires. Outcomes scatter across personal inboxes. No organisational recipe. You have scored glue, not loop engineering.",[191,2549,2550,2553,2554,2557,2558,2561,2562,2565],{},[204,2551,255],{"href":253,"rel":2552},[208]," reports wide experimentation and narrower scale. Loop engineering is how repeat work scales: ",[195,2555,2556],{},"compile"," what worked, ",[195,2559,2560],{},"trigger"," it reliably, ",[195,2563,2564],{},"record"," what happened. High-performing organisations in that survey are more likely to redesign workflows, not merely sprinkle assistants on the current process. This RFP is a redesign test.",[191,2567,2568],{},"Red flags you can mark in the room:",[263,2570,2571,2574,2577,2580,2583,2586],{},[266,2572,2573],{},"Outcomes live only in chat transcripts",[266,2575,2576],{},"“Success” when there was nothing to do — but Finance still got paged",[266,2578,2579],{},"Reuse means “find Sarah’s Slack thread from Q2”",[266,2581,2582],{},"Every write goes through a model — even fixed journal templates",[266,2584,2585],{},"No pause — cancel kills state without a resumable run",[266,2587,2588],{},"Notifications go to channels, not to a roster attached to the job",[191,2590,2591,2596],{},[204,2592,2595],{"href":2593,"rel":2594},"https:\u002F\u002Faiindex.stanford.edu\u002F",[208],"Stanford HAI’s AI Index"," is a useful external reminder that capability is not the scarce input. You are not scoring whether the model can write a polite email. You are scoring whether Tuesday’s job exists when the author is on leave.",[258,2598,2600],{"id":2599},"the-rfp-sheet-eight-tests","The RFP sheet — eight tests",[191,2602,2603],{},"Each test has a why, a when, a thing to do, and a thing to refuse. Run them in order if you are short on time: skip semantics first, then run page, then reuse. A fluent demo that fails test 1 is not a standing order.",[1690,2605,2607],{"className":1692,"code":2606,"language":1694,"meta":171,"style":171},"flowchart LR\n  skipTest[\"Can it skip quietly?\"] --> pageTest[\"Is there a run page?\"]\n  pageTest --> reuseTest[\"Can another team reuse it?\"]\n  reuseTest --> refuseTest[\"Can someone refuse a write?\"]\n",[273,2608,2609,2613,2618,2623],{"__ignoreMap":171},[1698,2610,2611],{"class":1700,"line":1701},[1698,2612,1704],{},[1698,2614,2615],{"class":1700,"line":172},[1698,2616,2617],{},"  skipTest[\"Can it skip quietly?\"] --> pageTest[\"Is there a run page?\"]\n",[1698,2619,2620],{"class":1700,"line":930},[1698,2621,2622],{},"  pageTest --> reuseTest[\"Can another team reuse it?\"]\n",[1698,2624,2626],{"class":1700,"line":2625},4,[1698,2627,2628],{},"  reuseTest --> refuseTest[\"Can someone refuse a write?\"]\n",[752,2630,2632],{"id":2631},"_1-can-it-skip-when-nothing-changed","1. Can it skip when nothing changed?",[191,2634,2635,2636,2639,2640,2643],{},"Run the loop on unchanged inputs. The run page should say ",[195,2637,2638],{},"skipped"," or ",[195,2641,2642],{},"no op"," — with timestamp and recipe version — not green-check spam.",[191,2645,2646,2647,2649],{},"Why it matters: month-end with zero exceptions is success. Bots that “succeed” on empty tables train operators to ignore alerts. See ",[204,2648,2482],{"href":2481}," for triggers that should dedupe.",[191,2651,2652],{},"When to insist: any scheduled or event-driven job that will run in unattended hours.",[191,2654,2655],{},"What to do: show ten consecutive skipped runs. Show alert volume — ideally zero.",[191,2657,2658],{},"What to refuse: a “success” email on an empty extract, or a skip that exists only as the absence of mail. Absence is not a record.",[752,2660,2662],{"id":2661},"_2-can-it-record-quiet-outcomes","2. Can it record quiet outcomes?",[191,2664,2665],{},"Quiet success is a first-class outcome: ran, nothing to write, roster optionally notified at digest frequency.",[191,2667,2668],{},"Why it matters: auditors and controllers ask what happened on the twelfth — including days nothing moved. Silence without a record is indistinguishable from failure. NIST’s Measure function is not optional because the week was quiet.",[191,2670,2671],{},"When to insist: regulated work, shared-service work, anything a second person will reconstruct.",[191,2673,2674],{},"What to do: export skipped runs for a month without vendor engineering.",[191,2676,2677],{},"What to refuse: a professional-services quote to produce last month’s quiet days. If export is a project, observation is not a product feature.",[752,2679,2681],{"id":2680},"_3-can-you-parameterise-a-write-without-a-model","3. Can you parameterise a write without a model?",[191,2683,2684,2685,2688],{},"Show a loop that posts or quotes a ",[195,2686,2687],{},"fixed-shape"," payload — accounts, amounts, CRM fields — from structured input, with no language model in the path.",[191,2690,2691,2692,813,2696,303],{},"Why it matters: many operational writes are templates, not essays. If the vendor routes everything through chat, you are paying inference tax on deterministic work and importing non-determinism into close. Compare ",[204,2693,2695],{"href":2694},"loop-vs-workflow-vs-agent","loop vs workflow vs agent",[204,2697,2699],{"href":2698},"a-loop-is-not-an-agent","a loop is not an agent",[191,2701,2702],{},"When to insist: journals, stage updates, status writes, any payload a controller could have typed from a spreadsheet.",[191,2704,2705],{},"What to do: disable the model. Does the loop still quote the write and wait on sign?",[191,2707,2708],{},"What to refuse: “the model is more flexible” as an answer to a fixed schema. Flexibility on a journal line is a defect.",[752,2710,2712],{"id":2711},"_4-can-it-pause-and-resume-cleanly","4. Can it pause and resume cleanly?",[191,2714,2715,2716,2719],{},"A loop waiting on a file, a signer, or an external system should ",[195,2717,2718],{},"pause"," with visible state — not vanish into a thread.",[191,2721,2722,2723,2725],{},"Why it matters: close week spans days. Operations must distinguish “waiting on a person” from “broken.” ",[204,2724,1893],{"href":395}," applies to standing orders too.",[191,2727,2728],{},"When to insist: any recipe that crosses a night, a weekend, or a named approver.",[191,2730,2731],{},"What to do: pause mid-run. Attach the missing file. Resume without restarting from scratch unless you choose to. Keep the same run identifier.",[191,2733,2734],{},"What to refuse: cancel-as-pause. If state dies, you do not have a pause. You have a restart with extra steps.",[752,2736,2738],{"id":2737},"_5-can-it-notify-the-roster-not-a-copied-channel","5. Can it notify the roster — not a copied channel?",[191,2740,2741,2742,2745,2746,2749,2750,2754],{},"Notifications should target the people on the job — controller, RevOps, counsel — with a link to the ",[195,2743,2744],{},"run page",". See ",[204,2747,2748],{"href":1983},"what is an AI workstream"," for the job-object idea, and ",[204,2751,2753],{"href":2752},"how-to-evaluate-collaborative-ai","how to evaluate collaborative AI"," for the roster questions.",[191,2756,2757],{},"Why it matters: Slack channels rot when people leave. Rosters follow the job. A copied channel is how a departed contractor keeps getting close packs, and how the new controller never does.",[191,2759,2760],{},"When to insist: any loop that another department will act on.",[191,2762,2763],{},"What to do: remove one person from the roster. Prove they stop receiving loop notifications without creating a new automation.",[191,2765,2766],{},"What to refuse: “we’ll update the webhook.” If membership is not data, notification is folklore.",[752,2768,2770],{"id":2769},"_6-can-you-reuse-a-recipe-without-copying-the-old-slack-channel","6. Can you reuse a recipe without copying the old Slack channel?",[191,2772,2773],{},"Clone the loop — triggers, steps, gates, notify rules — into a new team or region without re-prompting from memory.",[191,2775,2776,2777,2779,2780,2784],{},"Why it matters: ",[204,2778,2450],{"href":2477}," is an organisational capability, not hero prompts. If reuse requires export to JSON and a services quote, note the tax. ",[204,2781,2783],{"href":245,"rel":2782},[208],"Thoughtworks’ operating-system framing"," is useful here: ownership and durable state belong to the job, not to the person who first described it.",[191,2786,2787],{},"When to insist: any recipe you will need in a second business unit within a year.",[191,2789,2790],{},"What to do: stand up the same loop for a second team in under one hour — operator-led.",[191,2792,2793],{},"What to refuse: a clone that copies the prompt but drops the skip rules, the signer, or the run-page contract. That is a new folklore, not reuse.",[752,2795,2797],{"id":2796},"_7-does-every-trigger-land-on-the-same-run-page","7. Does every trigger land on the same run page?",[191,2799,2800,2801,303],{},"Schedule, file, data change, drop, ping, run now — several doors, one outcome surface. See ",[204,2802,2482],{"href":2481},[191,2804,2805],{},"Why it matters: operators should not learn six UIs. Audit should not merge six log formats. If the scheduled close and the emergency rerun do not look like the same object, you will get two classes of evidence.",[191,2807,2808],{},"When to insist: as soon as a team has more than one start condition for the same recipe.",[191,2810,2811],{},"What to do: fire two different triggers against the same recipe. Show both run pages side by side.",[191,2813,2814],{},"What to refuse: a “manual” path that writes with weaker gates than the scheduled path. Urgency is not a policy exception.",[752,2816,2818],{"id":2817},"_8-can-you-refuse-a-write-and-prove-the-system-of-record-unchanged","8. Can you refuse a write and prove the system of record unchanged?",[191,2820,2821,2822,303],{},"Even loops that only draft should demonstrate fail-closed behaviour when signers reject. Loops that write must show quote → sign → execute → read-back. See ",[204,2823,2824],{"href":390},"what is write-back governance",[191,2826,2827,2828,2830,2831,2834,2835,2837],{},"Why it matters: this is where loop evaluation meets harness evaluation. ",[204,2829,530],{"href":529}," test one — ",[195,2832,2833],{},"stop an action"," — applies to automated writes too. ",[204,2836,2274],{"href":2273}," is the category language.",[191,2839,2840],{},"When to insist: before any production write, including “just a status field.”",[191,2842,2843],{},"What to do: show a rejected payload. Show the CRM or ERP unchanged. Show the rejection on the run page, with who rejected and which policy version applied.",[191,2845,2846],{},"What to refuse: a write that cannot be shown in a quoted form. A paragraph the model later “applies” is not a payload.",[258,2848,2850],{"id":2849},"rfp-questions-paste-these","RFP questions — paste these",[191,2852,2853],{},"These are the short versions you can drop into a vendor questionnaire. They are not vendor-specific. They are not even AI-specific. They are standing-order tests.",[1082,2855,2856,2859,2866,2869,2872,2875,2878,2881],{},[266,2857,2858],{},"Show ten skipped runs with timestamps and recipe version.",[266,2860,2861,2862,2865],{},"Show a write path with ",[195,2863,2864],{},"no model call"," — structured in, quoted out.",[266,2867,2868],{},"Pause a run for 48 hours; resume; show a continuous run identifier.",[266,2870,2871],{},"Clone a loop to a second team without re-entering prompts.",[266,2873,2874],{},"Trigger the same recipe via schedule and via file drop; compare run pages.",[266,2876,2877],{},"Remove a roster member; prove notifications stop.",[266,2879,2880],{},"Reject a quoted write; prove the system of record unchanged; show who rejected.",[266,2882,2883],{},"Where do outputs live if email is down — still on the run page?",[191,2885,2886,2887,2889],{},"Add the harness sheet when loops call agents or share connectors with interactive work. Add ",[204,2888,2753],{"href":2752}," when the output is a signed pack rather than a silent write.",[258,2891,2893],{"id":2892},"proof-of-value-one-week","Proof of value — one week",[191,2895,2896],{},"Do not spend the week watching a prepared demo. Spend it on one real job.",[191,2898,2899,2902],{},[195,2900,2901],{},"Day 1–2:"," Pick one repeat job — weekly pipeline summary, bank file intake, redline folder watch. Name the trigger your team already watches. Write the skip condition in a sentence a controller would accept.",[191,2904,2905,2908],{},[195,2906,2907],{},"Day 3:"," Run ten times with empty or stale inputs. Demand skipped run pages. If you cannot get them, the rest of the week is theatre.",[191,2910,2911,2914],{},[195,2912,2913],{},"Day 4:"," Run once with a real change. Confirm roster notification links to the run — not a pasted screenshot. Confirm an independent reader can open the run without the author.",[191,2916,2917,2920],{},[195,2918,2919],{},"Day 5:"," Clone to a second roster or region. Time it. Attempt a refused write. Confirm the system of record did not move.",[191,2922,2923],{},"Pass criteria: skip semantics, run page, roster notify, reuse, and one refused or unsigned write that did not land. Fail any one and you do not have loop engineering. You have a demo that will not survive the first quiet week.",[258,2925,2218],{"id":2217},[549,2927,2928,2945],{},[552,2929,2930],{},[555,2931,2932,2935,2938],{},[558,2933,2934],{},"Topic",[558,2936,2937],{},"Loop engineering (this page)",[558,2939,2940,2941,2944],{},"Agent harness (",[204,2942,2943],{"href":529},"sibling sheet",")",[565,2946,2947,2958,2969,2980,2991,3002],{},[555,2948,2949,2952,2955],{},[570,2950,2951],{},"Unit of buy",[570,2953,2954],{},"Standing order \u002F recipe",[570,2956,2957],{},"Interactive runtime",[555,2959,2960,2963,2966],{},[570,2961,2962],{},"Hero metric",[570,2964,2965],{},"Skipped vs processed runs",[570,2967,2968],{},"Refused writes \u002F replay",[555,2970,2971,2974,2977],{},[570,2972,2973],{},"Trigger surface",[570,2975,2976],{},"Several signal types",[570,2978,2979],{},"User goal \u002F chat",[555,2981,2982,2985,2988],{},[570,2983,2984],{},"Quiet success",[570,2986,2987],{},"No op recorded",[570,2989,2990],{},"Waiting on signer",[555,2992,2993,2996,2999],{},[570,2994,2995],{},"Reuse",[570,2997,2998],{},"Recipe library",[570,3000,3001],{},"Versioned harness + tools",[555,3003,3004,3007,3010],{},[570,3005,3006],{},"Model role",[570,3008,3009],{},"Optional, bounded step",[570,3011,3012],{},"Often central",[191,3014,3015,3016,3019,3020,3024],{},"You need both sheets if ",[204,3017,3018],{"href":1761},"the four pillars"," describe your stack — Automate (loops) plus Collaboration (agents on workstreams) under governance. ",[204,3021,3023],{"href":3022},"loop-engineering-vs-harness-engineering","Loop engineering vs harness engineering"," explains the crafts. This page and the harness page are how you buy them separately.",[191,3026,3027,3031],{},[204,3028,3030],{"href":3029},"loop-vs-rpa","Loop vs RPA"," is the adjacent comparison if incumbents sell bots. Do not let an RPA success-email farm satisfy test 1. “The script finished” is not a skipped run.",[258,3033,3035],{"id":3034},"department-lenses","Department lenses",[191,3037,3038],{},"Use the same eight tests. Change the job you bring to the POV.",[191,3040,3041,3044,3045,303],{},[195,3042,3043],{},"Finance"," — scheduled close packs, parameterised journals, controller on the roster. Skip on a week with no exceptions. See ",[204,3046,3048],{"href":3047},"loops-for-finance-and-planning","loops for finance and planning",[191,3050,3051,3054,3055,303],{},[195,3052,3053],{},"RevOps"," — stage-triggered hygiene, skip when fields are unchanged, notify the people who own the stage definition. See ",[204,3056,3058],{"href":3057},"loops-for-revenue-operations","loops for revenue operations",[191,3060,3061,3063,3064,303],{},[195,3062,64],{}," — inbound redlines, notify counsel, no send without a workflow gate. The loop starts the job; the workflow owns the stop. See ",[204,3065,3067],{"href":3066},"loops-for-legal-and-compliance","loops for legal and compliance",[191,3069,3070,3071,286,3074,3076,3077,3079],{},"Related reading: ",[204,3072,3073],{"href":1635},"what is collaborative AI",[204,3075,2699],{"href":2698},", and ",[204,3078,2360],{"href":2359}," when you need the kernel metaphor rather than the RFP sheet.",[258,3081,1415],{"id":1414},[191,3083,3084,3085,3087,3088,3091],{},"Nimbus Loops are one implementation of the standing-order object this sheet scores: triggers, recipes, run pages, roster notify, and ",[204,3086,357],{"href":40}," gates on a ",[204,3089,3090],{"href":32},"workstream",". You can run the eight tests there. You should also run them on whoever else claims “unattended agents” or “intelligent automation.”",[191,3093,3094],{},"Nimbus is not loop engineering. Loop engineering is whether your organisation can compile repeat work, skip quietly, and leave a record a second person can open. The product should make those tests boring. If a Nimbus demo — or any demo — cannot show ten skipped runs and one refused write, treat it as a copilot evaluation and score it on the harness sheet instead.",[2390,3096,2392],{},{"title":171,"searchDepth":172,"depth":172,"links":3098},[3099,3100,3110,3111,3112,3113,3114],{"id":2523,"depth":172,"text":2524},{"id":2599,"depth":172,"text":2600,"children":3101},[3102,3103,3104,3105,3106,3107,3108,3109],{"id":2631,"depth":930,"text":2632},{"id":2661,"depth":930,"text":2662},{"id":2680,"depth":930,"text":2681},{"id":2711,"depth":930,"text":2712},{"id":2737,"depth":930,"text":2738},{"id":2769,"depth":930,"text":2770},{"id":2796,"depth":930,"text":2797},{"id":2817,"depth":930,"text":2818},{"id":2849,"depth":172,"text":2850},{"id":2892,"depth":172,"text":2893},{"id":2217,"depth":172,"text":2218},{"id":3034,"depth":172,"text":3035},{"id":1414,"depth":172,"text":1415},"Evaluating loop engineering means asking whether standing orders can skip quietly, record outcomes on a run page, parameterise a write without a model, pause, notify the roster, and reuse a recipe without copying the old Slack channel.",{"eyebrow":2414,"title":3117},"RFP tests for standing orders",[3119,3122,3124,3127],{"question":3120,"answer":3121},"Is this the same checklist as evaluating an agent harness?","Sibling, not duplicate, and scoring them on one sheet is how you buy a model for a Monday cron job. [How to evaluate an agent harness](how-to-evaluate-an-agent-harness) scores the interactive runtime — stops, replay, sensors around agents. This page scores standing orders — triggers, skip semantics, run pages, recipe reuse. Use both if you run agents and loops on the same roster. Use only this page if the job is unattended repeat work with a known path. Refuse a vendor that answers harness questions with a skipped-run demo, or loop questions with a fluent chat. They are different objects and they fail differently.",{"question":2424,"answer":3123},"Run the same standing order ten times with empty or unchanged inputs. You should get ten run pages marked skipped — and zero spurious writes or alert storms. Then run once with a real change and confirm the roster was notified with a link to the run, not a pasted screenshot. Do this on a job the team already watches, not on a synthetic demo the vendor prepared. Refuse a POV that only shows a beautiful summary of a file. That is a copilot test. It tells you nothing about whether Tuesday’s close exists as an object when the author is out.",{"question":3125,"answer":3126},"Do we need a model in every loop?","No, and a vendor that cannot show a loop without one is selling inference, not loop engineering. Many standing orders are deterministic — compare, route, notify, quote a write for sign. Ask for a parameterised write path that does not call a model. If everything routes through chat, you are scoring a copilot. Use a model step when the input is messy and the rest of the recipe is known. Refuse a design that puts a language model in the path of a fixed journal template. You will pay twice: tokens, and the day the model invents an account code.",{"question":3128,"answer":3129},"What should we refuse even if the demo is fluent?","Refuse outcomes that live only in transcripts, and refuse “success” on empty inputs that still pages Finance. Refuse reuse that means finding last quarter’s thread, and refuse a write that cannot be shown with the model disabled. Refuse a pause that kills state, and refuse notifications that go to a copied channel rather than a roster you can edit. NIST’s AI Risk Management Framework assumes you can observe outcomes and then manage them. If you cannot export a month of skipped runs without a professional-services ticket, you cannot Measure — and you should not buy.","\u002Fblog\u002Fhow-to-evaluate-loop-engineering",{"title":2439,"description":3115},"blog\u002Fhow-to-evaluate-loop-engineering",[3134,941,2434,3135],"loops","automation","0qwQuy9uhlMrnUTMza--HZ1qmoRcv2zODPJJZ7ggXZE",{"id":3138,"title":3139,"archived":165,"authors":3140,"badge":3142,"body":3143,"date":3360,"definedTerm":166,"department":166,"description":3361,"extension":174,"eyebrow":166,"faqHeader":3362,"faqs":3365,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":3378,"relatedHeading":166,"seo":3379,"series":941,"sitemap":131,"status":166,"stem":3380,"subhead":166,"tags":3381,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":3383},"content\u002Fblog\u002Fwhat-auditors-are-asking-for.md","What are auditors asking for around AI?",[3141],{"name":185,"to":136},{"label":187},{"type":168,"value":3144,"toc":3353},[3145,3148,3151,3164,3167,3170,3176,3182,3185,3188,3191,3198,3201,3205,3208,3216,3222,3228,3235,3243,3246,3250,3253,3256,3259,3276,3279,3282,3285,3293,3297,3300,3320,3327,3330,3333,3337,3340,3343,3346],[191,3146,3147],{},"Auditors asking about AI usually want to follow one change: who decided, whether software could write without a person, and which rulebook you claim to follow. They pick a journal, a credit, a customer email, or a model connection, and they walk it from prompt to record.",[191,3149,3150],{},"This is showing up now because models sit on live systems, and existing control texts already care how a number became the number. You do not need every framework on day one. You need artefacts you can produce without asking anyone to remember.",[191,3152,3153,3154,3156,3157,3159,3160,3163],{},"This guide is a first evidence pack you can start this quarter. ",[204,3155,2274],{"href":2273}," is the rest of the access picture. ",[204,3158,2021],{"href":2020}," is who may see the job. ",[204,3161,3162],{"href":390},"Write-back governance"," is the write checklist.",[258,3165,3139],{"id":3166},"what-are-auditors-asking-for-around-ai",[191,3168,3169],{},"Two operational questions arrive first.",[191,3171,3172,3175],{},[195,3173,3174],{},"Can you show who decided?"," A named person, on a clock the company trusts, bound to a quote that matches the write. “The team aligned” is not an answer. “The channel approved” is not an answer. “The bot user posted” is not an answer.",[191,3177,3178,3181],{},[195,3179,3180],{},"Can you show the model did not write unchecked?"," Write-back means the AI changes a live system. Fail-closed means if nobody approves, nothing happens. A prompt that says “ask first” is not the gate. A weekly sampling of logs is not the gate if the write already landed.",[191,3183,3184],{},"Role-based access control (RBAC) means who is allowed to do what. It explains why that person, and not a guest, was offered the button. Auditors understand roles. They do not understand “the workspace.”",[191,3186,3187],{},"A payload is the exact change: fields, old and new values, target record — or the exact text and recipient for a message.",[191,3189,3190],{},"Then comes the mapping question: which framework applies to us? Not every company is under every text. Pretending otherwise produces a pile of mappings and no artefact.",[191,3192,3193,3197],{},[204,3194,3196],{"href":3195},"collaborative-ai-for-legal-and-compliance-review","Collaborative AI for legal and compliance review"," still needs a signer when the review becomes a filing. Several departments on one job is not a shared identity.",[191,3199,3200],{},"If the decision was “we will not write,” that is still a decision. Store it. A read-only connector with a date and an owner is evidence.",[258,3202,3204],{"id":3203},"why-is-this-showing-up-now","Why is this showing up now?",[191,3206,3207],{},"Models are in the path of records that already had auditors: financial reporting, customer commitments, legal filings, operational tickets.",[191,3209,3210,3215],{},[204,3211,3214],{"href":3212,"rel":3213},"https:\u002F\u002Fwww.govinfo.gov\u002Fcontent\u002Fpkg\u002FPLAW-107publ204\u002Fhtml\u002FPLAW-107publ204.htm",[208],"Sarbanes-Oxley"," (2002) is still the text many US-listed teams feel first. Internal control over financial reporting does not care that the proposer is a model. If AI can post, the control environment includes that path.",[191,3217,2130,3218,3221],{},[204,3219,2134],{"href":399,"rel":3220},[208]," (2023) is organised as govern, map, measure, and manage. Measure, here, is the stored outcome, including the no. Govern is the roles and the owners. The framework will not click the refuse button for you.",[191,3223,3224,3227],{},[204,3225,905],{"href":2518,"rel":3226},[208]," (2023) adds a management system for AI: policies, roles, risk assessment, documented processes, and evidence that those processes run. Useful if you will be asked for a certificate. Not a substitute for a payload screen.",[191,3229,3230,3231,3234],{},"The ",[204,3232,711],{"href":709,"rel":3233},[208]," (2024\u002F1689) is from 2024. A deployer is the organisation that uses an AI system under its authority, as the Act defines that role. You may also be a provider if you place a system on the market. Map the role with counsel. Human oversight that cannot refuse a write is not oversight.",[191,3236,3237,3242],{},[204,3238,3241],{"href":3239,"rel":3240},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2022\u002F2554\u002Foj",[208],"DORA"," (2022) is about digital operational resilience for financial entities and their ICT third parties. If you are in that sector, the AI vendor is an ICT provider conversation, not only an innovation conversation.",[191,3244,3245],{},"Customers and boards ask for structure even when a text is voluntary. That is why the questions arrive before a regulator has written your company’s name.",[258,3247,3249],{"id":3248},"how-do-you-prepare-evidence-without-a-huge-project","How do you prepare evidence without a huge project?",[191,3251,3252],{},"Do the one-change walk before anyone external does.",[191,3254,3255],{},"Pick a change a model proposed. Follow it from prompt to record. See whether you can produce a named person, a frozen payload, and a stored outcome without anyone’s memory.",[191,3257,3258],{},"Show:",[263,3260,3261,3264,3267,3270,3273],{},[266,3262,3263],{},"A connector in read-only mode, and a failed write attempt.",[266,3265,3266],{},"One object class with a frozen payload and a named signer — or a dated decision that no class is enabled yet.",[266,3268,3269],{},"The live system’s own validation still firing, if a write ran.",[266,3271,3272],{},"A success and a rejection.",[266,3274,3275],{},"A person who was removed and could not sign the next day.",[191,3277,3278],{},"If you cannot show the failed attempt, assume an auditor will treat write as on.",[191,3280,3281],{},"Unchecked also includes send. A customer message is a write to the relationship. If mail can go out because the connector was on for retrieval, that is an unchecked write with no field names to screenshot.",[191,3283,3284],{},"Do not start with a coverage matrix against every clause. Breadth without a sample fails the first request. Depth on one change lets you map the same artefact twice if two texts apply.",[191,3286,3287,3292],{},[204,3288,3291],{"href":3289,"rel":3290},"https:\u002F\u002Fwww.law.cornell.edu\u002Frules\u002Ffrcp\u002Frule_37",[208],"Federal Rule of Civil Procedure 37(e)"," (2015) is about preserving electronically stored information you should have kept. Chat retention sliders are not that programme. Put approvals where a new manager can find them.",[258,3294,3296],{"id":3295},"what-is-a-reasonable-first-evidence-pack","What is a reasonable first evidence pack?",[191,3298,3299],{},"One page plus exports:",[263,3301,3302,3305,3308,3311,3314,3317],{},[266,3303,3304],{},"Job name, system, connector mode, date, owner.",[266,3306,3307],{},"Roster: guest, member, admin, signer — or “signer not yet named; write off.”",[266,3309,3310],{},"One stored refusal (sandbox is fine).",[266,3312,3313],{},"One stored success if you have enabled a class; otherwise omit.",[266,3315,3316],{},"Clock and retention note: where the artefact lives, how long, who can export it without a vendor ticket.",[266,3318,3319],{},"Which texts you claim: SOX ICFR if you file; NIST AI RMF as structure; ISO 42001 if you are on that path; EU AI Act role if in scope; DORA if you are a financial entity.",[191,3321,3322,3323,3326],{},"Your ",[204,3324,3325],{"href":126},"compliance"," programme should hold that page.",[191,3328,3329],{},"ISO 42001, if you take it seriously, adds an owner for AI, a statement of which systems models may connect to and in which mode, a way to handle incidents and model or prompt changes that alter write behaviour, and records that last longer than a chat default. It does not add object-level tokens. You can be certified and still have an admin token on a model. Ask the auditor of that management system to sample a stored rejection from a live job.",[191,3331,3332],{},"For deployers under the EU AI Act, the operational match is: know you are using AI, use it as intended, monitor, keep required records, and ensure human oversight where the Act requires it. “The vendor is the provider” does not move your ERP posting into their audit file. Your token, your records, your signer. High-risk classification is legal work. This guide will not guess it.",[258,3334,3336],{"id":3335},"how-do-you-start-this-quarter","How do you start this quarter?",[191,3338,3339],{},"This month: pick one real job. Run the one-change walk. Write the one-page pack. Fill blanks as findings, not as a reason to delay the page.",[191,3341,3342],{},"Next month: fix the first hole — usually the stored no, the read-only proof, or the named signer.",[191,3344,3345],{},"If you cannot complete the walk, keeping write off is the honest state of the control. Mapping will not replace it.",[191,3347,3348,813,3350,3352],{},[204,3349,3162],{"href":390},[204,3351,2021],{"href":2020}," are the two product habits that make the pack easier to gather later.",{"title":171,"searchDepth":172,"depth":172,"links":3354},[3355,3356,3357,3358,3359],{"id":3166,"depth":172,"text":3139},{"id":3203,"depth":172,"text":3204},{"id":3248,"depth":172,"text":3249},{"id":3295,"depth":172,"text":3296},{"id":3335,"depth":172,"text":3336},"2026-09-06","Who decided, did the model write unchecked, and which rulebook applies. How to prepare a first evidence pack this quarter without a huge project.",{"eyebrow":3363,"title":3364},"Short answers","One change you can walk",[3366,3369,3372,3375],{"question":3367,"answer":3368},"If we are not in the EU, can we ignore the AI Act?","You can ignore it as a legal duty only if you are not in its scope. You should still answer the same operational questions — who decided, and did the model write unchecked — because auditors and customers will ask them in other words.",{"question":3370,"answer":3371},"Does ISO 42001 certification mean our CRM writes are governed?","No. Certification speaks to a management system. It does not replace a named person on a payload, a stored rejection, or a connector that can be read-only. Ask to see those artefacts in your product, not only the certificate.",{"question":3373,"answer":3374},"What is the smallest evidence pack that still helps?","One change a model proposed: named person, frozen payload, stored outcome including a no, plus the connector mode and the roster on that job. Map that pack to whichever texts apply. Do not start with a matrix of empty controls.",{"question":3376,"answer":3377},"Do we need this if AI is still read-only?","A dated decision to stay read-only, with an owner and the connector name, is evidence. You need the full write pack before the first production write class.","\u002Fblog\u002Fwhat-auditors-are-asking-for",{"title":3139,"description":3361},"blog\u002Fwhat-auditors-are-asking-for",[941,3382,427,711,3325],"audit","EfFv_YZ08TZJm4MDym8B05aD4MGOwabpfOQ5EtWvDxw",{"id":3385,"title":3386,"archived":165,"authors":3387,"badge":3389,"body":3390,"date":3774,"definedTerm":166,"department":166,"description":3775,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":3776,"relatedHeading":166,"seo":3777,"series":941,"sitemap":131,"status":166,"stem":3778,"subhead":166,"tags":3779,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":3782},"content\u002Fblog\u002Fwhat-to-look-for-in-model-routing.md","What to Look for in Model Routing",[3388],{"name":185,"to":136},{"label":187},{"type":168,"value":3391,"toc":3757},[3392,3397,3400,3417,3419,3466,3477,3481,3484,3504,3511,3531,3537,3540,3544,3582,3592,3595,3599,3602,3628,3631,3633,3636,3643,3651,3653,3657,3663,3667,3670,3674,3677,3681,3688,3692,3699,3701,3712,3714],[191,3393,3394,3396],{},[195,3395,1065],{}," is the policy that maps a task class to a model class before inference runs. It is not a brand preference. It is not a dropdown labelled “best.”",[191,3398,3399],{},"Someone using the flagship model to label a ticket is how you pay frontier prices for work a compact model could have finished in a second. That is not a moral failing. It is a missing policy. The product either chooses before the call, or a person chooses in a menu, or the default is the largest model “for quality.” Only the first is routing.",[191,3401,3402,813,3407,3412,3413,3416],{},[204,3403,3406],{"href":3404,"rel":3405},"https:\u002F\u002Fopenai.com\u002Fapi\u002Fpricing\u002F",[208],"OpenAI’s API pricing",[204,3408,3411],{"href":3409,"rel":3410},"https:\u002F\u002Fwww.anthropic.com\u002Fpricing",[208],"Anthropic’s pricing"," make the same point in public: compact and frontier models are not the same invoice line. ",[204,3414,421],{"href":419,"rel":3415},[208]," has tracked how fast inference cost and capability moved — which is exactly why “best model” is not a routing policy. Best for a memo is not best for a classify step. Best last quarter is not best this quarter.",[258,3418,261],{"id":260},[263,3420,3421,3427,3433,3439,3450,3460],{},[266,3422,3423,3426],{},[195,3424,3425],{},"Frontier \u002F flagship model."," The strongest (and usually most expensive) model a lab currently sells. Reserved for synthesis, hard reasoning, and novel language. Not for labelling.",[266,3428,3429,3432],{},[195,3430,3431],{},"Compact \u002F small model."," Faster and cheaper. Often enough for extract, classify, and summarise. “Small” is a cost and latency class, not an insult.",[266,3434,3435,3438],{},[195,3436,3437],{},"Task class."," The kind of step: classify, retrieve, forecast, synthesise. If the platform cannot name the class, it cannot route. It can only default.",[266,3440,3441,3444,3445,3449],{},[195,3442,3443],{},"NTU."," A metered unit of useful work so you can quote and cap a loop. See ",[204,3446,3448],{"href":3447},"what-is-ai-token-economics","What is AI token economics",". Seats hide routing. NTU makes it visible.",[266,3451,3452,3455,3456,3459],{},[195,3453,3454],{},"Model-agnostic."," The platform can call more than one provider. That is a menu, not a policy, until it ",[225,3457,3458],{},"chooses"," by task class. Extra logos with a hidden flagship default is lock-in with branding.",[266,3461,3462,3465],{},[195,3463,3464],{},"Always-flagship."," Marketing for “we use the best model.” A classify job does not need a long-context reasoner. Quality theatre is a cost event.",[191,3467,3468,3469,3472,3473,3476],{},"Routing is also not ",[195,3470,3471],{},"fine-tuning"," (changing a model’s weights), and it is not ",[195,3474,3475],{},"orchestration"," (what steps exist). You can orchestrate a brilliant graph and still send every node to the flagship. You can fine-tune a compact model and still need a policy that sends classify there. Evaluate them separately.",[258,3478,3480],{"id":3479},"why-you-should-care","Why you should care",[191,3482,3483],{},"Teams do not wake up and choose waste. They inherit a default.",[263,3485,3486,3492,3498],{},[266,3487,3488,3491],{},[195,3489,3490],{},"Single-model shop."," Every label, every search, every memo calls the same flagship. Finance sees one invoice and cannot split labelling from reasoning. You cannot cap what you cannot see.",[266,3493,3494,3497],{},[195,3495,3496],{},"User-picked dropdown."," Power users pick the most expensive option “to be safe.” New hires copy that habit. Routing is now a training problem. Training problems do not survive quarter-end.",[266,3499,3500,3503],{},[195,3501,3502],{},"Always-flagship as quality theatre."," Best for whom? Best for a forecast interpolation is often a time-series path, not a frontier model inventing a number that looks fine in a short demo.",[191,3505,3506,3510],{},[204,3507,3509],{"href":253,"rel":3508},[208],"McKinsey’s 2025 State of AI survey"," keeps showing the operational gap: regular use, then a struggle to scale because cost and workflow were never treated as a system. Routing is that system for inference. Without it, scale is a token bill.",[191,3512,3513,3518,3519,3521,3522,813,3525,3530],{},[204,3514,3517],{"href":3515,"rel":3516},"https:\u002F\u002Fwww.gartner.com\u002Fen\u002Farticles\u002Fai-governance-trism",[208],"Gartner’s AI TRiSM framing"," implies you can ",[225,3520,2245],{}," which model ran, on which data, at what cost. A platform that cannot show that is not ready, regardless of its red-team slides. ",[204,3523,3241],{"href":3239,"rel":3524},[208],[204,3526,3529],{"href":3527,"rel":3528},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Fdir\u002F2022\u002F2555\u002Foj",[208],"NIS2"," change the evidence question: you should understand ICT dependencies. “We are not sure which model ran last Tuesday” is a dependency you cannot explain.",[191,3532,3533,3534,3536],{},"Choosing a model is choosing a brain. Choosing tools is choosing hands. Decide them separately. A compact model that extracts a refund still cannot write it without a quoted named signer if that is the workstream policy. Routing does not replace ",[204,3535,357],{"href":2273},". Governance does not replace routing. You need both.",[191,3538,3539],{},"Red flags: “we support many providers” with no task-class map; seat pricing that includes unlimited flagship; users pick the model in production; classify and memo share a model id in the demo; no NTU quote before a run expands; fallback is “switch the dropdown”; graph does not record model id per step; forecasting done by an LLM in the demo on a short series.",[258,3541,3543],{"id":3542},"what-to-look-for","What to look for",[263,3545,3546,3552,3558,3568,3574],{},[266,3547,3548,3551],{},[195,3549,3550],{},"They can refuse the flagship."," Run a classify-only job and show the model id. If classify used the same model as the memo, routing is a slide. Refusal is the proof. Support for many models is not.",[266,3553,3554,3557],{},[195,3555,3556],{},"Task-class map you can read."," Classify → compact. Retrieve → embeddings, not stuffing hundreds of tickets into a long window. Forecast → a time-series path, not a frontier model interpolating a spreadsheet. Reason → frontier. If they cannot name the classes, they cannot route them.",[266,3559,3560,3563,3564,303],{},[195,3561,3562],{},"NTU quotes before the run expands."," Operators see an estimate and can set a workstream cap. Seat licences hide routing. Unlimited flagship under a seat is always-flagship with a predictable opex line. See ",[204,3565,3567],{"href":3566},"total-cost-of-ownership-for-enterprise-ai","Total cost of ownership for enterprise AI",[266,3569,3570,3573],{},[195,3571,3572],{},"Fallback is a logged promotion",", not “users will switch the dropdown.” Compact models fail on novel schemas and policy-edge language. Temporarily raise that class, budget-aware, then revert. A promotion without a log is a silent cost change. A dropdown is a training problem.",[266,3575,3576,3579,3580,303],{},[195,3577,3578],{},"The graph records the model id per step."," Six months later you can answer “which model drafted this?” without grepping provider dashboards. That is audit as well as cost. See ",[204,3581,1173],{"href":1172},[191,3583,3584,3585,3588,3589,3591],{},"Ask for four artefacts from one workstream run: model id per step; NTU per task class; a classify job that did ",[225,3586,3587],{},"not"," use the flagship; a forecast that did ",[225,3590,3587],{}," use an LLM as the estimator. If the vendor can only show a chat transcript and a blended token total, routing is not in the product.",[191,3593,3594],{},"Why each artefact matters: model id is Measure in NIST language. NTU per class is how Finance splits labelling from reasoning. Classify-without-flagship is the refusal test. Forecast-without-LLM is whether they know the difference between narration and estimation. Demos are short series. Production is seasonality, holidays, and missing days.",[752,3596,3598],{"id":3597},"what-a-live-demo-should-prove","What a live demo should prove",[191,3600,3601],{},"Do not accept a provider logo wall.",[1082,3603,3604,3610,3613,3616,3619,3622,3625],{},[266,3605,3606,3607,3609],{},"Run one ",[204,3608,3090],{"href":1983}," with extract, classify, retrieve, and a memo.",[266,3611,3612],{},"Show model id per step. Classify and extract are compact. The memo may be frontier.",[266,3614,3615],{},"Show NTU (or tokens) per task class, quoted before the run grows, with a cap on the workstream.",[266,3617,3618],{},"Force a compact failure on a novel schema. Show a logged promotion, then a revert — not a user switching a dropdown.",[266,3620,3621],{},"Show a forecast path that is not an LLM interpolating a sheet. Narration can still be frontier.",[266,3623,3624],{},"Query the graph: which model drafted this payload? Answer from the ledger, not from a provider console.",[266,3626,3627],{},"Confirm a named-role gate still applies regardless of which model drafted. Routing chooses the brain. Governance still decides the write.",[191,3629,3630],{},"If they pass by opening a playground and picking “best,” you evaluated a dropdown.",[258,3632,1415],{"id":1414},[191,3634,3635],{},"Nimbus treats routing as an operating decision tied to workstream steps: task type, sensitivity, and cost — not “best everywhere.” Compact models handle extract. Frontier models are reserved for synthesis. Spend is NTU-metered, quoted per workstream, visible per step.",[191,3637,3638,3639,3642],{},"Release gates apply regardless of which model drafted the payload. Connectors stay read-only by default. Routing decides ",[225,3640,3641],{},"which brain"," reads them. Governance still decides whether anything writes.",[191,3644,616,3645,3647,3648,3650],{},[204,3646,44],{"href":45},". For the unit of account, ",[204,3649,3448],{"href":3447},". Score the four artefacts above. The product claim is the policy, not the catalogue.",[258,3652,750],{"id":749},[752,3654,3656],{"id":3655},"is-model-agnostic-the-same-as-routing","Is “model-agnostic” the same as routing?",[191,3658,3659,3660,3662],{},"No. Model-agnostic means more than one provider. Routing means it ",[195,3661,3458],{}," by task class, with a default that is cheap where cheap is correct. A hidden always-flagship default is lock-in with extra logos. Ask what happens if the operator never touches a dropdown. If the answer is flagship, you have your policy.",[752,3664,3666],{"id":3665},"should-operators-ever-pick-a-model","Should operators ever pick a model?",[191,3668,3669],{},"Rarely, and as an override. Production operators should brief outcomes. If quality depends on each user knowing which model is good at JSON, you have staffed a routing department by accident. Overrides should be logged, budget-aware, and exceptional. A dropdown on every run is how always-flagship returns through the side door.",[752,3671,3673],{"id":3672},"why-not-put-forecasting-in-the-llm-if-the-numbers-look-fine-in-the-demo","Why not put forecasting in the LLM if the numbers look fine in the demo?",[191,3675,3676],{},"Demos are short series. Production is seasonality, holidays, and missing days. Keep narration on the frontier model and estimation on a time-series path. A fluent number is not a control. The warehouse or the statistical path already owns the number. RAG plus a frontier model is for policy language, not for revenue by region.",[752,3678,3680],{"id":3679},"what-if-legal-requires-a-single-approved-model-vendor","What if legal requires a single approved model vendor?",[191,3682,3683,3684,3687],{},"Routing still applies ",[195,3685,3686],{},"inside"," that vendor’s catalogue: compact vs frontier vs embedding. Single-vendor is a contracting constraint, not an excuse to max tokens. Model-agnostic is nice. Task-class mapping inside one catalogue is the control. Do not skip routing because the RFP named one lab.",[752,3689,3691],{"id":3690},"does-nis2-or-dora-change-the-routing-question","Does NIS2 or DORA change the routing question?",[191,3693,3694,3695,3698],{},"They change the ",[195,3696,3697],{},"evidence"," question. DORA and NIS2 expect you to understand ICT dependencies. “We are not sure which model ran last Tuesday” is a dependency you cannot explain. Record model id per step on the graph. That is enough to start. You do not need a new product category. You need Measure.",[258,3700,807],{"id":806},[191,3702,3703,286,3707,3076,3710,303],{},[204,3704,3706],{"href":3705},"how-to-evaluate-ai-workstream-platforms","How to evaluate AI workstream platforms",[204,3708,3709],{"href":973},"How to evaluate an enterprise AI operating system",[204,3711,3567],{"href":3566},[258,3713,821],{"id":820},[263,3715,3716,3722,3728,3734,3739,3745,3751],{},[266,3717,3718],{},[204,3719,3721],{"href":3404,"rel":3720},[208],"OpenAI API pricing",[266,3723,3724],{},[204,3725,3727],{"href":3409,"rel":3726},[208],"Anthropic pricing",[266,3729,3730],{},[204,3731,3733],{"href":419,"rel":3732},[208],"Stanford HAI, 2025 AI Index Report",[266,3735,3736],{},[204,3737,887],{"href":253,"rel":3738},[208],[266,3740,3741],{},[204,3742,3744],{"href":3515,"rel":3743},[208],"Gartner, AI TRiSM \u002F AI governance",[266,3746,3747],{},[204,3748,3750],{"href":3239,"rel":3749},[208],"DORA (Regulation 2022\u002F2554)",[266,3752,3753],{},[204,3754,3756],{"href":3527,"rel":3755},[208],"NIS2 (Directive 2022\u002F2555)",{"title":171,"searchDepth":172,"depth":172,"links":3758},[3759,3760,3761,3764,3765,3772,3773],{"id":260,"depth":172,"text":261},{"id":3479,"depth":172,"text":3480},{"id":3542,"depth":172,"text":3543,"children":3762},[3763],{"id":3597,"depth":930,"text":3598},{"id":1414,"depth":172,"text":1415},{"id":749,"depth":172,"text":750,"children":3766},[3767,3768,3769,3770,3771],{"id":3655,"depth":930,"text":3656},{"id":3665,"depth":930,"text":3666},{"id":3672,"depth":930,"text":3673},{"id":3679,"depth":930,"text":3680},{"id":3690,"depth":930,"text":3691},{"id":806,"depth":172,"text":807},{"id":820,"depth":172,"text":821},"2026-08-17","Model routing is a policy that uses a cheaper model for simple steps and a stronger model only when the task needs it — not a dropdown labelled “best.”","\u002Fblog\u002Fwhat-to-look-for-in-model-routing",{"title":3386,"description":3775},"blog\u002Fwhat-to-look-for-in-model-routing",[941,622,3780,3781],"routing","cost","CHj_o8nPz97yUKukPMWLIFwyFs4cGj__ST6_gCIWkhg",{"fold":3784,"id":3789,"title":3790,"archived":165,"authors":166,"badge":166,"body":3791,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":3795,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":3799,"relatedHeading":166,"seo":3800,"series":166,"sitemap":165,"status":166,"stem":3801,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":3802},{"headline":3785,"description":3786,"primaryLabel":8,"primaryTo":3787,"secondaryLabel":3788,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Start on Free.","\u002Fsignup?plan=free","Explore the platform","content\u002Fshared\u002Fcta.md","Site CTAs",{"type":168,"value":3792,"toc":3793},[],{"title":171,"searchDepth":172,"depth":172,"links":3794},[],{"headline":3796,"description":3797,"primaryLabel":8,"primaryTo":3787,"secondaryLabel":3798,"secondaryTo":85},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger. Start on Free.","Talk to our team","\u002Fshared\u002Fcta",{"title":3790,"description":171},"shared\u002Fcta","PS2VPJsszmUpMBZT6nEp8cWXCdeiN6zDRl-p8d0uY2k",{"enabled":165,"message":3804,"linkLabel":79,"linkHref":80,"id":3805,"title":3806,"archived":165,"authors":166,"badge":166,"body":3807,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":3811,"relatedHeading":166,"seo":3812,"series":166,"sitemap":165,"status":166,"stem":3813,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":3814},"We're hiring! Join the team building the Sentient Enterprise.","content\u002Fshared\u002Fhiring.md","Hiring banner",{"type":168,"value":3808,"toc":3809},[],{"title":171,"searchDepth":172,"depth":172,"links":3810},[],"\u002Fshared\u002Fhiring",{"title":3806,"description":171},"shared\u002Fhiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",1790215700112]