[{"data":1,"prerenderedAt":1377},["ShallowReactive",2],{"site-nav-content":3,"blog:\u002Fblog\u002Fwhat-to-look-for-in-model-routing":179,"blog-index-copy":623,"blog:\u002Fblog\u002Fwhat-to-look-for-in-model-routing:surround":644,"hiring-banner-content":1346,"site-cta-content":1358},{"header":4,"productNav":9,"nav":42,"footer":61,"askAI":132,"id":163,"title":164,"archived":165,"authors":166,"badge":166,"body":167,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":175,"relatedHeading":166,"seo":176,"series":166,"sitemap":165,"status":166,"stem":177,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":178},{"productLabel":5,"loginLabel":6,"contactLabel":7,"contactSalesLabel":8},"Product","Log in","Contact","Get started for free",[10,14,18,22,26,30,34,38],{"label":11,"to":12,"description":13},"Overview","\u002Foverview","Seven layers. One closed loop.",{"label":15,"to":16,"description":17},"Conflux","\u002Fproduct\u002Fconflux","Where your team, workstreams, and agents meet.",{"label":19,"to":20,"description":21},"Agent Teams","\u002Fproduct\u002Fagent-teams","Specialist teams - governed from day one.",{"label":23,"to":24,"description":25},"Lifecycle Graph","\u002Fproduct\u002Flifecycle-graph","Intelligence that compounds across every interaction.",{"label":27,"to":28,"description":29},"Company Wiki","\u002Fproduct\u002Fwiki","Playbooks and policies where expertise stays.",{"label":31,"to":32,"description":33},"Workstreams","\u002Fproduct\u002Fworkstreams","From brief to signed-off deliverable on one canvas.",{"label":35,"to":36,"description":37},"Perception Console","\u002Fproduct\u002Fperception","Ask your whole business in plain English.",{"label":39,"to":40,"description":41},"Governance","\u002Fproduct\u002Fgovernance","Frontier AI you can actually sign off on.",[43,46,49,52,55,58],{"label":44,"to":45},"Models","\u002Fmodels",{"label":47,"to":48},"Pricing","\u002Fpricing",{"label":50,"to":51},"Integrations","\u002Fintegrations",{"label":53,"to":54},"Security","\u002Fsecurity",{"label":56,"to":57},"Partners","\u002Fpartners",{"label":59,"to":60},"Insights","\u002Fblog",{"productHeading":5,"companyHeading":62,"resourcesHeading":63,"legalHeading":64,"docsLabel":65,"docsUrl":66,"statementLines":67,"copyright":70,"companyLinks":71,"resourcesLinks":86,"legalLinks":102,"socialLinks":109,"bottomLinks":119},"Company","Resources","Legal","Docs","https:\u002F\u002Fdocs.gonimbus.ai",[68,69],"Stop training someone else's model.","Control your AI.","© 2026 Nimbus Intelligence, Inc. All rights reserved.",[72,73,74,75,76,78,81,84],{"label":47,"to":48},{"label":50,"to":51},{"label":53,"to":54},{"label":59,"to":60},{"label":77,"to":57},"Partner Program",{"label":79,"to":80},"Careers","\u002Fcareers",{"label":82,"to":83},"System status","\u002Fstatus",{"label":7,"to":85},"\u002Fcontact",[87,90,93,96,99],{"label":88,"to":89},"Glossary","\u002Fglossary",{"label":91,"to":92},"Compare","\u002Fcompare",{"label":94,"to":95},"Evaluate","\u002Fevaluate",{"label":97,"to":98},"Problems","\u002Fproblems",{"label":100,"to":101},"Use cases","\u002Fuse-cases",[103,106],{"label":104,"to":105},"Terms of Service","\u002Fterms",{"label":107,"to":108},"Privacy Policy","\u002Fprivacy",[110,113,116],{"label":111,"href":112},"LinkedIn","https:\u002F\u002Fwww.linkedin.com\u002Fcompany\u002Fgonimbusai\u002F",{"label":114,"href":115},"X","https:\u002F\u002Fx.com\u002Fgonimbusai",{"label":117,"href":118},"Instagram","https:\u002F\u002Fwww.instagram.com\u002Fgonimbus_ai\u002F",[120,122,124,127,128],{"label":121,"to":105},"Terms",{"label":123,"to":108},"Privacy",{"label":125,"to":126},"Compliance","\u002Fcompliance",{"label":82,"to":83},{"label":129,"to":130,"external":131},"LLMs.txt","\u002Fllms.txt",true,{"text":133,"prompt":134},"Ask AI about Nimbus",{"I'm researching enterprise intelligence platforms and want to know how Nimbus combines perception, collaboration, and autonomous agents to drive strategic decision-making":135,"platforms":137},{" Summarize the highlights from Nimbus's website":136},"https:\u002F\u002Fgonimbus.ai",[138,143,148,153,158],{"name":139,"label":140,"icon":141,"hrefPrefix":142},"chatgpt","ChatGPT","simple-icons:openai","https:\u002F\u002Fchatgpt.com\u002F?prompt=",{"name":144,"label":145,"icon":146,"hrefPrefix":147},"perplexity","Perplexity","mdi:magnify","https:\u002F\u002Fwww.perplexity.ai\u002Fsearch\u002Fnew?q=",{"name":149,"label":150,"icon":151,"hrefPrefix":152},"grok","Grok","simple-icons:x","https:\u002F\u002Fx.com\u002Fi\u002Fgrok?text=",{"name":154,"label":155,"icon":156,"hrefPrefix":157},"claude","Claude","simple-icons:anthropic","https:\u002F\u002Fclaude.ai\u002Fnew?q=",{"name":159,"label":160,"icon":161,"hrefPrefix":162},"google-ai","Google AI","simple-icons:google","https:\u002F\u002Fwww.google.com\u002Fsearch?udm=50&aep=11&q=","content\u002Fshared\u002Fnav.md","Site navigation",false,null,{"type":168,"value":169,"toc":170},"minimark",[],{"title":171,"searchDepth":172,"depth":172,"links":173},"",2,[],"md","\u002Fshared\u002Fnav",{"title":164,"description":171},"shared\u002Fnav","1dD7ahDRl0SQ4hz53-kKo0tEFrGaLuaztZ3PPfp6a9k",{"id":180,"title":181,"archived":165,"authors":182,"badge":185,"body":187,"date":612,"definedTerm":166,"department":166,"description":613,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":614,"relatedHeading":166,"seo":615,"series":616,"sitemap":131,"status":166,"stem":617,"subhead":166,"tags":618,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":622},"content\u002Fblog\u002Fwhat-to-look-for-in-model-routing.md","What to Look for in Model Routing",[183],{"name":184,"to":136},"Nimbus Research",{"label":186},"Evaluation",{"type":168,"value":188,"toc":594},[189,197,200,222,227,277,288,292,295,315,323,346,354,357,361,402,412,415,420,423,452,455,459,462,469,478,482,486,492,496,499,503,506,510,517,521,528,532,546,550],[190,191,192,196],"p",{},[193,194,195],"strong",{},"Model routing"," is the policy that maps a task class to a model class before inference runs. It is not a brand preference. It is not a dropdown labelled “best.”",[190,198,199],{},"Someone using the flagship model to label a ticket is how you pay frontier prices for work a compact model could have finished in a second. That is not a moral failing. It is a missing policy. The product either chooses before the call, or a person chooses in a menu, or the default is the largest model “for quality.” Only the first is routing.",[190,201,202,209,210,215,216,221],{},[203,204,208],"a",{"href":205,"rel":206},"https:\u002F\u002Fopenai.com\u002Fapi\u002Fpricing\u002F",[207],"nofollow","OpenAI’s API pricing"," and ",[203,211,214],{"href":212,"rel":213},"https:\u002F\u002Fwww.anthropic.com\u002Fpricing",[207],"Anthropic’s pricing"," make the same point in public: compact and frontier models are not the same invoice line. ",[203,217,220],{"href":218,"rel":219},"https:\u002F\u002Fhai.stanford.edu\u002Fai-index\u002F2025-ai-index-report",[207],"Stanford HAI’s 2025 AI Index"," has tracked how fast inference cost and capability moved — which is exactly why “best model” is not a routing policy. Best for a memo is not best for a classify step. Best last quarter is not best this quarter.",[223,224,226],"h2",{"id":225},"words-youll-hear","Words you’ll hear",[228,229,230,237,243,249,260,271],"ul",{},[231,232,233,236],"li",{},[193,234,235],{},"Frontier \u002F flagship model."," The strongest (and usually most expensive) model a lab currently sells. Reserved for synthesis, hard reasoning, and novel language. Not for labelling.",[231,238,239,242],{},[193,240,241],{},"Compact \u002F small model."," Faster and cheaper. Often enough for extract, classify, and summarise. “Small” is a cost and latency class, not an insult.",[231,244,245,248],{},[193,246,247],{},"Task class."," The kind of step: classify, retrieve, forecast, synthesise. If the platform cannot name the class, it cannot route. It can only default.",[231,250,251,254,255,259],{},[193,252,253],{},"NTU."," A metered unit of useful work so you can quote and cap a loop. See ",[203,256,258],{"href":257},"what-is-ai-token-economics","What is AI token economics",". Seats hide routing. NTU makes it visible.",[231,261,262,265,266,270],{},[193,263,264],{},"Model-agnostic."," The platform can call more than one provider. That is a menu, not a policy, until it ",[267,268,269],"em",{},"chooses"," by task class. Extra logos with a hidden flagship default is lock-in with branding.",[231,272,273,276],{},[193,274,275],{},"Always-flagship."," Marketing for “we use the best model.” A classify job does not need a long-context reasoner. Quality theatre is a cost event.",[190,278,279,280,283,284,287],{},"Routing is also not ",[193,281,282],{},"fine-tuning"," (changing a model’s weights), and it is not ",[193,285,286],{},"orchestration"," (what steps exist). You can orchestrate a brilliant graph and still send every node to the flagship. You can fine-tune a compact model and still need a policy that sends classify there. Evaluate them separately.",[223,289,291],{"id":290},"why-you-should-care","Why you should care",[190,293,294],{},"Teams do not wake up and choose waste. They inherit a default.",[228,296,297,303,309],{},[231,298,299,302],{},[193,300,301],{},"Single-model shop."," Every label, every search, every memo calls the same flagship. Finance sees one invoice and cannot split labelling from reasoning. You cannot cap what you cannot see.",[231,304,305,308],{},[193,306,307],{},"User-picked dropdown."," Power users pick the most expensive option “to be safe.” New hires copy that habit. Routing is now a training problem. Training problems do not survive quarter-end.",[231,310,311,314],{},[193,312,313],{},"Always-flagship as quality theatre."," Best for whom? Best for a forecast interpolation is often a time-series path, not a frontier model inventing a number that looks fine in a short demo.",[190,316,317,322],{},[203,318,321],{"href":319,"rel":320},"https:\u002F\u002Fwww.mckinsey.com\u002Fcapabilities\u002Fquantumblack\u002Four-insights\u002Fthe-state-of-ai",[207],"McKinsey’s 2025 State of AI survey"," keeps showing the operational gap: regular use, then a struggle to scale because cost and workflow were never treated as a system. Routing is that system for inference. Without it, scale is a token bill.",[190,324,325,330,331,334,335,209,340,345],{},[203,326,329],{"href":327,"rel":328},"https:\u002F\u002Fwww.gartner.com\u002Fen\u002Farticles\u002Fai-governance-trism",[207],"Gartner’s AI TRiSM framing"," implies you can ",[267,332,333],{},"see"," which model ran, on which data, at what cost. A platform that cannot show that is not ready, regardless of its red-team slides. ",[203,336,339],{"href":337,"rel":338},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2022\u002F2554\u002Foj",[207],"DORA",[203,341,344],{"href":342,"rel":343},"https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Fdir\u002F2022\u002F2555\u002Foj",[207],"NIS2"," change the evidence question: you should understand ICT dependencies. “We are not sure which model ran last Tuesday” is a dependency you cannot explain.",[190,347,348,349,353],{},"Choosing a model is choosing a brain. Choosing tools is choosing hands. Decide them separately. A compact model that extracts a refund still cannot write it without a quoted named signer if that is the workstream policy. Routing does not replace ",[203,350,352],{"href":351},"what-is-ai-governance","governance",". Governance does not replace routing. You need both.",[190,355,356],{},"Red flags: “we support many providers” with no task-class map; seat pricing that includes unlimited flagship; users pick the model in production; classify and memo share a model id in the demo; no NTU quote before a run expands; fallback is “switch the dropdown”; graph does not record model id per step; forecasting done by an LLM in the demo on a short series.",[223,358,360],{"id":359},"what-to-look-for","What to look for",[228,362,363,369,375,386,392],{},[231,364,365,368],{},[193,366,367],{},"They can refuse the flagship."," Run a classify-only job and show the model id. If classify used the same model as the memo, routing is a slide. Refusal is the proof. Support for many models is not.",[231,370,371,374],{},[193,372,373],{},"Task-class map you can read."," Classify → compact. Retrieve → embeddings, not stuffing hundreds of tickets into a long window. Forecast → a time-series path, not a frontier model interpolating a spreadsheet. Reason → frontier. If they cannot name the classes, they cannot route them.",[231,376,377,380,381,385],{},[193,378,379],{},"NTU quotes before the run expands."," Operators see an estimate and can set a workstream cap. Seat licences hide routing. Unlimited flagship under a seat is always-flagship with a predictable opex line. See ",[203,382,384],{"href":383},"total-cost-of-ownership-for-enterprise-ai","Total cost of ownership for enterprise AI",".",[231,387,388,391],{},[193,389,390],{},"Fallback is a logged promotion",", not “users will switch the dropdown.” Compact models fail on novel schemas and policy-edge language. Temporarily raise that class, budget-aware, then revert. A promotion without a log is a silent cost change. A dropdown is a training problem.",[231,393,394,397,398,385],{},[193,395,396],{},"The graph records the model id per step."," Six months later you can answer “which model drafted this?” without grepping provider dashboards. That is audit as well as cost. See ",[203,399,401],{"href":400},"how-to-evaluate-ai-audit-and-observability","How to evaluate AI audit and observability",[190,403,404,405,408,409,411],{},"Ask for four artefacts from one workstream run: model id per step; NTU per task class; a classify job that did ",[267,406,407],{},"not"," use the flagship; a forecast that did ",[267,410,407],{}," use an LLM as the estimator. If the vendor can only show a chat transcript and a blended token total, routing is not in the product.",[190,413,414],{},"Why each artefact matters: model id is Measure in NIST language. NTU per class is how Finance splits labelling from reasoning. Classify-without-flagship is the refusal test. Forecast-without-LLM is whether they know the difference between narration and estimation. Demos are short series. Production is seasonality, holidays, and missing days.",[416,417,419],"h3",{"id":418},"what-a-live-demo-should-prove","What a live demo should prove",[190,421,422],{},"Do not accept a provider logo wall.",[424,425,426,434,437,440,443,446,449],"ol",{},[231,427,428,429,433],{},"Run one ",[203,430,432],{"href":431},"what-is-an-ai-workstream","workstream"," with extract, classify, retrieve, and a memo.",[231,435,436],{},"Show model id per step. Classify and extract are compact. The memo may be frontier.",[231,438,439],{},"Show NTU (or tokens) per task class, quoted before the run grows, with a cap on the workstream.",[231,441,442],{},"Force a compact failure on a novel schema. Show a logged promotion, then a revert — not a user switching a dropdown.",[231,444,445],{},"Show a forecast path that is not an LLM interpolating a sheet. Narration can still be frontier.",[231,447,448],{},"Query the graph: which model drafted this payload? Answer from the ledger, not from a provider console.",[231,450,451],{},"Confirm a named-role gate still applies regardless of which model drafted. Routing chooses the brain. Governance still decides the write.",[190,453,454],{},"If they pass by opening a playground and picking “best,” you evaluated a dropdown.",[223,456,458],{"id":457},"how-this-shows-up-in-nimbus","How this shows up in Nimbus",[190,460,461],{},"Nimbus treats routing as an operating decision tied to workstream steps: task type, sensitivity, and cost — not “best everywhere.” Compact models handle extract. Frontier models are reserved for synthesis. Spend is NTU-metered, quoted per workstream, visible per step.",[190,463,464,465,468],{},"Release gates apply regardless of which model drafted the payload. Connectors stay read-only by default. Routing decides ",[267,466,467],{},"which brain"," reads them. Governance still decides whether anything writes.",[190,470,471,472,474,475,477],{},"See ",[203,473,44],{"href":45},". For the unit of account, ",[203,476,258],{"href":257},". Score the four artefacts above. The product claim is the policy, not the catalogue.",[223,479,481],{"id":480},"questions-people-actually-ask","Questions people actually ask",[416,483,485],{"id":484},"is-model-agnostic-the-same-as-routing","Is “model-agnostic” the same as routing?",[190,487,488,489,491],{},"No. Model-agnostic means more than one provider. Routing means it ",[193,490,269],{}," by task class, with a default that is cheap where cheap is correct. A hidden always-flagship default is lock-in with extra logos. Ask what happens if the operator never touches a dropdown. If the answer is flagship, you have your policy.",[416,493,495],{"id":494},"should-operators-ever-pick-a-model","Should operators ever pick a model?",[190,497,498],{},"Rarely, and as an override. Production operators should brief outcomes. If quality depends on each user knowing which model is good at JSON, you have staffed a routing department by accident. Overrides should be logged, budget-aware, and exceptional. A dropdown on every run is how always-flagship returns through the side door.",[416,500,502],{"id":501},"why-not-put-forecasting-in-the-llm-if-the-numbers-look-fine-in-the-demo","Why not put forecasting in the LLM if the numbers look fine in the demo?",[190,504,505],{},"Demos are short series. Production is seasonality, holidays, and missing days. Keep narration on the frontier model and estimation on a time-series path. A fluent number is not a control. The warehouse or the statistical path already owns the number. RAG plus a frontier model is for policy language, not for revenue by region.",[416,507,509],{"id":508},"what-if-legal-requires-a-single-approved-model-vendor","What if legal requires a single approved model vendor?",[190,511,512,513,516],{},"Routing still applies ",[193,514,515],{},"inside"," that vendor’s catalogue: compact vs frontier vs embedding. Single-vendor is a contracting constraint, not an excuse to max tokens. Model-agnostic is nice. Task-class mapping inside one catalogue is the control. Do not skip routing because the RFP named one lab.",[416,518,520],{"id":519},"does-nis2-or-dora-change-the-routing-question","Does NIS2 or DORA change the routing question?",[190,522,523,524,527],{},"They change the ",[193,525,526],{},"evidence"," question. DORA and NIS2 expect you to understand ICT dependencies. “We are not sure which model ran last Tuesday” is a dependency you cannot explain. Record model id per step on the graph. That is enough to start. You do not need a new product category. You need Measure.",[223,529,531],{"id":530},"related-reading","Related reading",[190,533,534,538,539,543,544,385],{},[203,535,537],{"href":536},"how-to-evaluate-ai-workstream-platforms","How to evaluate AI workstream platforms",", ",[203,540,542],{"href":541},"how-to-evaluate-an-enterprise-ai-operating-system","How to evaluate an enterprise AI operating system",", and ",[203,545,384],{"href":383},[223,547,549],{"id":548},"sources","Sources",[228,551,552,558,564,570,576,582,588],{},[231,553,554],{},[203,555,557],{"href":205,"rel":556},[207],"OpenAI API pricing",[231,559,560],{},[203,561,563],{"href":212,"rel":562},[207],"Anthropic pricing",[231,565,566],{},[203,567,569],{"href":218,"rel":568},[207],"Stanford HAI, 2025 AI Index Report",[231,571,572],{},[203,573,575],{"href":319,"rel":574},[207],"McKinsey, The state of AI in 2025",[231,577,578],{},[203,579,581],{"href":327,"rel":580},[207],"Gartner, AI TRiSM \u002F AI governance",[231,583,584],{},[203,585,587],{"href":337,"rel":586},[207],"DORA (Regulation 2022\u002F2554)",[231,589,590],{},[203,591,593],{"href":342,"rel":592},[207],"NIS2 (Directive 2022\u002F2555)",{"title":171,"searchDepth":172,"depth":172,"links":595},[596,597,598,602,603,610,611],{"id":225,"depth":172,"text":226},{"id":290,"depth":172,"text":291},{"id":359,"depth":172,"text":360,"children":599},[600],{"id":418,"depth":601,"text":419},3,{"id":457,"depth":172,"text":458},{"id":480,"depth":172,"text":481,"children":604},[605,606,607,608,609],{"id":484,"depth":601,"text":485},{"id":494,"depth":601,"text":495},{"id":501,"depth":601,"text":502},{"id":508,"depth":601,"text":509},{"id":519,"depth":601,"text":520},{"id":530,"depth":172,"text":531},{"id":548,"depth":172,"text":549},"2026-08-17","Model routing is a policy that uses a cheaper model for simple steps and a stronger model only when the task needs it — not a dropdown labelled “best.”","\u002Fblog\u002Fwhat-to-look-for-in-model-routing",{"title":181,"description":613},"evaluation","blog\u002Fwhat-to-look-for-in-model-routing",[616,619,620,621],"models","routing","cost","CHj_o8nPz97yUKukPMWLIFwyFs4cGj__ST6_gCIWkhg",{"hero":624,"id":626,"title":627,"archived":165,"authors":166,"badge":166,"body":628,"date":166,"definedTerm":166,"department":166,"description":632,"extension":174,"eyebrow":633,"faqHeader":166,"faqs":166,"footerBand":634,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":60,"relatedHeading":640,"seo":641,"series":166,"sitemap":131,"status":166,"stem":642,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":643},{"filename":625},"u2221455217_Flat_design_of_a_futuristic_minimalist_landscape__5d589295-cdea-4ea9-a262-be766881accf_1.png","content\u002Fblog\u002Findex.md","Exploring the future of intelligence.",{"type":168,"value":629,"toc":630},[],{"title":171,"searchDepth":172,"depth":172,"links":631},[],"Deep dives into pre-cognitive intelligence, sentient enterprises, and the evolving landscape of AI-driven business transformation.","Latest Research",{"headline":635,"description":636,"primaryLabel":637,"primaryTo":638,"secondaryLabel":639,"secondaryTo":12},"Stay at the frontier.","Subscribe for product updates and new insights.","Subscribe","\u002Fnewsletter","Explore the platform","More research",{"title":627,"description":632},"blog\u002Findex","BFSWGYO9bcTlaulivKYWyg08_DJHsdGg3OC6g_CG1Hw",[166,645],{"id":646,"title":647,"archived":165,"authors":648,"badge":650,"body":651,"date":1337,"definedTerm":166,"department":166,"description":1338,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":165,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1339,"relatedHeading":166,"seo":1340,"series":616,"sitemap":131,"status":166,"stem":1341,"subhead":166,"tags":1342,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1345},"content\u002Fblog\u002Fhow-to-evaluate-an-agent-harness.md","How to Evaluate an Agent Harness",[649],{"name":184,"to":136},{"label":186},{"type":168,"value":652,"toc":1318},[653,661,681,700,702,774,783,787,790,824,836,839,843,858,877,893,904,919,931,946,957,968,972,998,1010,1014,1020,1029,1032,1036,1039,1048,1058,1069,1080,1089,1097,1105,1119,1126,1129,1137,1139,1161,1163,1167,1170,1174,1179,1183,1186,1190,1193,1197,1213,1215,1225,1227],[190,654,655,656,660],{},"Evaluating an ",[203,657,659],{"href":658},"what-is-an-agent-harness","agent harness"," is checking whether the runtime around the model can finish a job under a stop you trust — not whether a demo answered a question.",[190,662,663,668,669,672,673,676,677,385],{},[203,664,667],{"href":665,"rel":666},"https:\u002F\u002Fdocs.langchain.com\u002Foss\u002Fpython\u002Flangchain\u002Fagents",[207],"LangChain"," defines the object: Agent = Model + Harness. The scoring sheet is therefore about the harness. If your RFP starts with context-window size and SWE-bench, you are scoring a model (and maybe an inner coding loop). You will miss whether an unsigned Salesforce PATCH is possible. ",[203,670,671],{"href":541},"How to evaluate an enterprise AI OS"," is the cousin sheet for wiki, workstreams, and routing as a ",[267,674,675],{},"product category",". This page is the runtime tests that apply to Claude Code, a LangGraph deployment, AIP, Agentforce, and Nimbus alike — then specialised by ",[203,678,680],{"href":679},"inner-vs-outer-agent-harness","inner vs outer",[190,682,683,687,688,693,694,699],{},[203,684,686],{"href":319,"rel":685},[207],"McKinsey’s 2025 State of AI"," already measured the trap: widespread use, limited scale. A fluent demo produces the first. A harness that can refuse, replay, and ratchet produces the second. ",[203,689,692],{"href":690,"rel":691},"https:\u002F\u002Fwww.nist.gov\u002Fitl\u002Fai-risk-management-framework\u002Fnist-ai-rmf-playbook",[207],"NIST’s AI RMF Playbook"," is the measurement language. ",[203,695,698],{"href":696,"rel":697},"https:\u002F\u002Fwww.iso.org\u002Fstandard\u002F42001",[207],"ISO\u002FIEC 42001"," is the management-system language. Neither is “the model seemed careful.”",[223,701,226],{"id":225},[228,703,704,715,733,745,755,765],{},[231,705,706,709,710,714],{},[193,707,708],{},"Harness vs framework."," Library versus running loop. ",[203,711,713],{"href":712},"agent-harness-vs-agent-framework","Harness vs framework",". “We use LangChain” is not a passed test.",[231,716,717,720,721,726,727,732],{},[193,718,719],{},"Sensor."," Independent check. ",[203,722,725],{"href":723,"rel":724},"https:\u002F\u002Fmartinfowler.com\u002Farticles\u002Fharness-engineering.html",[207],"Böckeler","; ",[203,728,731],{"href":729,"rel":730},"https:\u002F\u002Fwww.thoughtworks.com\u002Fen-us\u002Finsights\u002Fblog\u002Fgenerative-ai\u002Fharness-engineering-agent-feedback-exploring-ai-coding-sensors",[207],"Thoughtworks",". Inner: tests. Outer: quote vs SoR.",[231,734,735,738,739,744],{},[193,736,737],{},"Hook \u002F interceptor."," Always runs. ",[203,740,743],{"href":741,"rel":742},"https:\u002F\u002Fcode.claude.com\u002Fdocs\u002Fen\u002Fhooks",[207],"Claude Code hooks",". Outer: fail-closed adapter.",[231,746,747,750,751,385],{},[193,748,749],{},"Quote."," Structured payload, not a paragraph. ",[203,752,754],{"href":753},"what-is-write-back-governance","Write-back",[231,756,757,760,761,385],{},[193,758,759],{},"Replay."," Can you reconstruct signer, policy version, tool grants. ",[203,762,764],{"href":763},"what-is-a-lifecycle-graph","Lifecycle graph",[231,766,767,770,771,385],{},[193,768,769],{},"Model portability."," Swap weights without rewriting tools. Not a logo on a slide. ",[203,772,195],{"href":773},"what-is-model-routing",[190,775,776,777,209,780,782],{},"When you evaluate Nimbus, run these tests on ",[203,778,779],{"href":32},"workstreams",[203,781,352],{"href":40},", not on a homepage video. When you evaluate Claude Code, run them on a repo hook and CI, not on a blog SWE-bench screenshot. Same sheet, different workspace.",[223,784,786],{"id":785},"why-evaluation-usually-fails","Why evaluation usually fails",[190,788,789],{},"People score agents like they score chat: quality of the paragraph, latency, brand of the model. That produces three false passes:",[424,791,792,802,812],{},[231,793,794,797,798,385],{},[193,795,796],{},"The copilot pass."," SSO, a usage dashboard, a good answer. No loop ownership. ",[203,799,801],{"href":800},"how-to-choose-between-a-copilot-and-a-work-os","Copilot vs work OS",[231,803,804,807,808,385],{},[193,805,806],{},"The benchmark pass."," SWE-bench or Terminal-Bench for an outer job. Inner eval, outer purchase. ",[203,809,811],{"href":810},"eval-loops-for-enterprise-agent-harnesses","Eval loops",[231,813,814,817,818,823],{},[193,815,816],{},"The framework pass."," A graph in a notebook with every production tool attached. ",[203,819,822],{"href":820,"rel":821},"https:\u002F\u002Fgenai.owasp.org\u002Fllm-top-10\u002F",[207],"OWASP"," excessive agency with extra nodes.",[190,825,826,831,832,385],{},[203,827,830],{"href":828,"rel":829},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents",[207],"Anthropic"," is blunt: encode the job, bound the tools, define done. Your proof of value should force those three. Written answers without a failed action are still a slide. ",[203,833,835],{"href":834},"how-to-run-an-enterprise-ai-proof-of-value","How to run an enterprise AI proof of value",[190,837,838],{},"Red flags: chathe entire proof; “we integrate” with no scoped grant; governance as PDF; memory as a long window; “model-agnostic” with a flagship default and seat pricing; MCP write tools that inherit a god service account; vendor database offered as the new system of record.",[223,840,842],{"id":841},"checklist","Checklist",[190,844,845,848,849,853,854,385],{},[193,846,847],{},"1. Can it stop an action the model wants?"," Inner: ",[850,851,852],"code",{},"PreToolUse"," denies a matched command; tests fail the merge. Outer: unsigned write does not execute; reject leaves SoR unchanged. If the only stop is max tokens, you have a fuse, not a control plane. ",[203,855,857],{"href":856},"human-in-the-loop-approval-architecture","HITL architecture",[190,859,860,861,866,867,872,873,876],{},"Why this matters: ",[203,862,865],{"href":863,"rel":864},"https:\u002F\u002Fwww.cbc.ca\u002Fnews\u002Fcanada\u002Fbritish-columbia\u002Fair-canada-chatbot-lawsuit-1.7116416",[207],"Air Canada"," and the ",[203,868,871],{"href":869,"rel":870},"https:\u002F\u002Fwww.reuters.com\u002Flegal\u002Fnew-york-lawyers-sanctioned-using-fake-chatgpt-cases-legal-brief-2023-06-22\u002F",[207],"sanctioned ChatGPT brief"," are ungated generation reaching a record. Your demo must show a ",[267,874,875],{},"failed"," write.",[190,878,879,882,883,886,887,889,890,892],{},[193,880,881],{},"2. Can you replay who signed and which harness version ran?"," Signer identity, wiki or ",[850,884,885],{},"AGENTS.md"," revision, tool grants, payload hash, model class. If the answer is Slack search or “the transcript,” you do not have a ledger. ",[203,888,401],{"href":400},". Nimbus’s ",[203,891,23],{"href":24}," is one implementation; demand the export without a vendor engineer.",[190,894,895,898,899,903],{},[193,896,897],{},"3. Can you swap the model without rewriting tools?"," Change compact vs frontier on extract vs judgement. If tools are bound to one vendor’s function-calling dialect in application code with no adapter, portability is a hope. ",[203,900,902],{"href":665,"rel":901},[207],"LangChain’s model interface"," exists for this; product harnesses must expose it as policy, not as a rewrite.",[190,905,906,909,910,914,915,385],{},[193,907,908],{},"4. Are tools grants or a belt?"," Least privilege per job. Missing Salesforce is a configuration error, not a hallucination. ",[203,911,913],{"href":912},"connector-and-permissions-architecture","Connector architecture",". MCP servers inherit the same grant. ",[203,916,918],{"href":917},"mcp-for-enterprise-integrations","MCP for enterprise",[190,920,921,924,925,930],{},[193,922,923],{},"5. Is verification outside the generator?"," Inner: CI the agent cannot mark skip without a hook. Outer: schema of the quote; SoR row matches. Anthropic’s ",[203,926,929],{"href":927,"rel":928},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-harnesses-for-long-running-agents",[207],"long-running harness"," refuses “premature victory” by forcing artefacts and tests. Steal that instinct.",[190,932,933,936,937,941,942,385],{},[193,934,935],{},"6. Can an operator add a sensor without a six-month SOW?"," ",[203,938,940],{"href":939},"what-is-harness-engineering","Harness engineering"," is a ratchet. If only vendor FDE can add a gate, you bought a programme. Fine for AIP-scale. Wrong for a standard CRM field this quarter. ",[203,943,945],{"href":944},"self-service-vs-forward-deployed-ai-platforms","Self-service vs FDE",[190,947,948,951,952,956],{},[193,949,950],{},"7. Is the workspace the job you are buying?"," Repo vs company. ",[203,953,955],{"href":954},"how-to-choose-between-a-coding-harness-and-an-enterprise-harness","How to choose coding vs enterprise",". A single scoring sheet with no workspace column will buy the wrong loop.",[190,958,959,962,963,967],{},[193,960,961],{},"8. Economics of the loop."," Max steps, spend cap, routing. Seat “unlimited” is often always-flagship. ",[203,964,966],{"href":965},"ai-cost-control-architecture","AI cost control architecture",". Ask for a per-step model breakdown on a live run.",[416,969,971],{"id":970},"rfp-questions","RFP questions",[424,973,974,977,980,983,986,989,992,995],{},[231,975,976],{},"Show an action the model attempted that the harness refused. What fired?",[231,978,979],{},"After a successful write (or merge), show the signer, policy version, and payload (or diff) without Slack.",[231,981,982],{},"Change the model on extract this week. Which tools broke?",[231,984,985],{},"Attach a connector (or repo permission) as an operator, not as SE. Time?",[231,987,988],{},"Detach the grant mid-job. Does the write fail closed?",[231,990,991],{},"What is the independent sensor for “done”? Who can mark skip?",[231,993,994],{},"Two departments, different scopes, one job — or one god toolbox?",[231,996,997],{},"Price: seats, tokens, NTUs, or a services quote? What stops flagship on classify?",[190,999,1000,1001,1004,1005,1009],{},"Put these in the RFP, then run them in a ",[203,1002,1003],{"href":834},"PoV",". ",[203,1006,1008],{"href":1007},"rfp-questions-for-enterprise-ai-agents","RFP questions for enterprise AI agents"," overlaps; keep both. Agents without a harness test are a persona list.",[416,1011,1013],{"id":1012},"proof-of-value-short","Proof of value (short)",[190,1015,1016,1019],{},[193,1017,1018],{},"Inner job:"," real repo, required hook, red test the agent must fix, no production SoR token.",[190,1021,1022,1025,1026,1028],{},[193,1023,1024],{},"Outer job:"," real cross-department write, quoted payload, reject path, export. Nimbus should pass the same live sequence as anyone else: OAuth attach, blocked unsigned write, graph export. ",[203,1027,11],{"href":12}," is not the proof.",[190,1030,1031],{},"Skip any refuse\u002Freplay\u002Fswap and you evaluated a chat product, a benchmark, or a framework notebook.",[223,1033,1035],{"id":1034},"score-inner-and-outer-without-mixing-oracles","Score inner and outer without mixing oracles",[190,1037,1038],{},"Run two short scripts. Do not average them into one “AI score.”",[190,1040,1041,1044,1045,1047],{},[193,1042,1043],{},"Inner script (repo)."," Fresh checkout of a service you own. Required hook: deny a dangerous bash pattern. Agent must add a failing test then make it pass. CI is the merge sensor. No production CRM token in the environment. Record: did the hook fire, did CI stay independent, can you show the ",[850,1046,885],{}," revision. SWE-bench plots from the vendor are background, not this script.",[190,1049,1050,1053,1054,1057],{},[193,1051,1052],{},"Outer script (SoR)."," Sandbox Salesforce or equivalent. Operator (not SE) attaches OAuth. Model proposes a write. Unsigned path must fail. Reject path must leave records unchanged. Approve path: read-back matches hash. Export signer and wiki revision. Detach the connector and retry the write — must fail closed. ",[203,1055,1056],{"href":834},"Proof of value"," is this script with two departments on the canvas.",[190,1059,1060,1061,209,1063,1065,1066,385],{},"If a vendor refuses to run the outer script because “we are a coding tool,” believe them and buy them for inner only. If a vendor refuses the inner script because “we are an OS,” believe them and do not replace Cursor. If a vendor claims both and fails one script, you have a category error in their marketing. Nimbus should pass the outer script on ",[203,1062,779],{"href":32},[203,1064,352],{"href":40},". Claude Code should pass the inner script. ",[203,1067,1068],{"href":954},"How to choose",[190,1070,1071,1074,1075,1079],{},[193,1072,1073],{},"Thoughtworks’ layer check."," After the scripts, ask where layer 4 lives: who owns the policy when the agent did what it was allowed to do and harm still happened. If the answer is a steering committee with no interceptor, you evaluated theatre. ",[203,1076,1078],{"href":696,"rel":1077},[207],"ISO 42001"," will not save a missing refuse.",[190,1081,1082,1085,1086,385],{},[193,1083,1084],{},"Economics check."," Pull one live run’s step list: model class per step, tokens or NTUs, which sensor fired. Always-flagship with no cap is a failed harness eval even if the paragraph was good. ",[203,1087,1088],{"href":965},"Cost control",[190,1090,1091,1094,1095,385],{},[193,1092,1093],{},"MCP check."," One write-capable server. Which workspaces may use it. If the answer is “any host that can see the URL,” fail. ",[203,1096,918],{"href":917},[190,1098,1099,1100,1104],{},"Weight the eight checklist items; do not add a ninth called “brand.” ",[203,1101,1103],{"href":218,"rel":1102},[207],"Stanford AI Index"," is useful context for how fast coding tools moved. It is not a substitute for the outer script.",[190,1106,1107,1108,1113,1114,1118],{},"Score vendorsystems, not as essays. A beautiful ",[203,1109,1112],{"href":1110,"rel":1111},"https:\u002F\u002Fwww.langchain.com\u002Fblog\u002Fthe-anatomy-of-an-agent-harness",[207],"anatomy post"," does not pass the refuse test. A messy UI that blocks the unsigned PATCH does. Watch for “evaluation theatre”: the SE runs the happy path, the fail path is “we’ll configure that in phase two,” the ledger is a screenshot of LangSmith. Phase two is where ",[203,1115,1117],{"href":319,"rel":1116},[207],"McKinsey"," pilots go to die.",[190,1120,1121,1122,1125],{},"Bring your own oracle. For inner: a test the agent did not write. For outer: a sandbox row you control. If the vendor must supply the only success criterion, you are scoring their demo fixtures. Terminal-Bench’s strength is that the ",[267,1123,1124],{},"environment"," is the grader. Copy that.",[190,1127,1128],{},"People on the bake-off: an operator who will live in the product, someone who owns the SoR, someone who can say no for Legal, an engineer who will keep the inner harness. If only the vendor and an innovation lead attend, you will buy a narrative. Nimbus, AIP, Cursor, and a LangGraph SOW should all survive that room or be narrowed to the job they actually do.",[190,1130,1131,1132,1136],{},"Write the pass\u002Ffail before the demo so the SE cannot redefine success live. “Blocked unsigned write” is a boolean. “Felt enterprise-ready” is not. Record the session. If they cannot fail on camera, assume they cannot fail in production. ",[203,1133,1135],{"href":690,"rel":1134},[207],"NIST Playbook"," language helps here: you are Measuring a control, not a vibe.",[223,1138,458],{"id":457},[190,1140,1141,1142,1146,1147,1150,1151,1154,1155,1157,1158,1160],{},"Nimbus is an ",[203,1143,1145],{"href":1144},"what-is-an-enterprise-agent-harness","enterprise \u002F outer harness",": ",[203,1148,1149],{"href":28},"wiki"," as guides, connectors as grants, ",[203,1152,1153],{"href":20},"teams"," as the hiring object, ",[203,1156,352],{"href":40}," as the interceptor, graph as replay, ",[203,1159,619],{"href":45}," as routing. Score those surfaces against the eight tests. Do not accept “we are a harness” as a substitute for a failed write. AIP and Agentforce deserve the same eight.",[223,1162,481],{"id":480},[416,1164,1166],{"id":1165},"can-we-score-claude-code-and-nimbus-on-one-spreadsheet","Can we score Claude Code and Nimbus on one spreadsheet?",[190,1168,1169],{},"Yes, with a workspace column. Shared rows: refuse, replay, swap, sensors, operator change, economics. Inner-only rows: tests, sandbox, PR. Outer-only rows: SoR quote, roster signer, workstream isolation.",[416,1171,1173],{"id":1172},"the-vendor-sent-a-swe-bench-plot","The vendor sent a SWE-bench plot.",[190,1175,1176,1177,385],{},"File it under inner quality. If you are buying CRM writes, it is not sufficient. ",[203,1178,811],{"href":810},[416,1180,1182],{"id":1181},"we-already-completed-a-copilot-rfp","We already completed a copilot RFP.",[190,1184,1185],{},"Keep it for personal tools. This sheet is for loops that act. Different job.",[416,1187,1189],{"id":1188},"is-iso-42001-certification-the-eval","Is ISO 42001 certification the eval?",[190,1191,1192],{},"It is a management-system signal. Still watch a write fail. Certification without an interceptor is paperwork.",[416,1194,1196],{"id":1195},"what-should-i-read-next","What should I read next?",[190,1198,1199,1203,1204,1208,1209,1212],{},[203,1200,1202],{"href":1201},"agent-harness-architecture","Agent harness architecture"," to know the parts. ",[203,1205,1207],{"href":1206},"how-to-evaluate-write-back-governance","How to evaluate write-back governance"," for the outer stop in detail. ",[203,1210,1211],{"href":939},"What is harness engineering"," for the ratchet after you buy.",[223,1214,531],{"id":530},[190,1216,1217,209,1221,385],{},[203,1218,1220],{"href":1219},"how-to-evaluate-multi-agent-platforms","How to evaluate multi-agent platforms",[203,1222,1224],{"href":1223},"how-to-evaluate-ai-governance-platforms","How to evaluate AI governance platforms",[223,1226,549],{"id":548},[228,1228,1229,1235,1241,1247,1253,1259,1265,1271,1276,1282,1287,1293,1299,1305,1311],{},[231,1230,1231],{},[203,1232,1234],{"href":665,"rel":1233},[207],"LangChain, Agents",[231,1236,1237],{},[203,1238,1240],{"href":1110,"rel":1239},[207],"LangChain, The anatomy of an agent harness",[231,1242,1243],{},[203,1244,1246],{"href":723,"rel":1245},[207],"Böckeler, Harness engineering for coding agent users",[231,1248,1249],{},[203,1250,1252],{"href":729,"rel":1251},[207],"Thoughtworks, Harness engineering and agent feedback",[231,1254,1255],{},[203,1256,1258],{"href":828,"rel":1257},[207],"Anthropic, Building effective agents",[231,1260,1261],{},[203,1262,1264],{"href":927,"rel":1263},[207],"Anthropic, Effective harnesses for long-running agents",[231,1266,1267],{},[203,1268,1270],{"href":741,"rel":1269},[207],"Claude Code, Hooks",[231,1272,1273],{},[203,1274,575],{"href":319,"rel":1275},[207],[231,1277,1278],{},[203,1279,1281],{"href":690,"rel":1280},[207],"NIST AI RMF Playbook",[231,1283,1284],{},[203,1285,698],{"href":696,"rel":1286},[207],[231,1288,1289],{},[203,1290,1292],{"href":820,"rel":1291},[207],"OWASP Top 10 for LLM applications",[231,1294,1295],{},[203,1296,1298],{"href":863,"rel":1297},[207],"CBC, Air Canada chatbot lawsuit",[231,1300,1301],{},[203,1302,1304],{"href":869,"rel":1303},[207],"Reuters, ChatGPT legal brief sanctions",[231,1306,1307],{},[203,1308,1310],{"href":218,"rel":1309},[207],"Stanford HAI, 2025 AI Index",[231,1312,1313],{},[203,1314,1317],{"href":1315,"rel":1316},"https:\u002F\u002Fmodelcontextprotocol.io\u002Fspecification\u002F2025-11-25\u002Findex",[207],"Model Context Protocol specification",{"title":171,"searchDepth":172,"depth":172,"links":1319},[1320,1321,1322,1326,1327,1328,1335,1336],{"id":225,"depth":172,"text":226},{"id":785,"depth":172,"text":786},{"id":841,"depth":172,"text":842,"children":1323},[1324,1325],{"id":970,"depth":601,"text":971},{"id":1012,"depth":601,"text":1013},{"id":1034,"depth":172,"text":1035},{"id":457,"depth":172,"text":458},{"id":480,"depth":172,"text":481,"children":1329},[1330,1331,1332,1333,1334],{"id":1165,"depth":601,"text":1166},{"id":1172,"depth":601,"text":1173},{"id":1181,"depth":601,"text":1182},{"id":1188,"depth":601,"text":1189},{"id":1195,"depth":601,"text":1196},{"id":530,"depth":172,"text":531},{"id":548,"depth":172,"text":549},"2026-08-24","Evaluating an agent harness means checking whether it can stop a write, replay who signed, swap the model without rewriting tools, and fail a real sensor — not whether the demo answered a question.","\u002Fblog\u002Fhow-to-evaluate-an-agent-harness",{"title":647,"description":1338},"blog\u002Fhow-to-evaluate-an-agent-harness",[616,1343,352,1344],"agent-harness","rfp","1CUE6bav0yO3UQKgSvj4IpjPNqJHOWrGl0WKWVQII7k",{"enabled":165,"message":1347,"linkLabel":79,"linkHref":80,"id":1348,"title":1349,"archived":165,"authors":166,"badge":166,"body":1350,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":166,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1354,"relatedHeading":166,"seo":1355,"series":166,"sitemap":165,"status":166,"stem":1356,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1357},"We're hiring! Join the team building the Sentient Enterprise.","content\u002Fshared\u002Fhiring.md","Hiring banner",{"type":168,"value":1351,"toc":1352},[],{"title":171,"searchDepth":172,"depth":172,"links":1353},[],"\u002Fshared\u002Fhiring",{"title":1349,"description":171},"shared\u002Fhiring","1zs3boivKda1e-b-hAyuNcmZSKjZUAXmecnwHVgcHzk",{"fold":1359,"id":1363,"title":1364,"archived":165,"authors":166,"badge":166,"body":1365,"date":166,"definedTerm":166,"department":166,"description":171,"extension":174,"eyebrow":166,"faqHeader":166,"faqs":166,"footerBand":1369,"headline":166,"image":166,"industry":166,"jobType":166,"listed":131,"location":166,"navigation":131,"openRoles":166,"pageLayout":166,"path":1373,"relatedHeading":166,"seo":1374,"series":166,"sitemap":165,"status":166,"stem":1375,"subhead":166,"tags":166,"video":166,"whyJoin":166,"workplaceType":166,"__hash__":1376},{"headline":1360,"description":1361,"primaryLabel":8,"primaryTo":1362,"secondaryLabel":639,"secondaryTo":12},"Run frontier AI your business actually owns.","Governed agent swarms, 2,000+ integrations, and a knowledge graph that stays inside your walls. Start on Free.","\u002Fsignup?plan=free","content\u002Fshared\u002Fcta.md","Site CTAs",{"type":168,"value":1366,"toc":1367},[],{"title":171,"searchDepth":172,"depth":172,"links":1368},[],{"headline":1370,"description":1371,"primaryLabel":8,"primaryTo":1362,"secondaryLabel":1372,"secondaryTo":85},"See what governed AI looks like on your stack.","Connect your tools, run a workstream, and keep every decision on your ledger. Start on Free.","Talk to our team","\u002Fshared\u002Fcta",{"title":1364,"description":171},"shared\u002Fcta","PS2VPJsszmUpMBZT6nEp8cWXCdeiN6zDRl-p8d0uY2k",1790215701478]