[{"data":1,"prerenderedAt":496},["ShallowReactive",2],{"locale-alternates:\u002Fwhat-is-an-ai-harness":3,"post-\u002Fwhat-is-an-ai-harness":8},{"path":4,"alternates":5},"\u002Fwhat-is-an-ai-harness",{"en":4,"tr":6,"de":7},"\u002Ftr\u002Fai-harness-nedir","\u002Fde\u002Fwas-ist-ein-ki-harness",{"page":9,"translations":349,"nav":355,"related":476,"random":485},{"id":10,"title":11,"body":12,"categories":321,"category":324,"changeHistory":324,"date":325,"description":326,"disclosures":327,"draft":330,"extension":331,"firstLiveAt":324,"image":332,"imageAlt":333,"kind":334,"lang":335,"meta":336,"navigation":337,"omitGermanLocalizationDisclosure":330,"path":4,"publishedAt":324,"readingTime":338,"rights":324,"seo":339,"seoTitle":340,"slug":341,"sources":324,"stem":341,"tags":342,"translationKey":341,"type":347,"updated":324,"__hash__":348},"posts\u002Fwhat-is-an-ai-harness.md","What Is an AI Harness? A Plain-English Explanation",{"type":13,"value":14,"toc":311},"minimark",[15,19,22,30,33,38,41,44,47,50,57,60,85,88,91,95,98,176,183,186,189,193,196,199,206,209,218,222,225,228,231,234,237,240,248,252,255,263,266,269,273],[16,17,18],"p",{},"Imagine asking two products that use the same AI model to complete a task. One tells you how to do it. The other finds the information it needs, performs the action in the relevant software, and checks the result. If employees still have to finish the remaining work, that difference becomes very tangible in day-to-day use.",[16,20,21],{},"“LLMs talk; harnesses get work done.” I use this as a memorable simplification, not a technical definition. An LLM—a large language model—produces a response. A harness turns that response into part of a real task.",[16,23,24,25,29],{},"To understand the difference, look at the operating structure around the model. An ",[26,27,28],"em",{},"AI harness"," is the software layer that determines when the model runs, which instructions it receives, and what information it can use. A more capable harness can also manage tool use, task state, and the steps taken until the work reaches an acceptable result.",[16,31,32],{},"Let me explain the term with a sandwich. Then we can map that kitchen setup to a software product.",[34,35,37],"h2",{"id":36},"ask-a-very-knowledgeable-cook-for-a-sandwich","Ask a very knowledgeable cook for a sandwich",[16,39,40],{},"Imagine a highly knowledgeable cook standing in front of you. You say, “Make me a cheese sandwich.” The cook considers the request and replies: “Take two slices of bread and put cheese between them.”",[16,42,43],{},"You may have received a good recipe. Your plate is still empty.",[16,45,46],{},"That is one way to think about calling a language model on its own. You send the model a request, and it produces an output. The simplest version of the technical flow looks like this:",[16,48,49],{},"Prompt → model → response",[16,51,52,53,56],{},"The ",[26,54,55],{},"prompt"," contains the request and instructions you give the model. The model processes them and produces a response. If the task is to write text, that response may already be a usable output. If something must happen in another system, the product needs an additional connection to make it happen.",[16,58,59],{},"Now give the cook an operating setup: ingredients, kitchen tools, and rules to follow.",[61,62,63,67,70,73,76,79,82],"ol",{},[64,65,66],"li",{},"The task is clear: prepare a cheese sandwich.",[64,68,69],{},"The cook checks the ingredients. Is there bread and cheese? Does the person have an allergy? Which kitchen rules apply?",[64,71,72],{},"The cook selects the necessary tools: a knife, a plate, and a sandwich press if the bread should be toasted.",[64,74,75],{},"The cook prepares the sandwich.",[64,77,78],{},"The cook checks whether the requested ingredients were used and whether the sandwich is ready to serve.",[64,80,81],{},"If something is wrong and can be fixed, the cook fixes it. If a suitable ingredient is unavailable, the cook stops and asks.",[64,83,84],{},"The cook serves the sandwich.",[16,86,87],{},"In this example, the harness is the whole setup that organizes the cook's work. It defines which information to inspect, what can be used, how the work is checked, and when the task is complete.",[16,89,90],{},"The metaphor has a limit: a real cook has hands; a language model does not. In software, the model generates a request for a tool to perform an action. The tool—run by the application or provider—performs that action. A model saying “I made the sandwich” is not evidence that the sandwich is ready.",[34,92,94],{"id":93},"the-model-generates-the-harness-organizes-the-work","The model generates; the harness organizes the work",[16,96,97],{},"Separating the parts this way makes the terms easier to follow:",[99,100,101,117],"table",{},[102,103,104],"thead",{},[105,106,107,111,114],"tr",{},[108,109,110],"th",{},"Part",[108,112,113],{},"What does it do?",[108,115,116],{},"What is it in the sandwich metaphor?",[118,119,120,132,143,154,165],"tbody",{},[105,121,122,126,129],{},[123,124,125],"td",{},"Model",[123,127,128],{},"Produces a response, plan, or tool call.",[123,130,131],{},"The cook who thinks through and proposes what to do.",[105,133,134,137,140],{},[123,135,136],{},"Harness",[123,138,139],{},"Organizes model calls and the progress of the task.",[123,141,142],{},"The operating setup that brings together instructions, tool access, checks, and stopping conditions.",[105,144,145,148,151],{},[123,146,147],{},"Tools",[123,149,150],{},"Read information or perform an action.",[123,152,153],{},"The knife, plate, and sandwich press. In software, these may be functions that read a file or create a record.",[105,155,156,159,162],{},[123,157,158],{},"State \u002F memory",[123,160,161],{},"Stores task information and progress.",[123,163,164],{},"The order note, allergy information, and record of which preparations are complete.",[105,166,167,170,173],{},[123,168,169],{},"Agent",[123,171,172],{},"The model working within this system to carry out a task.",[123,174,175],{},"The cook continuing the job within a setup equipped with information and tools.",[16,177,178,179,182],{},"This is a practical mental model I use to make the concepts easier to understand. It is not a universal taxonomy, and products do not all draw the boundaries in the same place. Some sources, in particular, use ",[26,180,181],{},"agent"," to describe both the model and the surrounding system.",[16,184,185],{},"The model may be called repeatedly as the work progresses. It interprets incoming information, proposes the next step, or identifies the tool it needs. The harness provides the operating structure in which those decisions can be applied. Rules written in software also limit which steps are possible.",[16,187,188],{},"Memory does not mean that the model remembers everything by itself. The software stores the state of the task and supplies the relevant parts during the next model call. Stored information is only useful if it is presented to the model when needed and the model uses it correctly.",[34,190,192],{"id":191},"a-harness-can-begin-with-a-direct-api-call","A harness can begin with a direct API call",[16,194,195],{},"A harness is not limited to terminal commands, desktop applications, or enterprise agent frameworks. It can sit inside a product, independently of the interface the user sees. Nor does it require a ready-made package.",[16,197,198],{},"Suppose a product accepts two kinds of request: summarizing text and translating it into another language. The product identifies the task type, selects the corresponding system prompt and model, and sends both through the provider's API—the connection the software uses to request the model service. It then returns the result to the user.",[16,200,201,202,205],{},"A ",[26,203,204],{},"system prompt"," is the set of instructions that defines how the model should behave for that task. For a summary, you might tell it to preserve the key information. For a translation, you might tell it not to change the meaning. In the practical sense used in this article, code that selects the system prompt and model is already a minimal prompt-and-model routing harness.",[16,207,208],{},"This system may have no tools, persistent memory, or retry loop yet. It still has a layer that organizes how the model is used. If the user selects the task type in the interface, you do not even need another model to make that routing decision.",[16,210,211,212,217],{},"I examine different forms of model routing and the overhead they can add in ",[213,214,216],"a",{"href":215},"\u002Fhow-to-route-requests-across-multiple-ai-models","How Should You Route Requests Across Multiple AI Models?",". The important point here is that even this simple selection belongs to the operating structure around the model.",[34,219,221],{"id":220},"when-the-work-spans-several-steps","When the work spans several steps",[16,223,224],{},"As the task grows, you can add capabilities to the harness. Bringing together the instructions, documents, and previous results the model should see at a given step is often called context preparation. Connecting tools, storing completed steps, validating the output, and retrying recoverable failures all extend the same operating structure.",[16,226,227],{},"A multi-step system that uses tools might follow this flow:",[16,229,230],{},"Task → inspect context → use tools → execute → check the result → retry when needed → finish",[16,232,233],{},"The system receives the task, inspects the necessary information, uses tools to perform the action, and checks the result. It retries when appropriate, then completes the work. Tool use and execution are often parts of the same step. Reading a document may also require a tool. This sequence is a simple way to understand the work cycle, not a protocol that every harness must follow.",[16,235,236],{},"Checking the result does not always mean asking the same model, “Did you do it correctly?” Software can check whether a file exists. It can read from the relevant system to verify that a record was written to the right place. Deciding whether a piece of text preserves its intended meaning may require human judgment.",[16,238,239],{},"Retries also need limits. If a record was created but the response never arrived, blindly repeating the action could create a duplicate. Before retrying, the harness should check what actually happened. If the task is no longer progressing or the required permission is missing, it should stop and return control to a person.",[16,241,242,243,247],{},"Sometimes fixed software rules choose the next step in this cycle. Sometimes the model interprets new information and makes the choice. That is where ",[213,244,246],{"href":245},"\u002Fwhen-do-you-actually-need-an-ai-agent","the distinction between a workflow and an agent"," becomes important. Using a simple harness does not require handing the entire task to an open-ended agent.",[34,249,251],{"id":250},"look-beyond-the-model-when-evaluating-the-product","Look beyond the model when evaluating the product",[16,253,254],{},"The same cook works differently in a kitchen where the ingredients are prepared than in one where every ingredient has to be found first. Change the checking process, and the speed and result may change too. In software products, the information available to the model, the tools it can access, and the way failures are handled create a similar difference.",[16,256,257,258,262],{},"That is why you should look at the operating structure as well as the model name when considering ",[213,259,261],{"href":260},"\u002Fwhy-the-same-ai-model-produces-different-results","why the same model produces different results across applications",". More tools or a longer loop do not automatically produce a better result. Unnecessary steps can increase latency, cost, and the chance of failure.",[16,264,265],{},"For a summarization task, the right instruction, a suitable model, and a short check may be enough. If the task performs actions across several systems, context, permissions, state tracking, and verification become more important. Where the steps and rules are fully known, existing automation may also be sufficient.",[16,267,268],{},"When evaluating an AI product, choose a concrete task from your own work. Examine which information it uses, which action it actually performs, and how the system determines that the work is complete. Include the amount of correction employees must do afterward. The harness earns its value through the contribution it makes to completing that task with an acceptable result.",[34,270,272],{"id":271},"further-reading","Further reading",[274,275,276,293,302],"ul",{},[64,277,278,288,289,292],{},[213,279,287],{"href":280,"className":281,"rel":283,"target":286},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fbuilding-effective-agents",[282],"dofollow",[284,285],"nofollow","noopener","_blank","Building effective agents",": Provides basic design patterns for direct API use, routing, and models that use tools. It is one provider's architecture guide; it does not establish the broad definition of ",[26,290,291],{},"harness"," used here as an industry standard.",[64,294,295,301],{},[213,296,300],{"href":297,"className":298,"rel":299,"target":286},"https:\u002F\u002Fplatform.claude.com\u002Fdocs\u002Fen\u002Fagents-and-tools\u002Ftool-use\u002Foverview",[282],[284,285],"Tool use with Claude",": Separates the tool call generated by a model from the software that executes the action. It illustrates the difference between tools run by the application and by the provider through the Claude API.",[64,303,304,310],{},[213,305,309],{"href":306,"className":307,"rel":308,"target":286},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-harnesses-for-long-running-agents",[282],[284,285],"Effective harnesses for long-running agents",": Shows why progress records and result checks matter in long-running work. It draws on a web application development example and does not show that the same setup is superior for every task.",{"title":312,"searchDepth":313,"depth":313,"links":314},"",2,[315,316,317,318,319,320],{"id":36,"depth":313,"text":37},{"id":93,"depth":313,"text":94},{"id":191,"depth":313,"text":192},{"id":220,"depth":313,"text":221},{"id":250,"depth":313,"text":251},{"id":271,"depth":313,"text":272},[322,323],"ai","engineering",null,"2026-10-01","What is an AI harness? A sandwich metaphor explains models, tools, state, and agents—from simple API routing to systems that manage real work.",{"aiUse":328,"aiNote":329},"ai-assisted","Evren Bal's LinkedIn draft and conceptual explanation were developed into the Turkish source article with AI assistance. AI also supported source checking, structure, language editing, and localization.",false,"md","\u002Fimages\u002Fhero\u002Fai-harness-kitchen-robot.avif","A small kitchen robot assembles a cheese sandwich beside ingredients, tools, and a checked finished sandwich.","Explainer","en",{},true,9,{"title":11,"description":326},"What Is an AI Harness? A Simple, Practical Explanation","what-is-an-ai-harness",[343,344,345,346],"ai-harness","agent-harness","ai-models","model-routing","post","BLuFb5su8I8vCEEGpoVjcqV-8t10LC_yDSr2-U2Laog",{"en":350,"tr":351,"de":353},{"path":4,"title":11},{"path":6,"title":352},"Nedir bu harness? En basit hâliyle AI harness anlatıyorum",{"path":7,"title":354},"Was ist ein KI-Harness? Einfach erklärt",{"prev":356,"next":324,"others":359,"lucky":475,"readingTime":338},{"path":357,"title":358},"\u002Fai-editorial-disclosure","What an AI Disclosure Tells Readers About the Author’s Contribution",[360,363,366,369,372,373,376,379,382,385,388,391,394,397,400,403,406,408,411,414,417,420,423,426,429,432,435,436,439,442,445,448,451,454,457,460,463,466,469,472],{"path":361,"title":362},"\u002Fai-will-spread-across-turkiyes-businesses","AI in Türkiye: What Adoption Figures Leave Out",{"path":364,"title":365},"\u002Fwhen-ai-changes-work-training-employees-is-not-enough","When AI Changes Work, Training Employees Is Not Enough",{"path":367,"title":368},"\u002Fturkeys-first-real-time-mystery-shopping-reporting","Turkey's First Real-Time Mystery Shopping Reporting Platform",{"path":370,"title":371},"\u002Fwho-captures-ai-productivity-gains","Who Captures the Productivity Gains from AI?",{"path":357,"title":358},{"path":374,"title":375},"\u002Fwordpress-to-nuxt-ai-powered-content-pipeline","From WordPress to Nuxt: Building an AI-Powered Content Pipeline",{"path":377,"title":378},"\u002F1m-impressions-per-month-0-revenue-a-programmatic-seo-post-mortem","1M Impressions per Month, $0 Revenue: A Programmatic SEO Post-Mortem",{"path":380,"title":381},"\u002Fhow-ai-changes-the-experience-gap","Can a Junior Who Uses AI Well Outperform a Senior Expert?",{"path":383,"title":384},"\u002Fhow-employee-built-ai-systems-become-organizational-memory","How Employee-Built AI Systems Become Organizational Memory",{"path":386,"title":387},"\u002Fis-your-data-ready-for-ai","Is Your Data Ready for AI? Start With the Decision It Must Support",{"path":389,"title":390},"\u002Fan-seo-experiment-in-a-low-competition-serp-with-google-maps-and-openai","Building camiler.org: A Programmatic SEO Experiment with Google Maps and OpenAI",{"path":392,"title":393},"\u002Fthe-threshold-collapsed-to-zero","The Threshold Collapsed: What ProductLog Taught Me About Building in Public",{"path":395,"title":396},"\u002Fwriting-with-ai-means-thinking-with-your-archive","Writing with AI Means Thinking About Your Archive, Too",{"path":398,"title":399},"\u002Fthe-ai-productivity-baseline-is-moving-faster-than-we-remember","AI Wasn’t Always This Good. We Just Got Used to It.",{"path":401,"title":402},"\u002Fprotecting-organizational-memory-during-ai-transformation","Protect Organizational Memory During an AI Transformation",{"path":404,"title":405},"\u002Ffrom-rules-to-decisions-the-real-time-sales-intelligence-platform-we-built-at-vanity","Medical Tourism Lead Management: How We Moved from Manual Routing to a Real-Time Sales System",{"path":260,"title":407},"Why the Same AI Model Produces Different Results Across Applications",{"path":409,"title":410},"\u002Fturkey-national-ai-platform-evren","National AI Infrastructure Is More Than a GPU Count",{"path":412,"title":413},"\u002Fai-knowledge-management-from-documents-to-better-decisions","Why Searching Company Knowledge with AI Is Not Enough",{"path":415,"title":416},"\u002Fbeyond-the-bot-lessons-from-building-a-chat-system-for-global-patients","What We Learned Building a Healthcare Chatbot for International Patients",{"path":418,"title":419},"\u002Feu-ai-act-after-risk-classification","EU AI Act: What Should You Do Once You Know the Risk Level?",{"path":421,"title":422},"\u002Fwhen-process-automation-actually-needs-ai","When Does Process Automation Actually Need AI?",{"path":424,"title":425},"\u002Fwhat-ai-can-automate-in-cro-research","What Can AI Actually Automate in CRO Research?",{"path":427,"title":428},"\u002Fai-assisted-migrations-deferred-work","AI Can Make Deferred Work Worth Doing",{"path":430,"title":431},"\u002Fstart-with-the-business-problem-not-the-ai-model","Start With the Business Problem, Not the AI Model",{"path":433,"title":434},"\u002Fis-your-ai-system-delivering-business-results","Is Your AI System Actually Delivering Business Results?",{"path":215,"title":216},{"path":437,"title":438},"\u002Fhow-much-authority-should-ai-have","How Much Authority Should You Give an AI System?",{"path":440,"title":441},"\u002Fproductlog-the-platform-i-built-for-myself-first","ProductLog: The Platform I Built for Myself First",{"path":443,"title":444},"\u002Fredar-ai-powered-summaries-for-kap-disclosures-and-open-sources","Redar: AI-Powered Summaries for KAP Disclosures and Open Sources",{"path":446,"title":447},"\u002Fthe-era-of-the-previous-vibe-coder-begins","The Era of the \"Previous Vibe Coder\" Begins: The Invisibility of Clean Code and the Technical Debt Bill of AI",{"path":449,"title":450},"\u002Fhow-llms-identify-experts","What Makes an LLM Recommend Someone as an Expert?",{"path":452,"title":453},"\u002Fchatbot-data-third-party-ai-provider","What Happens When Chatbot Data Is Sent to a Third-Party AI Provider?",{"path":455,"title":456},"\u002Fif-ai-handles-the-execution-who-sets-the-strategy","If AI Handles the Execution, Who Sets the Strategy?",{"path":458,"title":459},"\u002Fwhen-does-ai-progress-improve-everyday-life","When Does AI Progress Become Progress for People?",{"path":461,"title":462},"\u002Fai-transformation-redesign-work-not-cut-roles","AI Transformation Starts with Redesigning Work",{"path":464,"title":465},"\u002Ffrom-asking-questions-to-delegating-work","From Asking Questions to Delegating Work: What the Codex Study Shows",{"path":467,"title":468},"\u002Fwhy-we-trust-ai-judgments","Why We Trust AI Judgments and What That Fails to Prove",{"path":470,"title":471},"\u002Fgoogle-generative-ai-data-ai-citation-timing","AI Visibility Dropped Before Search: The Google Data That Changed My Theory",{"path":473,"title":474},"\u002Faccessing-know-how-is-not-the-same-as-creating-it","Accessing Know-How Is Not the Same as Creating It",{"path":418,"title":419},[477,479,483],{"path":260,"title":407,"date":478},"2026-09-12",{"path":480,"title":481,"date":482},"\u002Fstripe-openrouter-which-model-for-which-job","Stripe’s OpenRouter Deal: Which Model for Which Job?","2026-09-21",{"path":215,"title":216,"date":484},"2026-09-02",[486,490,492],{"path":487,"title":488,"date":489},"\u002Fbuild-in-public-2-0","Build in Public in the AI Era: What to Share and What to Keep Private","2026-06-28",{"path":437,"title":438,"date":491},"2026-08-30",{"path":493,"title":494,"date":495},"\u002Fthe-job-ai-wont-take-and-the-five-it-prevents","AI Is Reducing Hiring Without Layoffs","2026-06-25",1790828475562]