Панель сбора материалов10.09.2026
← К материалу

SOURCE DOCUMENT

Raw HTML

How to evaluate the performance of AI agents? · n8n AI

Оригинал ↗
<!DOCTYPE html>
<html lang="en">
	<head>
		<meta charset="utf-8">
		<meta http-equiv="X-UA-Compatible" content="IE=edge">
			<title>How to evaluate the performance of AI agents? – n8n Blog</title>
		<meta name="HandheldFriendly" content="True">
		<meta name="viewport" content="width=device-width, initial-scale=1">
		<!-- <link rel="preconnect" href="https://fonts.googleapis.com"> 
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link rel="preload" as="style" href="https://fonts.googleapis.com/css2?family=Nunito:ital,wght@0,400;0,800;0,900;1,400;1,800;1,900&display=swap">
<link rel="stylesheet" href="https://fonts.googleapis.com/css2?family=Nunito:ital,wght@0,400;0,800;0,900;1,400;1,800;1,900&display=swap"> -->
		<link rel="stylesheet" type="text/css" href="https://blog.n8n.io/assets/css/screen.css?v=M18h2_Kr5nq7cywd">
		<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.3.0/css/fontawesome.min.css" crossorigin="anonymous" referrerpolicy="no-referrer">
		<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.3.0/css/solid.min.css" crossorigin="anonymous" referrerpolicy="no-referrer">
		
		<script src="https://blog.n8n.io/assets/js/slugify.js?v=pgb4I6LvVtEl3iCf"></script>
		<script src="https://blog.n8n.io/assets/js/workflow-banner.js?v=tSVISzLW2Ja5gfUV"></script>
		<script src="https://blog.n8n.io/assets/js/popular-tags.js?v=X0tYuT5LRchLCeSS"></script>
		
		<!-- Google Tag Manager (blog.n8n.io) -->
		<script>(function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':
		new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0],
		j=d.createElement(s),dl=l!='dataLayer'?'&l='+l:'';j.async=true;j.src=
		'https://www.googletagmanager.com/gtm.js?id='+i+dl;f.parentNode.insertBefore(j,f);
		})(window,document,'script','dataLayer','GTM-MQTZKBJ');</script>
		<!-- End Google Tag Manager (blog.n8n.io) -->
		<meta name="description" content="Evaluate AI agents effectively: learn offline vs. online testing, key metrics, and methods like deterministic checks, LLM-as-a-judge, and human review to improve performance and reliability.">
    <link rel="icon" href="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w256h256/2022/06/logo-n8n-symbol-512x512.png" type="image/png">
    <link rel="canonical" href="https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/">
    <meta name="referrer" content="no-referrer-when-downgrade">
    
    <meta property="og:site_name" content="n8n Blog">
    <meta property="og:type" content="article">
    <meta property="og:title" content="How to evaluate the performance of AI agents?">
    <meta property="og:description" content="Evaluate AI agents effectively: learn offline vs. online testing, key metrics, and methods like deterministic checks, LLM-as-a-judge, and human review to improve performance and reliability.">
    <meta property="og:url" content="https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/">
    <meta property="og:image" content="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg">
    <meta property="article:published_time" content="2026-04-21T17:45:41.000Z">
    <meta property="article:modified_time" content="2026-05-19T10:41:26.000Z">
    <meta property="article:tag" content="AI">
    <meta property="article:tag" content="Tutorial">
    
    <meta property="article:publisher" content="https://www.facebook.com/n8nio">
    <meta name="twitter:card" content="summary_large_image">
    <meta name="twitter:title" content="How to evaluate the performance of AI agents?">
    <meta name="twitter:description" content="Evaluate AI agents effectively: learn offline vs. online testing, key metrics, and methods like deterministic checks, LLM-as-a-judge, and human review to improve performance and reliability.">
    <meta name="twitter:url" content="https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/">
    <meta name="twitter:image" content="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg">
    <meta name="twitter:label1" content="Written by">
    <meta name="twitter:data1" content="Yulia Dmitrievna">
    <meta name="twitter:label2" content="Filed under">
    <meta name="twitter:data2" content="AI, Tutorial">
    <meta name="twitter:site" content="@n8n_io">
    <meta property="og:image:width" content="1200">
    <meta property="og:image:height" content="675">
    
    <script type="application/ld+json">
{
    "@context": "https://schema.org",
    "@type": "Article",
    "publisher": {
        "@type": "Organization",
        "name": "n8n Blog",
        "url": "https://blog.n8n.io/",
        "logo": {
            "@type": "ImageObject",
            "url": "https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2025/09/n8n-Logo.svg",
            "width": 145,
            "height": 26
        }
    },
    "author": {
        "@type": "Person",
        "name": "Yulia Dmitrievna",
        "image": {
            "@type": "ImageObject",
            "url": "https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2023/01/profile-pik.jpg",
            "width": 1024,
            "height": 1024
        },
        "url": "https://blog.n8n.io/author/yulia/",
        "sameAs": []
    },
    "contributor": [
        {
            "@type": "Person",
            "name": "Eduard Parsadanyan",
            "image": {
                "@type": "ImageObject",
                "url": "https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2022/05/new_pic.png"
            },
            "url": "https://blog.n8n.io/author/eduard/",
            "sameAs": [
                "https://zeeg.me/parsadanyan/free-n8n-session"
            ]
        }
    ],
    "headline": "How to evaluate the performance of AI agents?",
    "url": "https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/",
    "datePublished": "2026-04-21T17:45:41.000Z",
    "dateModified": "2026-05-19T10:41:26.000Z",
    "image": {
        "@type": "ImageObject",
        "url": "https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg",
        "width": 1200,
        "height": 675
    },
    "keywords": "AI, Tutorial",
    "description": "Evaluate AI agents effectively: learn offline vs. online testing, key metrics, and methods like deterministic checks, LLM-as-a-judge, and human review to improve performance and reliability.",
    "mainEntityOfPage": "https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/"
}
    </script>

    <meta name="generator" content="Ghost 6.63">
    <link rel="alternate" type="application/rss+xml" title="n8n Blog" href="https://blog.n8n.io/rss/">
    <script defer src="https://cdn.jsdelivr.net/ghost/portal@~2.71/umd/portal.min.js" data-i18n="true" data-ghost="https://blog.n8n.io/" data-key="8626993d9b363277120ea42164" data-api="https://n8n-cloud-blog.ghost.io/ghost/api/content/" data-locale="en" crossorigin="anonymous"></script><style id="gh-members-styles">.gh-post-upgrade-cta-content,
.gh-post-upgrade-cta {
    display: flex;
    flex-direction: column;
    align-items: center;
    font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Oxygen, Ubuntu, Cantarell, 'Open Sans', 'Helvetica Neue', sans-serif;
    text-align: center;
    width: 100%;
    color: #ffffff;
    font-size: 16px;
}

.gh-post-upgrade-cta-content {
    border-radius: 8px;
    padding: 40px 4vw;
}

.gh-post-upgrade-cta h2 {
    color: #ffffff;
    font-size: 28px;
    letter-spacing: -0.2px;
    margin: 0;
    padding: 0;
}

.gh-post-upgrade-cta p {
    margin: 20px 0 0;
    padding: 0;
}

.gh-post-upgrade-cta small {
    font-size: 16px;
    letter-spacing: -0.2px;
}

.gh-post-upgrade-cta a {
    color: #ffffff;
    cursor: pointer;
    font-weight: 500;
    box-shadow: none;
    text-decoration: underline;
}

.gh-post-upgrade-cta a:hover {
    color: #ffffff;
    opacity: 0.8;
    box-shadow: none;
    text-decoration: underline;
}

.gh-post-upgrade-cta a.gh-btn {
    display: block;
    background: #ffffff;
    text-decoration: none;
    margin: 28px 0 0;
    padding: 8px 18px;
    border-radius: 4px;
    font-size: 16px;
    font-weight: 600;
}

.gh-post-upgrade-cta a.gh-btn:hover {
    opacity: 0.92;
}</style>
    <script defer src="https://cdn.jsdelivr.net/ghost/sodo-search@~1.8/umd/sodo-search.min.js" data-key="8626993d9b363277120ea42164" data-styles="https://cdn.jsdelivr.net/ghost/sodo-search@~1.8/umd/main.css" data-sodo-search="https://n8n-cloud-blog.ghost.io/" data-locale="en" crossorigin="anonymous"></script>
    
    <link href="https://blog.n8n.io/webmentions/receive/" rel="webmention">
    <script defer src="/public/cards.min.js?v=H-xYiMJpfVIWffjV"></script>
    <link rel="stylesheet" type="text/css" href="/public/cards.min.css?v=yumsKhOGY54VXmGs">
    <script defer src="/public/ghost-stats.min.js?v=vFcCUf6ZQ0Hyhc8h" data-stringify-payload="false" data-datasource="analytics_events" data-storage="localStorage" data-host="https://blog.n8n.io/.ghost/analytics/api/v1/page_hit"  tb_site_uuid="0d78b34c-0c5f-4975-900e-61d00ccb1c2d" tb_post_uuid="7431ab54-fad3-4e6d-b960-a5918590a1be" tb_post_type="post" tb_member_uuid="undefined" tb_member_status="undefined" tb_gift_link=""></script><style>:root {--ghost-accent-color: #FD8925;}</style>
	</head>
	<body class="post-template tag-ai tag-tutorial">
		
		<!-- Google Tag Manager (noscript) -->
		<noscript><iframe src="https://www.googletagmanager.com/ns.html?id=GTM-MQTZKBJ"
		height="0" width="0" style="display:none;visibility:hidden"></iframe></noscript>
		<!-- End Google Tag Manager (noscript) -->
		<div class="global-wrap">
			<div class="global-content">
				<div class="hero-background-decoration"></div>
				<header class="header-section">
	<div class="header-container">
		<div class="header-navbar">
			<!-- Left Section: Logo + Navigation -->
			<div class="header-left">
				<div class="header-logo">
					<a href="https://blog.n8n.io" class="logo-link">
						<img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2025/09/n8n-Logo.svg" alt="n8n Blog" class="logo-image">
					</a>
				</div>
				<nav class="header-navigation">
					<ul class="nav-list">
						<li class="nav-item"><a href="https://n8n.io/" class="nav-link">n8n home</a></li>
						<li class="nav-item"><a href="/tag/tutorial/" class="nav-link">Tutorials</a></li>
						<li class="nav-item"><a href="/tag/guide/" class="nav-link">Guides</a></li>
						<li class="nav-item"><a href="/tag/tips/" class="nav-link">Tips</a></li>
						<li class="nav-item"><a href="/tag/by-partners/" class="nav-link">By partners</a></li>
						<li class="nav-item"><a href="/tag/news/" class="nav-link">News</a></li>
                        <li class="nav-item"><a href="/tag/interview/" class="nav-link">Interviews</a></li>
                    </ul>
				</nav>
			</div>
			
			<!-- Right Section: GitHub Badge + Sign In + Get Started -->
			<div class="header-right">
				<a href="https://github.com/n8n-io/n8n" class="github-badge" target="_blank" rel="noopener noreferrer">
					<div class="github-icon">
						<svg width="24" height="24" viewBox="0 0 24 24" fill="none" xmlns="http://www.w3.org/2000/svg">
							<path fill-rule="evenodd" clip-rule="evenodd" d="M12 0C5.37 0 0 5.37 0 12C0 17.31 3.435 21.795 8.205 23.385C8.805 23.49 9.03 23.13 9.03 22.815C9.03 22.53 9.015 21.585 9.015 20.58C6 21.135 5.22 19.845 4.98 19.17C4.845 18.825 4.26 17.76 3.75 17.475C3.33 17.25 2.73 16.695 3.735 16.68C4.68 16.665 5.355 17.55 5.58 17.91C6.66 19.725 8.385 19.215 9.075 18.9C9.18 18.12 9.495 17.595 9.84 17.295C7.17 16.995 4.38 15.96 4.38 11.37C4.38 10.065 4.845 8.985 5.61 8.145C5.49 7.845 5.07 6.615 5.73 4.965C5.73 4.965 6.735 4.65 9.03 6.195C9.99 5.925 11.01 5.79 12.03 5.79C13.05 5.79 14.07 5.925 15.03 6.195C17.325 4.635 18.33 4.965 18.33 4.965C18.99 6.615 18.57 7.845 18.45 8.145C19.215 8.985 19.68 10.05 19.68 11.37C19.68 15.975 16.875 16.995 14.205 17.295C14.64 17.67 15.015 18.39 15.015 19.515C15.015 21.12 15 22.41 15 22.815C15 23.13 15.225 23.505 15.825 23.385C18.2072 22.5807 20.2772 21.0497 21.7437 19.0074C23.2101 16.965 23.9993 14.5143 24 12C24 5.37 18.63 0 12 0Z" fill="currentColor"/>
						</svg>
					</div>
					<span class="github-text">Github ★ 162k</span>
				</a>
				<a href="https://app.n8n.cloud/" class="sign-in-link">Sign in</a>
				<a href="https://app.n8n.cloud/register" class="get-started-btn">Get Started</a>
			</div>

			<!-- Mobile Menu Toggle -->
			<div class="mobile-menu-toggle">
				<input id="mobile-toggle" class="mobile-checkbox" type="checkbox">
				<label class="mobile-toggle-label" for="mobile-toggle">
					<span class="hamburger">
						<span class="bar"></span>
						<span class="bar"></span>
						<span class="bar"></span>
					</span>
				</label>
				<!-- Mobile Navigation Menu -->
				<nav class="mobile-navigation">
					<ul class="mobile-nav-list">
						<li class="mobile-nav-item"><a href="https://n8n.io/" class="mobile-nav-link">n8n home</a></li>
						<li class="mobile-nav-item"><a href="/tag/tutorial/" class="mobile-nav-link">Tutorials</a></li>
						<li class="mobile-nav-item"><a href="/tag/guide/" class="mobile-nav-link">Guides</a></li>
						<li class="mobile-nav-item"><a href="/tag/tips/" class="mobile-nav-link">Tips</a></li>
                        <li class="mobile-nav-item"><a href="/tag/by-partners/" class="mobile-nav-link">By partners</a></li>
                        <li class="mobile-nav-item"><a href="/tag/news/" class="mobile-nav-link">News</a></li>
                        <li class="mobile-nav-item"><a href="/tag/interview/" class="mobile-nav-link">Interviews</a></li>
						<li class="mobile-nav-item mobile-auth">
							<a href="https://app.n8n.cloud/" class="mobile-nav-link">Sign in</a>
							<a href="https://app.n8n.cloud/register" class="mobile-get-started-btn">Get Started</a>
						</li>
					</ul>
				</nav>
			</div>
		</div>
	</div>
</header>				<main class="global-main">
					<progress class="post-progress"></progress>

<!-- Hero Section for Blog Post -->
<div class="post-hero-section">
	<div class="post-hero-container">
		<div class="post-hero-content">
			<div class="post-hero-tags">
				<a href="/tag/ai/" class="post-hero-chip">AI</a>
			</div>
			<div class="post-hero-copy">
				<div class="post-hero-header">
					<h1 class="post-hero-title">How to evaluate the performance of AI agents?</h1>
				</div>
				<div class="post-hero-subheader">
					<p>Evaluate AI agents effectively: learn offline vs. online testing, key metrics, and methods like deterministic checks, LLM-as-a-judge, and human review to improve performance and reliability.</p>
				</div>
			</div>
			<div class="post-hero-authors">
				<div class="post-hero-avatars">
					<div class="post-hero-avatar" style="background-image: url(https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2023/01/profile-pik.jpg)">
					</div>
					<div class="post-hero-avatar" style="background-image: url(https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2022/05/new_pic.png)">
					</div>
				</div>
				<div class="post-hero-author-info">
					<div class="post-hero-author-name">
						<a href="/author/yulia/">Yulia Dmitrievna</a>, <a href="/author/eduard/">Eduard Parsadanyan</a>
					</div>
					<div class="post-hero-author-meta">
						<time datetime="2026-04-21">April 21, 2026</time> ∙ 12 minutes read
					</div>
				</div>
			</div>
		</div>
		<div class="post-hero-node-box">
			<div class="post-hero-image-wrapper">
				<img class="post-hero-image"
					srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w400/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg 400w,
							https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg 600w,
							https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w800/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg 1000w,
							https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg 1600w"
					sizes="(max-width: 480px) 400px,
						   (max-width: 768px) 600px,
						   (max-width: 1024px) 800px,
						   544px"
					src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w800/2026/04/BlogHeader_How-to-evaluate-the-performance-of-AI-agents--1-.jpg"
					alt="How to evaluate the performance of AI agents?"
					loading="eager">
			</div>
		</div>
	</div>
</div>

<!-- Post Content -->
<article class="post-article">
	<div class="post-content-wrapper">
		<div id="toc"></div>
		<div class="post-content">
		<p>Traditional software testing is straightforward: you give input X and expect output Y. If the function returns the wrong value, the test fails.</p><p>LLM-based agents don't work that way. They're non-deterministic which means the same prompt can produce different outputs across runs. They operate over multiple steps, making decisions about which tools to call, what parameters to pass, and how to interpret results.&nbsp;</p><p>An agent can complete an execution without errors and still hallucinate facts, miss the user's intent, or take unnecessary steps. Classical testing may not catch problematic outputs produced by an AI Agent.</p><p>When <a href="https://n8n.io/ai-agents/"><u>building AI Agents</u></a>, you face three main evaluation challenges:</p><ol><li><strong>You're evaluating trajectories, instead of just outputs.</strong> An agent might give the correct final answer but call the wrong tools, use the wrong parameters, or take five steps when one would do. If you only check the final result, you'll overlook these issues.</li><li><strong>Successful performance is harder to define.</strong> "Good" output often involves subjective qualities such as tone, helpfulness, and policy compliance. You need different evaluation methods for different quality dimensions.</li><li><strong>One-time testing isn't enough.</strong> Models get upgraded, new edge cases emerge over time, and user behavior may shift. This means that agents that work today might degrade tomorrow.</li></ol><p>Systematic evaluation allows you to overcome these challenges and bridge the gap between AI Agent changes and their performance impact. The way you approach it depends on where you are in your journey.</p><h2 id="how-do-i-start-evaluating-an-ai-agent">How do I start evaluating an AI Agent?</h2><p>As you scale your use of AI agents, evaluation typically evolves through stages. Most teams start with manual testing and expand as agents move toward production. Where you begin depends on your current stage and your risk tolerance.</p><table>
<thead>
<tr>
<th>Stage</th>
<th>When it's used</th>
<th>How it's used</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ad-hoc</td>
<td>Prototypes,<br>early development</td>
<td>Manual spot-checking, eyeball results</td>
</tr>
<tr>
<td>Curated test<br>suites</td>
<td>During development,<br>before major changes</td>
<td>Structured datasets with a number of<br>pre-defined test cases, manually<br>triggered or semi-automated</td>
</tr>
<tr>
<td>CI-integrated<br>evaluations</td>
<td>Automated validation<br>on every commit</td>
<td>Automated tests in pipeline,<br>pass/fail gates</td>
</tr>
<tr>
<td>Production<br>monitoring</td>
<td>Live systems</td>
<td>Continuous scoring, A/B tests,<br>alerts on degradation</td>
</tr>
</tbody>
</table>
<h2 id="what-are-the-main-approaches-for-evaluating-ai-agents">What are the main approaches for evaluating AI Agents?</h2><p>At a high level, evaluation happens in two contexts: offline (on test datasets) and online (in production). Each approach helps to catch different issues.</p><h3 id="offline-evaluation">Offline evaluation</h3><p><strong>Offline evaluation</strong> runs your agent against curated test datasets during development or in CI pipelines. You define inputs, expected outputs, and success criteria. Then either manually check whether the agent meets your criteria or set up automated testing before you push to production.</p><p>Offline evaluation helps you:</p><ul><li>Catch regressions before they reach users</li><li>Compare performance across prompts or model changes</li><li>Validate that edge cases still work after updates</li></ul><p>The limitation of this approach is that your test dataset only covers scenarios that you anticipated. Real users may hit edge cases outside your test coverage.</p><h3 id="online-evaluation">Online evaluation</h3><p><strong>Online evaluation</strong> scores live traffic and collects feedback from actual usage. It catches issues your test suite didn’t foresee, such as unexpected inputs, edge cases you didn't think of, gradual performance degradation as user behavior shifts.</p><p>Online evaluation helps you:</p><ul><li>Detect model drift over time</li><li>Surface failure patterns from real conversations</li><li>Collect user feedback on what actually matters</li></ul><p>The downside of online evaluation is that you're catching issues after they've already affected users.</p><p><strong>In practice, you may benefit from using both approaches.</strong> Offline evaluation catches issues before deployment. Online evaluation reveals unknown issues once they occur. A typical setup runs offline checks on every change, then monitors key metrics in production to catch what slipped through.</p><p>The next question to ask is how to evaluate the output of AI Agents. What methods work best for various quality dimensions?</p><div class="kg-card kg-callout-card kg-callout-card-blue"><div class="kg-callout-emoji">💡</div><div class="kg-callout-text">Agent evaluation builds on LLM evaluation. LLM evals check single outputs and agent evals add a layer on top: tool usage, step efficiency, reasoning paths. This article focuses on the agent-level evaluation, for LLM-level methods see the <a href="https://blog.n8n.io/practical-evaluation-methods-for-enterprise-ready-llms/"><u>practical evaluation methods for enterprise-ready LLMs</u></a>.</div></div><h2 id="which-evaluation-methods-exist-for-ai-agents">Which evaluation methods exist for AI Agents?</h2><p>Different quality dimensions require different evaluation methods. A single method won't cover all dimensions – you'll typically combine several.</p><p>Let’s look at the brief summary of the methods that most production setups layer:</p>
<!--kg-card-begin: html-->
<table>
<thead>
<tr>
<th>Method</th>
<th>Use for</th>
<th>Scales?</th>
<th>Cost</th>
</tr>
</thead>
<tbody>
<tr>
<td>Deterministic checks</td>
<td>Objective criteria, formats, actions</td>
<td>✅ Yes</td>
<td>Low</td>
</tr>
<tr>
<td>LLM-as-a-judge</td>
<td>Subjective qualities, open-ended outputs</td>
<td>✅ Yes</td>
<td>Medium</td>
</tr>
<tr>
<td>Human review</td>
<td>High-stakes, complex reasoning, calibration</td>
<td>❌ No</td>
<td>High</td>
</tr>
<tr>
<td>User feedback</td>
<td>Real-world value, satisfaction</td>
<td>✅ Yes</td>
<td>Low</td>
</tr>
</tbody>
</table>
<!--kg-card-end: html-->
<div class="kg-card kg-callout-card kg-callout-card-blue"><div class="kg-callout-emoji">💡</div><div class="kg-callout-text">These methods work across both offline and online approaches.&nbsp;<br>You can combine evaluation methods: deterministic checks for fast validation before and during production, LLM-as-a-judge for offline quality assessment and sampled production monitoring, user feedback for capturing real-world satisfaction, and human review for calibration and high-stakes spot checks.</div></div><p>No matter which methods you prefer, you need to understand what exactly to measure. The right metrics depend on which quality parameters matter most for your use case.&nbsp;</p><h3 id="deterministic-checks">Deterministic checks</h3><p>Rule-based validation for objective criteria: regex patterns, schema validation (JSON validity checks), exact string matching, required field presence, output length limits.&nbsp;</p><ul><li><strong>Best for:</strong> structured outputs, classification tasks, format compliance, action validation ("did the agent call the correct tool?")</li><li><strong>Strengths:</strong> fast, cheap, fully reproducible, no ambiguity</li><li><strong>Limitations:</strong> can't assess subjective qualities like helpfulness or tone</li></ul><p>Start here. If you can define your success criteria as deterministic rules, this can be considered as the most straightforward evaluation method</p><h3 id="llm-as-a-judge">LLM-as-a-judge</h3><p>Use an LLM to score outputs against criteria like correctness, helpfulness, or tone. The evaluator model reads the agent's response (and optionally a reference answer) and assigns a score.</p><ul><li><strong>Best for:</strong> subjective qualities, open-ended responses, cases where "correct" isn't binary</li><li><strong>Strengths:</strong> scales better than human review, handles nuance</li><li><strong>Limitations:</strong> costs tokens, can inherit biases (i.e. may prefer longer responses), results may shift when evaluation models update</li></ul><p>Validate LLM-as-a-judge metrics against human ratings periodically. If your evaluator drifts, your scores may not reflect the actual outcome quality.</p><h3 id="human-review">Human review</h3><p>Domain experts or annotators grade outputs using structured rubrics. For agents, this often means reviewing execution traces – not just the final answer, but which tools were called, what parameters were passed, and how the agent reasoned through the task.</p><ul><li><strong>Best for:</strong> high-stakes decisions, nuanced domain knowledge, calibrating automated evaluators</li><li><strong>Strengths:</strong> catches what automated methods miss, builds ground truth datasets</li><li><strong>Limitations:</strong> expensive, slow, doesn't scale</li></ul><p>Use human review strategically – for validating edge cases, auditing samples, and building reference datasets that improve your automated checks.</p><h3 id="user-feedback">User feedback</h3><p>Direct signals from the people using your agent: thumbs up/down, ratings, follow-up questions, escalation requests.</p><ul><li><strong>Best for:</strong> understanding real-world value, catching "technically correct but unhelpful" responses</li><li><strong>Strengths:</strong> reflects actual user needs, requires no ground truth</li><li><strong>Limitations:</strong> skews toward extremes (people respond when delighted or frustrated), low response rates, noisy signal</li></ul><p>Combine user feedback with automated metrics. If evaluations show 95% correctness but users answer you with 3 stars, the agent might be accurate but unhelpful.</p><h2 id="which-metrics-can-i-use-to-evaluate-ai-agent-performance">Which metrics can I use to evaluate AI Agent performance?</h2><p>Broadly speaking, you can group metrics into two big categories based on how you measure them.</p><h3 id="deterministic-metrics">Deterministic metrics</h3><p>Deterministic metrics are based on rules and fully automated. They're fast to compute, reproducible across runs, and don't add cost per evaluation.</p>
<!--kg-card-begin: html-->
<table>
<thead>
<tr>
<th>Common deterministic metrics</th>
<th>What each metric measures</th>
</tr>
</thead>
<tbody>
<tr>
<td>Task completion rate</td>
<td>Did the agent finish the task without errors?</td>
</tr>
<tr>
<td>Tool usage accuracy</td>
<td>Was the correct tool called with the right parameters?</td>
</tr>
<tr>
<td>Exact match / categorization</td>
<td>Did the output match the expected value?</td>
</tr>
<tr>
<td>Format compliance</td>
<td>Are all required fields present, is the schema valid?</td>
</tr>
<tr>
<td>Step efficiency</td>
<td>Did the agent avoid unnecessary steps and loops?</td>
</tr>
<tr>
<td>Operational metrics</td>
<td>Are latency, token usage, and cost within targets?</td>
</tr>
</tbody>
</table>
<!--kg-card-end: html-->
<p>Use deterministic metrics when you can clearly define the rules without relying on LLMs. These methods are suitable for both offline and online testing and indicate potentially problematic cases that require a closer look.</p><h3 id="model-based-metrics">Model-based metrics</h3><p><strong>Model-based metrics</strong> use an LLM to judge outputs. Their main advantage lies in handling subjective qualities that rules can't capture. The downside of LLM-based evaluations is that they cost tokens and may shift when language models update.</p>
<!--kg-card-begin: html-->
<table>
<thead>
<tr>
<th>Common LLM-based metrics</th>
<th>What each metric measures</th>
</tr>
</thead>
<tbody>
<tr>
<td>Correctness</td>
<td>Does the output semantically match the reference answer?</td>
</tr>
<tr>
<td>Helpfulness</td>
<td>Does the agent answer the user's actual question?</td>
</tr>
<tr>
<td>Groundedness</td>
<td>Is the output supported by retrieved sources? (RAG)</td>
</tr>
<tr>
<td>Reasoning quality</td>
<td>Does the chain-of-thought make sense?</td>
</tr>
<tr>
<td>Tone / policy compliance</td>
<td>Does the output follow appropriate style and guidelines?</td>
</tr>
<tr>
<td>Instruction following</td>
<td>Does it respect system prompt constraints?</td>
</tr>
</tbody>
</table>
<!--kg-card-end: html-->
<p>For offline evaluation, it’s worth running model-based metrics on your full test dataset. For online evaluation, it's better to be selective and run them on a random sample of production cases, or trigger LLM-based checks only when deterministic checks show drops in quality. This way, you can keep costs manageable while still catching subjective issues.</p><p>Finally, some metrics work either way depending on implementation. For example, PII detection, toxicity, prompt injection can use keyword pattern matching (deterministic rules) or a classifier model. Choose based on your accuracy needs and cost constraints.&nbsp;</p><p>With your evaluation approach, methods and metrics defined, you need tools to put them into practice.</p><h2 id="which-tools-can-i-use-for-evaluating-ai-agents">Which tools can I use for evaluating AI agents?</h2><p>Evaluation software has evolved alongside LLM applications. Early tools focused on checking whether an LLM gave correct answers to benchmark questions. As teams moved from simple prompts to RAG pipelines and multi-step agents, the tools expanded too.</p><p>Today's platforms handle traces across multiple steps, score tool usage, and track metrics over time. But most of them weren't built specifically for agentic workflows and adapted to the new use-cases later. This means that you'll often need to combine tools. For example, you may use one tool for prompt-level testing, and another one for traceability.</p><p>Evaluation tools generally falls into several categories based on their overall functionality:</p><table>
<thead>
<tr>
<th>Tool</th>
<th>Category</th>
<th>What it does</th>
</tr>
</thead>
<tbody>
<tr>
<td>DeepEval</td>
<td>Evaluation library</td>
<td>Pytest-style testing for LLM apps with dozens of built-in metrics</td>
</tr>
<tr>
<td>RAGAS</td>
<td>Evaluation library</td>
<td>Focused on RAG evaluation, measures retrieval precision and<br>generation quality</td>
</tr>
<tr>
<td>Promptfoo</td>
<td>CLI and evaluation<br>library</td>
<td>CLI-first prompt testing with red-teaming and security scanning,<br>integrates with CI pipelines</td>
</tr>
<tr>
<td>LangSmith</td>
<td>Observability platform</td>
<td>Full tracing and evaluation from the LangChain team,<br>strong for LangChain-based agents</td>
</tr>
<tr>
<td>Langfuse</td>
<td>Observability platform</td>
<td>Open-source, supports custom evaluators and human annotation</td>
</tr>
<tr>
<td>Arize Phoenix</td>
<td>Observability +<br>evaluation platform</td>
<td>Open-source observability with evaluation hooks,<br>good for teams with existing ML monitoring infrastructure</td>
</tr>
<tr>
<td>n8n</td>
<td>Workflow automation +<br>evaluation</td>
<td>Build agents and evaluate them on the same platform,<br>with native data tables, built-in metrics, and feedback collection</td>
</tr>
</tbody>
</table>
<p>Most tools in the list focus on evaluation as a standalone concept.&nbsp;</p><div class="kg-card kg-callout-card kg-callout-card-blue"><div class="kg-callout-emoji">💡</div><div class="kg-callout-text">n8n takes a different approach: <a href="https://docs.n8n.io/advanced-ai/evaluations/overview/"><u>evaluation </u></a>is built into the same workflow where your agent runs. You can create test datasets, define metrics, and collect user feedback on the same platform. In addition to that, it’s possible to wire up external services, such as LangSmith.</div></div><p>This integrated approach means you can go from building your agent to evaluating it without switching tools. Let's see how this works in practice with real examples.</p><h2 id="how-to-evaluate-ai-agents-in-n8n">How to evaluate AI agents in n8n?</h2><p>n8n is a workflow automation platform with a visual builder for creating AI agents. You can connect LLMs to hundreds of integrations, add tools, memory, conditional logic and then deploy to production agents that interact with your actual business systems.</p><div class="kg-card kg-callout-card kg-callout-card-blue"><div class="kg-callout-emoji">💡</div><div class="kg-callout-text">New to n8n? Check this blog post on <a href="https://blog.n8n.io/how-to-build-ai-agent/"><u>how to build your first AI agent</u></a>.</div></div><p>n8n includes built-in evaluation capabilities alongside the agent building tools. You can create test datasets, run evaluations, add human oversight, and inspect execution traces – all within the same platform where you build and deploy your agents.</p><p>Both offline and online evaluation patterns work natively in n8n.</p><h3 id="offline-evaluation-with-n8n">Offline evaluation with n8n </h3><p>n8n's <a href="https://docs.n8n.io/advanced-ai/evaluations/overview/"><u>Evaluations feature</u></a> lets you run test datasets through your agent workflow before deploying changes.</p><p><strong>Store test cases in built-in Data Tables.</strong> Keep inputs and expected outputs in <a href="https://docs.n8n.io/data/data-tables/"><u>Data Tables</u></a> or Google Sheets. Add context columns if your agent needs them: user type, conversation history, session data.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-ee23e760-5c9b-432f-9594-1ca626e7e7e3.png" class="kg-image" alt="Here’s a sample dataset for evaluating LLM classification outputs. Source: https://blog.n8n.io/llm-evaluation-framework/" loading="lazy" width="1000" height="296" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-ee23e760-5c9b-432f-9594-1ca626e7e7e3.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-ee23e760-5c9b-432f-9594-1ca626e7e7e3.png 1000w"><figcaption><i><em class="italic" style="white-space: pre-wrap;">Here’s a sample dataset for evaluating LLM classification outputs. Source: https://blog.n8n.io/llm-evaluation-framework/</em></i></figcaption></figure><p><strong>Run tests through your workflow.</strong> The <a href="https://docs.n8n.io/integrations/builtin/core-nodes/n8n-nodes-base.evaluationtrigger/"><u>Evaluation Trigger</u></a> node runs each test case through your actual agent logic. The same workflow handles both evaluation runs and production traffic.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-501b7824-4671-4a53-bd5f-ba2191b9794e.png" class="kg-image" alt="An example of the online evaluation: the agent’s output is evaluated even in the production runs." loading="lazy" width="1202" height="554" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-501b7824-4671-4a53-bd5f-ba2191b9794e.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1000/2026/04/data-src-image-501b7824-4671-4a53-bd5f-ba2191b9794e.png 1000w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-501b7824-4671-4a53-bd5f-ba2191b9794e.png 1202w" sizes="(min-width: 1200px) 1200px"><figcaption><i><em class="italic" style="white-space: pre-wrap;">An example of the online evaluation: the agent’s output is evaluated even in the production runs.</em></i></figcaption></figure><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-1c61b20a-b6ba-4a8b-9bf2-53baa9d8c4c2.png" class="kg-image" alt="An example of the offline evaluation: when the user interacts with the agent, it reacts in a normal way. The evaluation trigger initiates the checks only when started manually. " loading="lazy" width="1600" height="539" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-1c61b20a-b6ba-4a8b-9bf2-53baa9d8c4c2.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1000/2026/04/data-src-image-1c61b20a-b6ba-4a8b-9bf2-53baa9d8c4c2.png 1000w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-1c61b20a-b6ba-4a8b-9bf2-53baa9d8c4c2.png 1600w" sizes="(min-width: 1200px) 1200px"><figcaption><i><em class="italic" style="white-space: pre-wrap;">An example of the offline evaluation: when the user interacts with the agent, it reacts in a normal way. The evaluation trigger initiates the checks only when started manually. </em></i></figcaption></figure><p><strong>Score outputs.</strong> Use built-in metrics (Correctness, Helpfulness, String Similarity, Categorization) or define custom scoring logic with the Code node.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-b35d63e3-71ce-4b02-9a56-40df0631624f.png" class="kg-image" alt="Workflow with Evaluation Trigger node connected to agent. Source; https://n8n.io/workflows/11832-sales-lead-routing-with-gemini-sentiment-analysis-and-model-evaluation-framework/" loading="lazy" width="1000" height="733" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-b35d63e3-71ce-4b02-9a56-40df0631624f.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-b35d63e3-71ce-4b02-9a56-40df0631624f.png 1000w"><figcaption><i><em class="italic" style="white-space: pre-wrap;">Workflow with Evaluation Trigger node connected to agent. Source; </em></i><span style="white-space: pre-wrap;">https://n8n.io/workflows/11832-sales-lead-routing-with-gemini-sentiment-analysis-and-model-evaluation-framework/</span></figcaption></figure><p><strong>Review results.</strong> Check rates and quality scores before deploying changes to production.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-fd28a264-8aea-4a3a-b520-f3b6e0641140.png" class="kg-image" alt="Evaluation results summary" loading="lazy" width="936" height="767" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-fd28a264-8aea-4a3a-b520-f3b6e0641140.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-fd28a264-8aea-4a3a-b520-f3b6e0641140.png 936w"><figcaption><i><em class="italic" style="white-space: pre-wrap;">Evaluation results summary</em></i></figcaption></figure><p>Check n8n workflow templates with different Evaluation scenarios (tool usage, answer similarity, LLM as a judge, programmatic metrics).</p>
<!--kg-card-begin: html-->
<script>
  workflowBanner([5523, 4423, 4271, 4274], document.currentScript);
</script>
<!--kg-card-end: html-->
<h3 id="online-evaluation-with-n8n">Online evaluation with n8n</h3><p>For live agents, n8n supports several monitoring patterns:</p><p><strong>Guardrails on inputs and outputs.</strong> Place checks before and after your agent node to catch PII, policy violations, or malformed requests in real time.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-80adfe7f-f6ce-48c8-b886-73391ef76f49.png" class="kg-image" alt="Workflow with guardrail nodes before and after the agent. Source: <https://n8n.io/workflows/13853-guardrail-ai-inputs-and-outputs-with-gpt-4o-mini-and-n8n-guardrails/>Workflow with guardrail nodes before and after the agent. Source: <https://n8n.io/workflows/13853-guardrail-ai-inputs-and-outputs-with-gpt-4o-mini-and-n8n-guardrails/>" loading="lazy" width="1419" height="469" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-80adfe7f-f6ce-48c8-b886-73391ef76f49.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1000/2026/04/data-src-image-80adfe7f-f6ce-48c8-b886-73391ef76f49.png 1000w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-80adfe7f-f6ce-48c8-b886-73391ef76f49.png 1419w" sizes="(min-width: 1200px) 1200px"><figcaption><i><em class="italic" style="white-space: pre-wrap;">Workflow with guardrail nodes before and after the agent. Source</em></i><span style="white-space: pre-wrap;">: &lt;https://n8n.io/workflows/13853-guardrail-ai-inputs-and-outputs-with-gpt-4o-mini-and-n8n-guardrails/&gt;</span></figcaption></figure><p><strong>Log metrics to a dataset.</strong> Capture data points from live executions, such as response quality scores, tool usage, latency and write them to a Data Table for trend analysis.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-a7b9f2e7-1ff7-4c7a-881d-455465c329a6.png" class="kg-image" alt="The modified version of the evaluation workflow. In addition to offline evaluation the workflow now logs all live agent’s answers in the separate data table. Source: Original workflowThe modified version of the evaluation workflow. In addition to offline evaluation the workflow now logs all live agent’s answers in the separate data table. " loading="lazy" width="1576" height="707" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-a7b9f2e7-1ff7-4c7a-881d-455465c329a6.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1000/2026/04/data-src-image-a7b9f2e7-1ff7-4c7a-881d-455465c329a6.png 1000w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-a7b9f2e7-1ff7-4c7a-881d-455465c329a6.png 1576w" sizes="(min-width: 1200px) 1200px"><figcaption><i><em class="italic" style="white-space: pre-wrap;">The modified version of the evaluation workflow. In addition to offline evaluation the workflow now logs all live agent’s answers in the separate data table. </em></i></figcaption></figure><p><strong>Separate monitoring workflow.</strong> Run periodic checks against your logged data to detect degradation or trigger alerts when metrics drop. You can launch a second workflow and apply the metrics on real agent's answers.</p><p><strong>User feedback collection.</strong> This is how we can implement human-in-the-loop evaluation after the agent runs. Add feedback prompts (Slack reactions, email links) to agent responses once the task is complete.&nbsp;</p><p>Users rate the output, and their responses route back to n8n via webhooks for logging and review. Unlike usual approval nodes that pause execution mid-workflow, this setup lets you capture quality signals without slowing down the agent.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-3fa94feb-80dd-4266-851a-a21f8a22add0.png" class="kg-image" alt="Send a message to users once the agent finished working to collect quick feedback and store it in a data table.Send a message to users once the agent finished working to collect quick feedback and store it in a data table." loading="lazy" width="1310" height="613" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-3fa94feb-80dd-4266-851a-a21f8a22add0.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1000/2026/04/data-src-image-3fa94feb-80dd-4266-851a-a21f8a22add0.png 1000w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-3fa94feb-80dd-4266-851a-a21f8a22add0.png 1310w" sizes="(min-width: 1200px) 1200px"><figcaption><i><em class="italic" style="white-space: pre-wrap;">Send a message to users once the agent finished working to collect quick feedback and store it in a data table.</em></i></figcaption></figure><p><strong>Connect external tools.</strong> For deeper observability, you can integrate your n8n workflow with such tools as <a href="https://docs.n8n.io/advanced-ai/langchain/langsmith/"><u>LangSmith</u></a>, or <a href="https://www.promptfoo.dev/docs/integrations/n8n/"><u>Promptfoo</u></a>.</p><figure class="kg-card kg-image-card kg-width-wide kg-card-hascaption"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-f1fbd2e9-3234-41c8-86c5-161a9ef0e2c4.png" class="kg-image" alt="LangSmith allows for an in-depth agent tracing analysis. Source: https://www.langchain.com/LangSmith allows for an in-depth agent tracing analysis. Source: https://www.langchain.com/" loading="lazy" width="1600" height="512" srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/04/data-src-image-f1fbd2e9-3234-41c8-86c5-161a9ef0e2c4.png 600w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1000/2026/04/data-src-image-f1fbd2e9-3234-41c8-86c5-161a9ef0e2c4.png 1000w, https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/data-src-image-f1fbd2e9-3234-41c8-86c5-161a9ef0e2c4.png 1600w" sizes="(min-width: 1200px) 1200px"><figcaption><i><em class="italic" style="white-space: pre-wrap;">LangSmith allows for an in-depth agent tracing analysis. Source: https://www.langchain.com/</em></i></figcaption></figure><h2 id="how-does-n8n-fit-into-ai-evaluation-concepts">How does n8n fit into AI evaluation concepts?</h2><p>Here's a quick reference connecting the evaluation methods we’ve covered to their n8n implementation:</p><table>
<thead>
<tr>
<th>Concept</th>
<th>How to implement it in n8n/</th>
</tr>
</thead>
<tbody>
<tr>
<td>Test datasets</td>
<td>Data Tables store inputs, agent responses and expected<br>outputs (ground truth)</td>
</tr>
<tr>
<td>Offline evaluation</td>
<td><a href="https://docs.n8n.io/integrations/builtin/core-nodes/n8n-nodes-base.evaluationtrigger/">The Evaluation Trigger node</a> processes test cases through<br>the same workflow that runs in production</td>
</tr>
<tr>
<td>Deterministic checks</td>
<td><a href="https://docs.n8n.io/code/expressions/">n8n expressions</a>, some <a href="https://docs.n8n.io/advanced-ai/evaluation/metrics/">built-in metrics</a> like string similarity,<br>categorisation and tools used; <a href="https://docs.n8n.io/integrations/builtin/core-nodes/n8n-nodes-base.code/">Code node</a> for custom validation</td>
</tr>
<tr>
<td>LLM-as-a-judge</td>
<td>Correctness and Helpfulness (the other two built-in metrics) or<br>custom metrics</td>
</tr>
<tr>
<td>Manual review</td>
<td>Check the trends in the evaluation summary dashboard, view<br>the evaluation table directly, analyse the step-by-step execution logs</td>
</tr>
<tr>
<td>User feedback</td>
<td>"Send and Wait for Response" operation in various nodes to collect<br>user ratings after the agent completes</td>
</tr>
</tbody>
</table>
<div class="kg-card kg-callout-card kg-callout-card-blue"><div class="kg-callout-emoji">💡</div><div class="kg-callout-text">Run evaluations before every prompt or model change: even 5-10 well-chosen test cases can catch more issues than hundreds of random ones. As you encounter edge cases in production or user-reported problems, add them to your test dataset to prevent regressions.</div></div><h2 id="wrap-up">Wrap up</h2><p>This article walks through how to systematically evaluate AI agents, from key challenges to offline and online approaches, as well as core evaluation methods.</p><p>It also shows how n8n supports agent evaluation with Data Tables, the Evaluation Trigger, built-in metrics, human-in-the-loop nodes, and external integrations.</p>
<!--kg-card-begin: html-->
<div class="content-banner">
  <div>
    <h3>Build and evaluate in n8n</h3>
    <p>Go from prototype to production-ready agent without switching tools</p>
  </div>
  <a href="https://app.n8n.cloud/register" class="global-button blog-banner-signup">Try n8n now</a>
</div>
<!--kg-card-end: html-->
<p>A practical way to get started is to choose a small set of meaningful test cases, run evaluations before each change, and continuously expand your dataset with real production failures.</p><h2 id="whats-next">What's next?</h2><p>With the foundation on how to evaluate the performance of AI Agents, you might want to dive deeper into the topic and explore our hands-on tutorial on building a complete evaluation system: <a href="https://blog.n8n.io/llm-evaluation-framework/"><strong><u>Building your own LLM evaluation framework with n8n</u></strong></a>.&nbsp;</p><p>This guide covers the "LLM-as-a-Judge" pattern we’ve discussed today in more detail and shows you how to create custom evaluation pipelines.</p><p>Take your n8n knowledge further:</p><ul><li>Explore the <a href="https://n8n.io/integrations/categories/ai/"><u>AI integrations catalog</u></a> to see available models and tools</li><li>Browse through <a href="https://n8n.io/workflows/?integrations=Evaluation"><u>evaluation workflow templates</u></a> for working examples</li><li><u>Or</u><a href="https://app.n8n.cloud/register"><u> start here</u></a> if you’re new to n8n!</li></ul>
		<!-- Share with us section -->
		<div class="post-share-section">
			<div class="post-share-content">
				<h2 class="post-share-title">Share with us</h2>
				<div class="post-share-body">
					<p>n8n users come from a wide range of backgrounds, experience levels, and interests. We have been looking to highlight different users and their projects in our blog posts. If you're working with n8n and would like to inspire the community, <a href="https://n8n-community.typeform.com/to/VYiRI7WN?ref=blog.n8n.io">contact us</a> 💌</p>
				</div>
			</div>


			<!-- Share buttons -->
			<div class="post-share-buttons">
				<span class="post-share-label">SHARE</span>
				<div class="post-share-icons">
					<a href="#" class="post-share-link" onclick="copyToClipboard(window.location.href); return false;">
						<svg width="24" height="24" viewBox="0 0 24 24" fill="none" xmlns="http://www.w3.org/2000/svg">
							<path d="M11.293 8.293L7.58604 4.58604C6.80499 3.80499 5.53866 3.80499 4.75761 4.58604C3.97656 5.36709 3.97656 6.63342 4.75761 7.41447L8.46458 11.1214C9.24563 11.9025 10.512 11.9025 11.293 11.1214C12.0741 10.3404 12.0741 9.07405 11.293 8.293Z" stroke="#E8E5EB" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
							<path d="M12.707 15.707L16.414 19.414C17.195 20.195 18.4613 20.195 19.2424 19.414C20.0234 18.6329 20.0234 17.3666 19.2424 16.5855L15.5354 12.8786C14.7544 12.0975 13.488 12.0975 12.707 12.8786C11.9259 13.6596 11.9259 14.9259 12.707 15.707Z" stroke="#E8E5EB" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
							<path d="M14.8285 9.17145L9.17145 14.8285" stroke="#E8E5EB" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
						</svg>
					</a>
					<a href="https://twitter.com/intent/tweet?text=How%20to%20evaluate%20the%20performance%20of%20AI%20agents%3F&url=https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/" target="_blank" rel="noopener" class="post-share-twitter">
						<svg width="24" height="24" viewBox="0 0 24 24" fill="none" xmlns="http://www.w3.org/2000/svg">
							<path d="M17.75 3.75H20.5L13.93 11.18L21.64 20.25H15.67L10.99 14.32L5.64 20.25H2.89L9.91 12.4L2.5 3.75H8.64L12.87 9.16L17.75 3.75ZM16.49 18.47H18.14L7.74 5.44H5.97L16.49 18.47Z" fill="#E8E5EB"/>
						</svg>
					</a>
					<a href="https://www.linkedin.com/sharing/share-offsite/?url=https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/" target="_blank" rel="noopener" class="post-share-linkedin">
						<svg width="24" height="24" viewBox="0 0 24 24" fill="none" xmlns="http://www.w3.org/2000/svg">
							<path d="M20.447 20.452H16.893V14.883C16.893 13.555 16.866 11.846 15.041 11.846C13.188 11.846 12.905 13.291 12.905 14.785V20.452H9.351V9H12.765V10.561H12.811C13.288 9.661 14.448 8.711 16.181 8.711C19.782 8.711 20.448 11.081 20.448 14.166V20.452H20.447ZM5.337 7.433C4.193 7.433 3.274 6.507 3.274 5.368C3.274 4.23 4.194 3.305 5.337 3.305C6.477 3.305 7.401 4.23 7.401 5.368C7.401 6.507 6.476 7.433 5.337 7.433ZM7.119 20.452H3.555V9H7.119V20.452ZM22.225 0H1.771C0.792 0 0 0.774 0 1.729V22.271C0 23.227 0.792 24 1.771 24H22.222C23.2 24 24 23.227 24 22.271V1.729C24 0.774 23.2 0 22.222 0H22.225Z" fill="#E8E5EB"/>
						</svg>
					</a>
				</div>
			</div>
		</div>
		</div>
	</div>
</article>

<aside class="post-navigation-section">
	<div class="post-navigation-wrap">
		<a href="/human-in-the-loop-vs-human-on-the-loop/" class="post-navigation-card" aria-label="Human-in-the-Loop vs. Human-on-the-Loop: When To Use Each System">
			<div class="post-navigation-thumbnail">
				<div class="post-navigation-image" style="background-image: url(https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/TL-1_BlogHeader_Human-in-the-Loop-vs-Human-on-the-Loop.jpg)"></div>
			</div>
			<div class="post-navigation-content">
				<div class="post-navigation-header">
					<h3 class="post-navigation-title">Human-in-the-Loop vs. Human-on-the-Loop: When To Use Each System</h3>
				</div>
				<div class="post-navigation-label">Newer post</div>
			</div>
		</a>
		<a href="/workflow-vs-orchestration/" class="post-navigation-card" aria-label="Workflow Automation vs. Orchestration: Architectural Differences That Matter at Scale">
			<div class="post-navigation-thumbnail">
				<div class="post-navigation-image" style="background-image: url(https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2026/04/TL-1_BlogHeader_workflow-vs-orchestration-1.jpg)"></div>
			</div>
			<div class="post-navigation-content">
				<div class="post-navigation-header">
					<h3 class="post-navigation-title">Workflow Automation vs. Orchestration: Architectural Differences That Matter at Scale</h3>
				</div>
				<div class="post-navigation-label">Older post</div>
			</div>
		</a>
	</div>
</aside><div class="comments-section">
	<div class="comments-wrap">
			</div>
</div>
<section class="post-related-section"
         data-related-api
         data-post-url="https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/"
         data-tag-name="AI"
         style="display:none">
  <div class="post-related-container">
    <div class="post-related-header">
      <h2 class="post-related-title">Other guides to get you going</h2>
    </div>
    <div class="post-related-grid"></div>
  </div>
</section>
<section class="post-workflows-section"
         data-workflows-api
         data-post-url="https://blog.n8n.io/how-to-evaluate-the-performance-of-ai-agents/"
         style="display:none">
  <div class="post-workflows-container">
    <div class="post-workflows-header">
      <h2 class="post-workflows-title">Where there's a will, there's already a workflow</h2>
    </div>
    <div class="post-workflows-grid"></div>
  </div>
</section>
<!--<section class="post-latest-section">
	<div class="post-latest-container">
		<div class="post-latest-header">
			<h2 class="post-latest-title">Latest n8n guides</h2>
		</div>
		<div class="post-latest-grid">
			<article class="post-latest-card post tag-guides tag-ai-9">
	<a href="/mcp-vs-api/" class="global-link" aria-label="MCP vs. API: Key Differences and When To Use Each"></a>
	<div class="post-latest-card-container">
		<a href="/mcp-vs-api/" class="post-latest-card-image global-image">
			<img srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w400/2026/09/TL--17-.jpeg 400w,
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/09/TL--17-.jpeg 600w, 
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w800/2026/09/TL--17-.jpeg 800w,
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/09/TL--17-.jpeg 1200w"
	 sizes="(max-width:480px) 350px, 500px"
	 src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/09/TL--17-.jpeg"
	 loading="lazy"
	 alt="MCP vs. API: Key Differences and When To Use Each">
		</a>
		<div class="post-latest-card-content">
			<div class="post-latest-card-tags global-tags">
								<a href="/tag/guides/">Guides</a><a href="/tag/ai-9/">AI</a>
			</div>
			<div class="post-latest-card-copy">
				<h3 class="post-latest-card-title"><a href="/mcp-vs-api/">MCP vs. API: Key Differences and When To Use Each</a></h3>
				<div class="post-latest-card-authors global-meta">
					<a href="/author/marketing/">n8n team</a>, <a href="/author/yulia/">Yulia Dmitrievna</a>
				</div>
			</div>
		</div>
	</div>
</article>			<article class="post-latest-card post tag-guides tag-ai-9">
	<a href="/autonomous-ai-agents/" class="global-link" aria-label="Autonomous AI Agents: Architecture and Risk Mitigation"></a>
	<div class="post-latest-card-container">
		<a href="/autonomous-ai-agents/" class="post-latest-card-image global-image">
			<img srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w400/2026/09/TL-2_BlogHeader_autonomous-ai-agents--1-.jpg 400w,
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/09/TL-2_BlogHeader_autonomous-ai-agents--1-.jpg 600w, 
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w800/2026/09/TL-2_BlogHeader_autonomous-ai-agents--1-.jpg 800w,
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/09/TL-2_BlogHeader_autonomous-ai-agents--1-.jpg 1200w"
	 sizes="(max-width:480px) 350px, 500px"
	 src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/09/TL-2_BlogHeader_autonomous-ai-agents--1-.jpg"
	 loading="lazy"
	 alt="Autonomous AI Agents: Architecture and Risk Mitigation">
		</a>
		<div class="post-latest-card-content">
			<div class="post-latest-card-tags global-tags">
								<a href="/tag/guides/">Guides</a><a href="/tag/ai-9/">AI</a>
			</div>
			<div class="post-latest-card-copy">
				<h3 class="post-latest-card-title"><a href="/autonomous-ai-agents/">Autonomous AI Agents: Architecture and Risk Mitigation</a></h3>
				<div class="post-latest-card-authors global-meta">
					<a href="/author/marketing/">n8n team</a>, <a href="/author/yulia/">Yulia Dmitrievna</a>
				</div>
			</div>
		</div>
	</div>
</article>			<article class="post-latest-card post">
	<a href="/introducing-n8n-assistant/" class="global-link" aria-label="Introducing n8n Assistant"></a>
	<div class="post-latest-card-container">
		<a href="/introducing-n8n-assistant/" class="post-latest-card-image global-image">
			<img srcset="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w400/2026/09/Blog-header---n8n-Assistant-Launch.jpg 400w,
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/09/Blog-header---n8n-Assistant-Launch.jpg 600w, 
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w800/2026/09/Blog-header---n8n-Assistant-Launch.jpg 800w,
			 https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w1200/2026/09/Blog-header---n8n-Assistant-Launch.jpg 1200w"
	 sizes="(max-width:480px) 350px, 500px"
	 src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/size/w600/2026/09/Blog-header---n8n-Assistant-Launch.jpg"
	 loading="lazy"
	 alt="Introducing n8n Assistant">
		</a>
		<div class="post-latest-card-content">
			<div class="post-latest-card-tags global-tags">
								
			</div>
			<div class="post-latest-card-copy">
				<h3 class="post-latest-card-title"><a href="/introducing-n8n-assistant/">Introducing n8n Assistant</a></h3>
				<div class="post-latest-card-authors global-meta">
					<a href="/author/paul/">Paul Gordon, n8n</a>
				</div>
			</div>
		</div>
	</div>
</article>		</div>
	</div>
</section>
-->

					<!-- <div class="subscribe-section">
	<div class="subscribe-wrap">
		<h3>Subscribe to new posts</h3>
		<form data-members-form="subscribe" class="subscribe-form">
			<input data-members-email type="email" placeholder="Your email address" aria-label="Your email address" required>
			<button class="global-button" type="submit">Subscribe</button>
		</form>
		<div class="subscribe-alert">
			<small class="alert-loading global-alert">Processing your application</small>
			<small class="alert-success global-alert">Please check your inbox and click the link to confirm your subscription</small>
			<small class="alert-error global-alert">There was an error sending the email</small>
		</div>
	</div>
</div>
 -->
				</main>
				<!-- Mobile Footer (hidden on desktop, shown on mobile) -->
<footer class="footer-mobile">
	<div class="footer-mobile-container">

		<!-- Logo and Tagline Section -->
		<div class="footer-mobile-brand">
			<div class="footer-mobile-logo">
				<a href="https://blog.n8n.io"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2025/09/n8n-Logo.svg" alt="n8n Blog"></a>
			</div>
			<p class="footer-mobile-tagline">Automate without limits</p>

			<!-- Social Icons -->
			<div class="footer-mobile-social">
				<a href="https://x.com/n8n_io" aria-label="X (Twitter)"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M23.954 4.569c-.885.389-1.83.654-2.825.775 1.014-.611 1.794-1.574 2.163-2.723-.951.555-2.005.959-3.127 1.184-.896-.959-2.173-1.559-3.591-1.559-2.717 0-4.92 2.203-4.92 4.917 0 .39.045.765.127 1.124C7.691 8.094 4.066 6.13 1.64 3.161c-.427.722-.666 1.561-.666 2.475 0 1.71.87 3.213 2.188 4.096-.807-.026-1.566-.248-2.228-.616v.061c0 2.385 1.693 4.374 3.946 4.827-.413.111-.849.171-1.296.171-.314 0-.615-.03-.916-.086.631 1.953 2.445 3.377 4.604 3.417-1.68 1.319-3.809 2.105-6.102 2.105-.39 0-.779-.023-1.17-.067 2.189 1.394 4.768 2.209 7.557 2.209 9.054 0 13.999-7.496 13.999-13.986 0-.209 0-.42-.015-.63.961-.689 1.8-1.56 2.46-2.548l-.047-.02z"/></svg></a>
				<a href="https://github.com/n8n-io/n8n" aria-label="GitHub"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M12 .297c-6.63 0-12 5.373-12 12 0 5.303 3.438 9.8 8.205 11.385.6.113.82-.258.82-.577 0-.285-.01-1.04-.015-2.04-3.338.724-4.042-1.61-4.042-1.61C4.422 18.07 3.633 17.7 3.633 17.7c-1.087-.744.084-.729.084-.729 1.205.084 1.838 1.236 1.838 1.236 1.07 1.835 2.809 1.305 3.495.998.108-.776.417-1.305.76-1.605-2.665-.3-5.466-1.332-5.466-5.93 0-1.31.465-2.38 1.235-3.22-.135-.303-.54-1.523.105-3.176 0 0 1.005-.322 3.3 1.23.96-.267 1.98-.399 3-.405 1.02.006 2.04.138 3 .405 2.28-1.552 3.285-1.23 3.285-1.23.645 1.653.24 2.873.12 3.176.765.84 1.23 1.91 1.23 3.22 0 4.61-2.805 5.625-5.475 5.92.42.36.81 1.096.81 2.22 0 1.606-.015 2.896-.015 3.286 0 .315.21.69.825.57C20.565 22.092 24 17.592 24 12.297c0-6.627-5.373-12-12-12"/></svg></a>
				<a href="https://discord.gg/n8n" aria-label="Discord"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M20.222 0c1.406 0 2.54 1.137 2.607 2.475V24l-2.677-2.273-1.47-1.338-1.604-1.398.67 2.205H3.71c-1.402 0-2.54-1.065-2.54-2.476V2.48C1.17 1.142 2.31.003 3.715.003h16.5L20.222 0zm-6.118 5.683h-.03l-.202.2c2.073.6 3.076 1.537 3.076 1.537-1.336-.668-2.54-1.002-3.744-1.137-.87-.135-1.74-.064-2.475 0h-.2c-.47 0-1.47.2-2.81.735-.467.203-.735.336-.735.336s1.002-1.002 3.21-1.537l-.135-.135s-1.672-.064-3.477 1.27c0 0-1.805 3.144-1.805 7.02 0 0 1 1.74 3.743 1.806 0 0 .4-.533.805-1.002-1.54-.468-2.14-1.404-2.14-1.404s.134.066.335.2h.06c.03 0 .044.015.06.03v.006c.016.016.03.03.06.03.33.136.66.27.93.4.466.202 1.065.403 1.8.536.93.135 1.996.2 3.21 0 .6-.135 1.2-.267 1.8-.535.39-.2.87-.4 1.397-.737 0 0-.6.936-2.205 1.404.33.466.795 1 .795 1 2.744-.06 3.81-1.8 3.87-1.726 0-3.87-1.815-7.02-1.815-7.02-1.635-1.214-3.165-1.26-3.435-1.26l.056-.02zm.168 4.413c.703 0 1.27.6 1.27 1.335 0 .74-.57 1.34-1.27 1.34-.7 0-1.27-.6-1.27-1.334.002-.74.573-1.338 1.27-1.338zm-4.543 0c.7 0 1.266.6 1.266 1.335 0 .74-.57 1.34-1.27 1.34-.7 0-1.27-.6-1.27-1.334 0-.74.57-1.338 1.27-1.338z"/></svg></a>
				<a href="https://www.linkedin.com/company/n8n/" aria-label="LinkedIn"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M20.447 20.452h-3.554v-5.569c0-1.328-.027-3.037-1.852-3.037-1.853 0-2.136 1.445-2.136 2.939v5.667H9.351V9h3.414v1.561h.046c.477-.9 1.637-1.85 3.37-1.85 3.601 0 4.267 2.37 4.267 5.455v6.286zM5.337 7.433c-1.144 0-2.063-.926-2.063-2.065 0-1.138.92-2.063 2.063-2.063 1.14 0 2.064.925 2.064 2.063 0 1.139-.925 2.065-2.064 2.065zm1.782 13.019H3.555V9h3.564v11.452zM22.225 0H1.771C.792 0 0 .774 0 1.729v20.542C0 23.227.792 24 1.771 24h20.451C23.2 24 24 23.227 24 22.271V1.729C24 .774 23.2 0 22.222 0h.003z"/></svg></a>
				<a href="https://www.youtube.com/c/n8n-io" aria-label="YouTube"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path class="a" d="M23.495 6.205a3.007 3.007 0 0 0-2.088-2.088c-1.87-.501-9.396-.501-9.396-.501s-7.507-.01-9.396.501A3.007 3.007 0 0 0 .527 6.205a31.247 31.247 0 0 0-.522 5.805 31.247 31.247 0 0 0 .522 5.783 3.007 3.007 0 0 0 2.088 2.088c1.868.502 9.396.502 9.396.502s7.506 0 9.396-.502a3.007 3.007 0 0 0 2.088-2.088 31.247 31.247 0 0 0 .5-5.783 31.247 31.247 0 0 0-.5-5.805zM9.609 15.601V8.408l6.264 3.602z"/></svg></a>
				

			</div>
		</div>

		<!-- Primary Links Section -->
		<div class="footer-mobile-links">
			<div class="footer-mobile-section">
				<ul class="footer-mobile-list">
					<li><a href="https://n8n.io/careers">Careers</a></li>
					<li><a href="https://n8n.io/contact">Contact</a></li>
					<li><a href="https://merch.n8n.io/">Merch</a></li>
					<li><a href="https://n8n.io/press">Press</a></li>
					<li><a href="https://n8n.io/security">Security</a></li>
				</ul>
			</div>

			<!-- Secondary Links Section -->
			<div class="footer-mobile-section">
				<ul class="footer-mobile-list">
					<li><a href="https://n8n.io/case-studies">Case studies</a></li>
					<li><a href="https://n8n.io/vs/zapier/">Zapier vs n8n</a></li>
					<li><a href="https://n8n.io/vs/make/">Make vs n8n</a></li>
					<li><a href="https://n8n.io/workflows/160-convert-xml-to-json/">XML to JSON converter</a></li>
				</ul>
			</div>

			<!-- Tertiary Links Section -->
			<div class="footer-mobile-section">
				<ul class="footer-mobile-list">
					<li><a href="https://n8n.io/affiliates/">Affiliate program</a></li>
					<li><a href="https://n8n.io/become-an-expert">Become an expert</a></li>
					<li><a href="https://luma.com/n8n-events">Events</a></li>
				</ul>
			</div>
		</div>

		<!-- Dropdown Sections -->
		<div class="footer-mobile-dropdowns">
			<div class="footer-mobile-dropdown">
				<button class="footer-dropdown-toggle" data-dropdown="popular-integrations">
					<span>Popular integrations</span>
					<svg class="dropdown-chevron" width="20" height="20" viewBox="0 0 20 20" fill="none">
						<path d="M5 7.5L10 12.5L15 7.5" stroke="currentColor" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
					</svg>
				</button>
				<div class="footer-dropdown-content" id="popular-integrations">
					<ul>
						<li><a href="https://n8n.io/integrations/google-sheets">Google Sheets</a></li>
						<li><a href="https://n8n.io/integrations/telegram">Telegram</a></li>
						<li><a href="https://n8n.io/integrations/mysql">MySQL</a></li>
						<li><a href="https://n8n.io/integrations/slack">Slack</a></li>
						<li><a href="https://n8n.io/integrations/discord">Discord</a></li>
						<li><a href="https://n8n.io/integrations/postgres">PostgreSQL</a></li>
					</ul>
					<a href="https://n8n.io/integrations" class="footer-show-more">Show more</a>
				</div>
			</div>

			<div class="footer-mobile-dropdown">
				<button class="footer-dropdown-toggle" data-dropdown="trending-combinations">
					<span>Trending combinations</span>
					<svg class="dropdown-chevron" width="20" height="20" viewBox="0 0 20 20" fill="none">
						<path d="M5 7.5L10 12.5L15 7.5" stroke="currentColor" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
					</svg>
				</button>
				<div class="footer-dropdown-content" id="trending-combinations">
					<ul>
						<li><a href="https://n8n.io/workflows">Google Sheets + Slack</a></li>
						<li><a href="https://n8n.io/workflows">Discord + MySQL</a></li>
						<li><a href="https://n8n.io/workflows">Telegram + PostgreSQL</a></li>
					</ul>
					<a href="https://n8n.io/workflows" class="footer-show-more">Show more</a>
				</div>
			</div>

			<div class="footer-mobile-dropdown">
				<button class="footer-dropdown-toggle" data-dropdown="technical-teams">
					<span>Used by technical teams</span>
					<svg class="dropdown-chevron" width="20" height="20" viewBox="0 0 20 20" fill="none">
						<path d="M5 7.5L10 12.5L15 7.5" stroke="currentColor" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
					</svg>
				</button>
				<div class="footer-dropdown-content" id="technical-teams">
					<ul>
						<li><a href="https://n8n.io/use-cases/development">Development Teams</a></li>
						<li><a href="https://n8n.io/use-cases/devops">DevOps</a></li>
						<li><a href="https://n8n.io/use-cases/infrastructure">Infrastructure</a></li>
					</ul>
					<a href="https://n8n.io/use-cases" class="footer-show-more">Show more</a>
				</div>
			</div>

			<div class="footer-mobile-dropdown">
				<button class="footer-dropdown-toggle" data-dropdown="integration-categories">
					<span>Top integration categories</span>
					<svg class="dropdown-chevron" width="20" height="20" viewBox="0 0 20 20" fill="none">
						<path d="M5 7.5L10 12.5L15 7.5" stroke="currentColor" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
					</svg>
				</button>
				<div class="footer-dropdown-content" id="integration-categories">
					<ul>
						<li><a href="https://n8n.io/integrations/databases">Databases</a></li>
						<li><a href="https://n8n.io/integrations/communication">Communication</a></li>
						<li><a href="https://n8n.io/integrations/productivity">Productivity</a></li>
					</ul>
					<a href="https://n8n.io/integrations" class="footer-show-more">Show more</a>
				</div>
			</div>

			<div class="footer-mobile-dropdown">
				<button class="footer-dropdown-toggle" data-dropdown="trending-templates">
					<span>Trending templates</span>
					<svg class="dropdown-chevron" width="20" height="20" viewBox="0 0 20 20" fill="none">
						<path d="M5 7.5L10 12.5L15 7.5" stroke="currentColor" stroke-width="1.5" stroke-linecap="round" stroke-linejoin="round"/>
					</svg>
				</button>
				<div class="footer-dropdown-content" id="trending-templates">
					<ul>
						<li><a href="https://n8n.io/workflows">Automated Data Sync</a></li>
						<li><a href="https://n8n.io/workflows">Lead Generation</a></li>
						<li><a href="https://n8n.io/workflows">Customer Support</a></li>
					</ul>
					<a href="https://n8n.io/workflows" class="footer-show-more">Show more</a>
				</div>
			</div>
		</div>

		<!-- Legal Links and Copyright -->
		<div class="footer-mobile-bottom">
			<div class="footer-mobile-legal">
				<a href="https://n8n.io/imprint/">Imprint</a>
				<span class="footer-separator">|</span>
				<a href="https://n8n.io/legal/security/">Security</a>
				<span class="footer-separator">|</span>
				<a href="https://n8n.io/legal/privacy/">Privacy</a>
				<span class="footer-separator">|</span>
				<a href="https://n8n.io/legal/vulnerability-disclosure-policy/">Report a vulnerability</a>
			</div>
			<div class="footer-mobile-copyright">
				<p>© 2026 n8n | All rights reserved.</p>
			</div>
		</div>
	</div>
</footer>
<!-- Desktop Footer (shown on desktop, hidden on mobile) -->
<footer class="blog-footer footer-desktop">
	<div class="blog-footer-container">

		<!-- Logo and Social Section -->
		<div class="footer-brand">
			<div class="footer-brand-left">
				<div class="footer-logo">
					<a href="https://blog.n8n.io"><img src="https://storage.ghost.io/c/0d/78/0d78b34c-0c5f-4975-900e-61d00ccb1c2d/content/images/2025/09/n8n-Logo.svg" alt="n8n Blog"></a>
				</div>
				<p class="footer-tagline">Automate without limits</p>
				<div class="footer-social">
					<a href="https://x.com/n8n_io" aria-label="X (Twitter)"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M23.954 4.569c-.885.389-1.83.654-2.825.775 1.014-.611 1.794-1.574 2.163-2.723-.951.555-2.005.959-3.127 1.184-.896-.959-2.173-1.559-3.591-1.559-2.717 0-4.92 2.203-4.92 4.917 0 .39.045.765.127 1.124C7.691 8.094 4.066 6.13 1.64 3.161c-.427.722-.666 1.561-.666 2.475 0 1.71.87 3.213 2.188 4.096-.807-.026-1.566-.248-2.228-.616v.061c0 2.385 1.693 4.374 3.946 4.827-.413.111-.849.171-1.296.171-.314 0-.615-.03-.916-.086.631 1.953 2.445 3.377 4.604 3.417-1.68 1.319-3.809 2.105-6.102 2.105-.39 0-.779-.023-1.17-.067 2.189 1.394 4.768 2.209 7.557 2.209 9.054 0 13.999-7.496 13.999-13.986 0-.209 0-.42-.015-.63.961-.689 1.8-1.56 2.46-2.548l-.047-.02z"/></svg></a>
					<a href="https://github.com/n8n-io/n8n" aria-label="GitHub"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M12 .297c-6.63 0-12 5.373-12 12 0 5.303 3.438 9.8 8.205 11.385.6.113.82-.258.82-.577 0-.285-.01-1.04-.015-2.04-3.338.724-4.042-1.61-4.042-1.61C4.422 18.07 3.633 17.7 3.633 17.7c-1.087-.744.084-.729.084-.729 1.205.084 1.838 1.236 1.838 1.236 1.07 1.835 2.809 1.305 3.495.998.108-.776.417-1.305.76-1.605-2.665-.3-5.466-1.332-5.466-5.93 0-1.31.465-2.38 1.235-3.22-.135-.303-.54-1.523.105-3.176 0 0 1.005-.322 3.3 1.23.96-.267 1.98-.399 3-.405 1.02.006 2.04.138 3 .405 2.28-1.552 3.285-1.23 3.285-1.23.645 1.653.24 2.873.12 3.176.765.84 1.23 1.91 1.23 3.22 0 4.61-2.805 5.625-5.475 5.92.42.36.81 1.096.81 2.22 0 1.606-.015 2.896-.015 3.286 0 .315.21.69.825.57C20.565 22.092 24 17.592 24 12.297c0-6.627-5.373-12-12-12"/></svg></a>
					<a href="https://discord.gg/n8n" aria-label="Discord"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M20.222 0c1.406 0 2.54 1.137 2.607 2.475V24l-2.677-2.273-1.47-1.338-1.604-1.398.67 2.205H3.71c-1.402 0-2.54-1.065-2.54-2.476V2.48C1.17 1.142 2.31.003 3.715.003h16.5L20.222 0zm-6.118 5.683h-.03l-.202.2c2.073.6 3.076 1.537 3.076 1.537-1.336-.668-2.54-1.002-3.744-1.137-.87-.135-1.74-.064-2.475 0h-.2c-.47 0-1.47.2-2.81.735-.467.203-.735.336-.735.336s1.002-1.002 3.21-1.537l-.135-.135s-1.672-.064-3.477 1.27c0 0-1.805 3.144-1.805 7.02 0 0 1 1.74 3.743 1.806 0 0 .4-.533.805-1.002-1.54-.468-2.14-1.404-2.14-1.404s.134.066.335.2h.06c.03 0 .044.015.06.03v.006c.016.016.03.03.06.03.33.136.66.27.93.4.466.202 1.065.403 1.8.536.93.135 1.996.2 3.21 0 .6-.135 1.2-.267 1.8-.535.39-.2.87-.4 1.397-.737 0 0-.6.936-2.205 1.404.33.466.795 1 .795 1 2.744-.06 3.81-1.8 3.87-1.726 0-3.87-1.815-7.02-1.815-7.02-1.635-1.214-3.165-1.26-3.435-1.26l.056-.02zm.168 4.413c.703 0 1.27.6 1.27 1.335 0 .74-.57 1.34-1.27 1.34-.7 0-1.27-.6-1.27-1.334.002-.74.573-1.338 1.27-1.338zm-4.543 0c.7 0 1.266.6 1.266 1.335 0 .74-.57 1.34-1.27 1.34-.7 0-1.27-.6-1.27-1.334 0-.74.57-1.338 1.27-1.338z"/></svg></a>
					<a href="https://www.linkedin.com/company/n8n/" aria-label="LinkedIn"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path d="M20.447 20.452h-3.554v-5.569c0-1.328-.027-3.037-1.852-3.037-1.853 0-2.136 1.445-2.136 2.939v5.667H9.351V9h3.414v1.561h.046c.477-.9 1.637-1.85 3.37-1.85 3.601 0 4.267 2.37 4.267 5.455v6.286zM5.337 7.433c-1.144 0-2.063-.926-2.063-2.065 0-1.138.92-2.063 2.063-2.063 1.14 0 2.064.925 2.064 2.063 0 1.139-.925 2.065-2.064 2.065zm1.782 13.019H3.555V9h3.564v11.452zM22.225 0H1.771C.792 0 0 .774 0 1.729v20.542C0 23.227.792 24 1.771 24h20.451C23.2 24 24 23.227 24 22.271V1.729C24 .774 23.2 0 22.222 0h.003z"/></svg></a>
					<a href="https://www.youtube.com/c/n8n-io" aria-label="YouTube"><svg role="img" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg"><path class="a" d="M23.495 6.205a3.007 3.007 0 0 0-2.088-2.088c-1.87-.501-9.396-.501-9.396-.501s-7.507-.01-9.396.501A3.007 3.007 0 0 0 .527 6.205a31.247 31.247 0 0 0-.522 5.805 31.247 31.247 0 0 0 .522 5.783 3.007 3.007 0 0 0 2.088 2.088c1.868.502 9.396.502 9.396.502s7.506 0 9.396-.502a3.007 3.007 0 0 0 2.088-2.088 31.247 31.247 0 0 0 .5-5.783 31.247 31.247 0 0 0-.5-5.805zM9.609 15.601V8.408l6.264 3.602z"/></svg></a>
					

				</div>
			</div>
			<div class="footer-brand-right">
				<ul class="footer-primary-links">
					<li><a href="https://n8n.io/">n8n home</a></li>
					<li><a href="https://n8n.io/features/">Features</a></li>
					<li><a href="https://n8n.io/pricing/">Pricing</a></li>
					<li><a href="https://docs.n8n.io/">Docs</a></li>
					<li><a href="https://community.n8n.io/">Community</a></li>
				</ul>
			</div>
		</div>

		<!-- Bottom Section -->
		<div class="footer-bottom">
			<div class="footer-legal">
				<a href="https://n8n.io/imprint/">Imprint</a>
				<span class="footer-separator">|</span>
				<a href="https://n8n.io/legal/security/">Security</a>
				<span class="footer-separator">|</span>
				<a href="https://n8n.io/legal/privacy/">Privacy</a>
				<span class="footer-separator">|</span>
				<a href="https://n8n.io/legal/vulnerability-disclosure-policy/">Report a vulnerability</a>
			</div>
			<div class="footer-copyright">
				<p>© 2026 n8n | All rights reserved.</p>
			</div>
		</div>
	</div>
</footer>
			</div>
		</div>
		<div id="notifications" class="global-notification">
	<div class="subscribe">You’ve successfully subscribed to n8n Blog</div>
	<div class="signin">Welcome back! You’ve successfully signed in.</div>
	<div class="signup">Great! You’ve successfully signed up.</div>
	<div class="expired">Your link has expired</div>
	<div class="checkout-success">Success! Check your email for magic link to sign-in.</div>
</div>
				<script src="https://blog.n8n.io/assets/js/global.js?v=Wrm8OemR0lUQ9qP5"></script>
		<script src="https://blog.n8n.io/assets/js/post.js?v=mR1N_JBqRdiTUgjo"></script>
		<script src="https://blog.n8n.io/assets/js/toc.js?v=oPtDPMixJsjbTkh9"></script>
		
		<script>
!function(){"use strict";const p=new URLSearchParams(window.location.search),isAction=p.has("action"),isStripe=p.has("stripe"),success=p.get("success"),action=p.get("action"),stripe=p.get("stripe"),n=document.getElementById("notifications"),a="is-subscribe",b="is-signin",c="is-signup",d="is-expired",e="is-checkout-success";

function hideNotification() {
    n.classList.remove(a,b,c,d,e);
    window.history.replaceState(null,null,window.location.pathname);
}

if(p&&(isAction||isStripe)) {
    if(isAction) {
        if(action=="subscribe"&&success=="true") n.classList.add(a);
        if(action=="signin"&&success=="true") n.classList.add(b);
        if(action=="signup"&&success=="true") n.classList.add(c);
        if(success=="false") n.classList.add(d);
    }
    if(isStripe&&stripe=="success") n.classList.add(e);
    
    // Auto hide after 5 seconds
    setTimeout(hideNotification, 5000);
    
    // Click to close
    n.addEventListener('click', hideNotification);
}}();
</script>

		
		<script src="https://blog.n8n.io/assets/js/external-links.js?v=DikCkQ8Bvh7jO-3I"></script>
		
	</body>
</html>