<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>AI Agent on Blowing in the wind</title>
    <link>https://zheng-bobo.github.io/en/tags/ai-agent/</link>
    <description>Recent content in AI Agent on Blowing in the wind</description>
    <generator>Hugo -- gohugo.io</generator>
    <language>en</language>
    <lastBuildDate>Sat, 26 Sep 2026 23:00:00 +0200</lastBuildDate>

  <atom:link href="https://zheng-bobo.github.io/en/tags/ai-agent/index.xml" rel="self" type="application/rss+xml" />


    <item>
      <title>A Field Guide to LLM and AI Agent Benchmarks</title>
      <link>https://zheng-bobo.github.io/en/post/llm-agent-benchmarks-guide/</link>
      <pubDate>Sat, 26 Sep 2026 23:00:00 +0200</pubDate>

      <guid>https://zheng-bobo.github.io/en/post/llm-agent-benchmarks-guide/</guid>
      <description>&lt;p&gt;Model reports often list ARC-E, ARC-C, MMLU, GPQA, GSM8K, HumanEval, SWE-bench, GAIA, WebArena, and OSWorld side by side. Their scores are not interchangeable: each benchmark uses different tasks, tools, environments, inference budgets, and scoring rules.&lt;/p&gt;

&lt;p&gt;This guide maps common evaluations from static question answering to agents completing real tasks, and explains what each benchmark can—and cannot—tell us.&lt;/p&gt;</description>
    </item>

  </channel>
</rss>