<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Benchmark - News</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Benchmark News feed</description>
    <item>
      <title>AWS Releases aws-bench to Evaluate Agents on Cloud Tasks</title>
      <link>https://www.infoq.com/news/2026/08/aws-bench-agent-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Benchmark-news</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/aws-bench-agent-evaluation/en/headerimage/generatedHeaderImage-1787307655540.jpg"/&gt;&lt;p&gt;AWS has released aws-bench, an open-source benchmark for evaluating AI agents on real AWS tasks such as misconfigurations and infrastructure provisioning. Unlike traditional benchmarks, it uses real resources in disposable AWS accounts, scoring agent performance through automated verifiers.&lt;/p&gt; &lt;i&gt;By Gianmarco Nalin&lt;/i&gt;</description>
      <category>AWS</category>
      <category>Open Source</category>
      <category>Cloud</category>
      <category>Benchmark</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>AI Development</category>
      <category>Development</category>
      <category>Architecture &amp; Design</category>
      <category>news</category>
      <pubDate>Sat, 22 Aug 2026 08:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/aws-bench-agent-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Benchmark-news</guid>
      <dc:creator>Gianmarco Nalin</dc:creator>
      <dc:date>2026-08-22T08:00:00Z</dc:date>
      <dc:identifier>/news/2026/08/aws-bench-agent-evaluation/en</dc:identifier>
    </item>
  </channel>
</rss>
