<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Benchmark</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Benchmark feed</description>
    <item>
      <title>AWS Releases aws-bench to Evaluate Agents on Cloud Tasks</title>
      <link>https://www.infoq.com/news/2026/08/aws-bench-agent-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Benchmark</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/aws-bench-agent-evaluation/en/headerimage/generatedHeaderImage-1787307655540.jpg"/&gt;&lt;p&gt;AWS has released aws-bench, an open-source benchmark for evaluating AI agents on real AWS tasks such as misconfigurations and infrastructure provisioning. Unlike traditional benchmarks, it uses real resources in disposable AWS accounts, scoring agent performance through automated verifiers.&lt;/p&gt; &lt;i&gt;By Gianmarco Nalin&lt;/i&gt;</description>
      <category>Agents</category>
      <category>AI Development</category>
      <category>Open Source</category>
      <category>Large language models</category>
      <category>Cloud</category>
      <category>Benchmark</category>
      <category>AWS</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>news</category>
      <pubDate>Sat, 22 Aug 2026 08:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/aws-bench-agent-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Benchmark</guid>
      <dc:creator>Gianmarco Nalin</dc:creator>
      <dc:date>2026-08-22T08:00:00Z</dc:date>
      <dc:identifier>/news/2026/08/aws-bench-agent-evaluation/en</dc:identifier>
    </item>
    <item>
      <title>Presentation: From Fab To Token - The State Of The Market</title>
      <link>https://www.infoq.com/presentations/ai-hardware-tokenomics/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Benchmark</link>
      <description>&lt;img src="https://res.infoq.com/presentations/ai-hardware-tokenomics/en/mediumimage/jordan-nanos-medium-1786538855211.jpg"/&gt;&lt;p&gt;Jordan Nanos discusses how semiconductor constraints, data center expansion, and networking bottlenecks impact AI software architecture. Drawing from SemiAnalysis research, he shares insights on benchmark performance, GPU scaling, and tokenomics from chip fab to model inference.&lt;/p&gt; &lt;i&gt;By Jordan Nanos&lt;/i&gt;</description>
      <category>QCon AI Boston 2026</category>
      <category>AI Architecture</category>
      <category>AI Security</category>
      <category>GPU</category>
      <category>Large language models</category>
      <category>Model Inference</category>
      <category>Hardware</category>
      <category>Benchmark</category>
      <category>Infrastructure</category>
      <category>Performance</category>
      <category>Data Analytics</category>
      <category>Transcripts</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Tue, 18 Aug 2026 16:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/ai-hardware-tokenomics/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Benchmark</guid>
      <dc:creator>Jordan Nanos</dc:creator>
      <dc:date>2026-08-18T16:00:00Z</dc:date>
      <dc:identifier>/presentations/ai-hardware-tokenomics/en</dc:identifier>
    </item>
  </channel>
</rss>
