<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Large language models - News</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Large language models News feed</description>
    <item>
      <title>AI Code Review at Scale: LinkedIn's Multi-Agent Approach</title>
      <link>https://www.infoq.com/news/2026/08/linkedin-ai-code-review/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/linkedin-ai-code-review/en/headerimage/linkedin-code-review-1787387463447.jpeg"/&gt;&lt;p&gt;At LinkedIn's scale, relying solely on human reviewers or simply putting an off-the-shelf AI reviewer in front of GitHub is not an effective way to manage PRs. To address this, LinkedIn engineers built a multi-agent AI code review platform that understands the organizationâ€™s coding context, treats code review as production infrastructure, and minimizes hallucinations and low-signal feedback.&lt;/p&gt; &lt;i&gt;By Sergio De Simone&lt;/i&gt;</description>
      <category>LinkedIn</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>Kubernetes</category>
      <category>Code Reviews</category>
      <category>Development</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>DevOps</category>
      <category>news</category>
      <pubDate>Sat, 22 Aug 2026 09:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/linkedin-ai-code-review/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</guid>
      <dc:creator>Sergio De Simone</dc:creator>
      <dc:date>2026-08-22T09:00:00Z</dc:date>
      <dc:identifier>/news/2026/08/linkedin-ai-code-review/en</dc:identifier>
    </item>
    <item>
      <title>AWS Releases Aws-Bench to Evaluate Agents on Cloud Tasks</title>
      <link>https://www.infoq.com/news/2026/08/aws-bench-agent-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/aws-bench-agent-evaluation/en/headerimage/generatedHeaderImage-1787307655540.jpg"/&gt;&lt;p&gt;AWS has released aws-bench, an open-source benchmark for evaluating AI agents on real AWS tasks such as misconfigurations and infrastructure provisioning. Unlike traditional benchmarks, it uses real resources in disposable AWS accounts, scoring agent performance through automated verifiers.&lt;/p&gt; &lt;i&gt;By Gianmarco Nalin&lt;/i&gt;</description>
      <category>AWS</category>
      <category>Open Source</category>
      <category>Cloud</category>
      <category>Benchmark</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>AI Development</category>
      <category>Development</category>
      <category>Architecture &amp; Design</category>
      <category>news</category>
      <pubDate>Sat, 22 Aug 2026 08:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/aws-bench-agent-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</guid>
      <dc:creator>Gianmarco Nalin</dc:creator>
      <dc:date>2026-08-22T08:00:00Z</dc:date>
      <dc:identifier>/news/2026/08/aws-bench-agent-evaluation/en</dc:identifier>
    </item>
    <item>
      <title>The Open-Sourcing of DeepSeek Harness Opens the Door to Modular, Unbundled AI Agent Infrastructure</title>
      <link>https://www.infoq.com/news/2026/08/deep-seek-harness/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;DeepSeek has released a developer preview of DeepSeek Harness (dsh), an open-source execution runtime for building autonomous AI agents. The software features a micro-kernel architecture with modular plugins for various functional units. The release includes an append-only event logging system for tracking execution activities. Adoption may depend on plugin ecosystem stability and API maintenance.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>AI Coding</category>
      <category>Large language models</category>
      <category>AI Assisted Coding</category>
      <category>Agentic AI Architecture</category>
      <category>AI Harness</category>
      <category>Development</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>news</category>
      <pubDate>Thu, 20 Aug 2026 05:05:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/deep-seek-harness/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-08-20T05:05:00Z</dc:date>
      <dc:identifier>/news/2026/08/deep-seek-harness/en</dc:identifier>
    </item>
    <item>
      <title>SpaceXAI Launches Grok Bot for Autonomous AI Agents</title>
      <link>https://www.infoq.com/news/2026/08/grok-bot-agent/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/grok-bot-agent/en/headerimage/generatedHeaderImage-1786979342385.jpg"/&gt;&lt;p&gt;SpaceXAI has introduced Grok Bot, a system of persistent AI agents that operate on dedicated cloud computers and can interact with websites, applications, inboxes, and other tools.&lt;/p&gt; &lt;i&gt;By Daniel Dominguez&lt;/i&gt;</description>
      <category>Anthropic</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>Artificial Intelligence</category>
      <category>Agentic AI Architecture</category>
      <category>Software Development</category>
      <category>OpenAI</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>news</category>
      <pubDate>Mon, 17 Aug 2026 18:02:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/grok-bot-agent/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</guid>
      <dc:creator>Daniel Dominguez</dc:creator>
      <dc:date>2026-08-17T18:02:00Z</dc:date>
      <dc:identifier>/news/2026/08/grok-bot-agent/en</dc:identifier>
    </item>
    <item>
      <title>Meta Open-Sources Muse Glimmer: a 30B Local Agentic Model Optimised for On-Device Execution</title>
      <link>https://www.infoq.com/news/2026/08/meta-muse-glimmer/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Meta AI Research has introduced Muse Glimmer, a 30-billion-parameter open-weight model under the Apache 2.0 license, designed for local workflows. It enables autonomous agents and complex task execution on consumer GPUs without relying on cloud APIs. The model employs a multi-stage training approach for efficient performance and supports multimodal inputs, enhancing coding and automation tasks.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>Foundation Model</category>
      <category>Large language models</category>
      <category>AI Architecture</category>
      <category>Local Inference</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>news</category>
      <pubDate>Fri, 14 Aug 2026 05:05:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/meta-muse-glimmer/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-08-14T05:05:00Z</dc:date>
      <dc:identifier>/news/2026/08/meta-muse-glimmer/en</dc:identifier>
    </item>
    <item>
      <title>Anthropic's Claude Breaches Sandbox During Model Security Evaluations</title>
      <link>https://www.infoq.com/news/2026/08/claude-sandox-breach/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Anthropic conducted an audit of 141006 evaluation runs after OpenAI's sandbox escape disclosure. The review identified three incidents where Claude models accessed the internet due to misconfigurations. These incidents involved unauthorised attacks on live targets. Anthropic has suspended offensive evaluations and plans to enhance security measures and collaborate with external auditors.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>Frontier Model</category>
      <category>Cloud Security</category>
      <category>Security Breach</category>
      <category>Large language models</category>
      <category>AI Security</category>
      <category>Model Evaluation</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>DevOps</category>
      <category>news</category>
      <pubDate>Thu, 13 Aug 2026 10:10:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/claude-sandox-breach/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models-news</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-08-13T10:10:00Z</dc:date>
      <dc:identifier>/news/2026/08/claude-sandox-breach/en</dc:identifier>
    </item>
  </channel>
</rss>
