<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Large language models</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Large language models feed</description>
    <item>
      <title>Presentation: From AI Agent Demo to Production: Automated Testing and Evaluation</title>
      <link>https://www.infoq.com/presentations/ai-agent-testing-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/presentations/ai-agent-testing-evaluation/en/mediumimage/zhou-yu-medium-1787813732120.jpeg"/&gt;&lt;p&gt;Zhou Yu discusses why AI agents stall in demo phase and shares how simulation-driven testing solves compliance and reliability bottlenecks. Learn how Columbia and Arklex AI use synthetic user personas, trajectory entropy, and automated CI/CD pipelines to evaluate multi-turn agents, catch edge cases before deployment, and scale self-learning workflows in production.&lt;/p&gt; &lt;i&gt;By Zhou Yu&lt;/i&gt;</description>
      <category>Transcripts</category>
      <category>Reliability</category>
      <category>Testing</category>
      <category>QCon AI Boston 2026</category>
      <category>Large language models</category>
      <category>Quality</category>
      <category>Performance Evaluation</category>
      <category>Agents</category>
      <category>Simulation</category>
      <category>Performance</category>
      <category>Automated testing</category>
      <category>Data</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Mon, 07 Sep 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/ai-agent-testing-evaluation/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Zhou Yu</dc:creator>
      <dc:date>2026-09-07T11:00:00Z</dc:date>
      <dc:identifier>/presentations/ai-agent-testing-evaluation/en</dc:identifier>
    </item>
    <item>
      <title>Google Mantis: an Agentic Vulnerability Scanning Harness for Reducing False Positives</title>
      <link>https://www.infoq.com/news/2026/09/google-mantis-vulnerability-scan/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/09/google-mantis-vulnerability-scan/en/headerimage/google-mantis-scanner-1788693601725.jpeg"/&gt;&lt;p&gt;Google has open-sourced Mantis, an AI-agent framework designed to automate the software vulnerability lifecycle, from identifying and validating vulnerabilities to reproducing and fixing them. Google says it developed Mantis to address the high rate of false positives and hallucinated vulnerabilities produced by conventional AI-powered code scanning.&lt;/p&gt; &lt;i&gt;By Sergio De Simone&lt;/i&gt;</description>
      <category>Google</category>
      <category>Open Source</category>
      <category>Security Vulnerabilities</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>news</category>
      <pubDate>Sun, 06 Sep 2026 12:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/09/google-mantis-vulnerability-scan/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Sergio De Simone</dc:creator>
      <dc:date>2026-09-06T12:00:00Z</dc:date>
      <dc:identifier>/news/2026/09/google-mantis-vulnerability-scan/en</dc:identifier>
    </item>
    <item>
      <title>Shopify Introduces Gisting: Compressing LLM System Prompts into Learned Tokens</title>
      <link>https://www.infoq.com/news/2026/09/spotify-gisting-llm-performance/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/09/spotify-gisting-llm-performance/en/headerimage/spotify-app-size-growth-process-1788462838475.jpeg"/&gt;&lt;p&gt;Shopify's engineering introduced gisting, a novel technique for compressing long LLM prompts into a smaller set of learned "gist" tokens, improving throughput and reducing inference cost.&lt;/p&gt; &lt;i&gt;By Sergio De Simone&lt;/i&gt;</description>
      <category>Model Inference</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>Performance</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>news</category>
      <pubDate>Thu, 03 Sep 2026 20:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/09/spotify-gisting-llm-performance/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Sergio De Simone</dc:creator>
      <dc:date>2026-09-03T20:00:00Z</dc:date>
      <dc:identifier>/news/2026/09/spotify-gisting-llm-performance/en</dc:identifier>
    </item>
    <item>
      <title>Cohere’s Parse 5 Promises Efficient Multi-Modal Information Extraction from Complex Documents</title>
      <link>https://www.infoq.com/news/2026/09/cohere-multimodal-parse/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Cohere has launched Parse 5, a multimodal foundation model designed to extract structured data from complex enterprise documents. The 2.3-billion-parameter system converts visually rich PDFs into Markdown while providing bounding box coordinates for visual grounding. It has been evaluated against over 2,000 enterprise pages, achieving an average score of 79.2 in key performance areas.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>OCR</category>
      <category>Large language models</category>
      <category>Machine Learning</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>news</category>
      <pubDate>Thu, 03 Sep 2026 06:06:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/09/cohere-multimodal-parse/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-09-03T06:06:00Z</dc:date>
      <dc:identifier>/news/2026/09/cohere-multimodal-parse/en</dc:identifier>
    </item>
    <item>
      <title>Presentation: Beyond Prompting: Context Engineering for Production-Grade AI</title>
      <link>https://www.infoq.com/presentations/context-engineering-redis-llm-architecture/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/presentations/context-engineering-redis-llm-architecture/en/mediumimage/ricardo-ferreira-medium-1787820318843.jpg"/&gt;&lt;p&gt;Ricardo Ferreira discusses moving beyond simple prompt engineering to build production-grade AI applications. He shares practical architectural strategies for integrating long-term and short-term memory using Redis, managing LLM token limits via summarization, mitigating context rot with reranking and semantic caching, and controlling exponential API costs under strict latency constraints.&lt;/p&gt; &lt;i&gt;By Ricardo Ferreira&lt;/i&gt;</description>
      <category>Retrieval-Augmented Generation</category>
      <category>Transcripts</category>
      <category>Architecture</category>
      <category>Redis</category>
      <category>Caching</category>
      <category>Generative AI</category>
      <category>QCon AI Boston 2026</category>
      <category>LangChain4j</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Wed, 02 Sep 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/context-engineering-redis-llm-architecture/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Ricardo Ferreira</dc:creator>
      <dc:date>2026-09-02T11:00:00Z</dc:date>
      <dc:identifier>/presentations/context-engineering-redis-llm-architecture/en</dc:identifier>
    </item>
    <item>
      <title>OpenClaw 2.0 Releases with Simplified Setup and Collaborative Agents</title>
      <link>https://www.infoq.com/news/2026/09/openclaw-2-release/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/09/openclaw-2-release/en/headerimage/generatedHeaderImage-1788278004063.jpg"/&gt;&lt;p&gt;OpenClaw has released OpenClaw 2.0, a major update to the open-source personal AI agent that changes its installation process, browser interface, memory, skills, automations, plugins, security, and collaboration features.&lt;/p&gt; &lt;i&gt;By Daniel Dominguez&lt;/i&gt;</description>
      <category>OpenAI</category>
      <category>Anthropic</category>
      <category>Artificial Intelligence</category>
      <category>ChatGPT</category>
      <category>Large language models</category>
      <category>Agents</category>
      <category>Claude</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>news</category>
      <pubDate>Tue, 01 Sep 2026 18:47:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/09/openclaw-2-release/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Daniel Dominguez</dc:creator>
      <dc:date>2026-09-01T18:47:00Z</dc:date>
      <dc:identifier>/news/2026/09/openclaw-2-release/en</dc:identifier>
    </item>
    <item>
      <title>Podcast: Scott Jenson on Evolving Desktop OS, Local-First, &amp; Agentic UX</title>
      <link>https://www.infoq.com/podcasts/evolving-desktop-agentic-ux/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/podcasts/evolving-desktop-agentic-ux/en/smallimage/infoq-podcast-500-1787734073812.jpg"/&gt;&lt;p&gt;In this episode, Scott Jenson, a veteran UX designer known for his work on the Macintosh, Google Maps, and Chrome, examines the long-term stagnation of desktop operating systems and the limitations of current mobile and cloud-centric models.&lt;/p&gt; &lt;i&gt;By Scott Jenson&lt;/i&gt;</description>
      <category>UX</category>
      <category>Local First</category>
      <category>Artificial Intelligence</category>
      <category>The InfoQ Podcast</category>
      <category>Design</category>
      <category>Large language models</category>
      <category>Operating Systems</category>
      <category>Desktop</category>
      <category>Privacy</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Culture &amp; Methods</category>
      <category>Development</category>
      <category>Architecture &amp; Design</category>
      <category>DevOps</category>
      <category>podcast</category>
      <pubDate>Mon, 31 Aug 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/podcasts/evolving-desktop-agentic-ux/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Scott Jenson</dc:creator>
      <dc:date>2026-08-31T11:00:00Z</dc:date>
      <dc:identifier>/podcasts/evolving-desktop-agentic-ux/en</dc:identifier>
    </item>
    <item>
      <title>Cloudflare Extends AI Search to Make it Easier for Agents and Developers to Search Custom Data</title>
      <link>https://www.infoq.com/news/2026/08/cloudflare-ai-search/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/cloudflare-ai-search/en/headerimage/cloudflare-ai-search-1788105575745.jpeg"/&gt;&lt;p&gt;Cloudflare AI Search is a built-in search and retrieval service designed to give AI agents and applications a ready-to-use search engine over custom data. It supports agent integration, multimodal search, and seamless integration with other Cloudflare tools.&lt;/p&gt; &lt;i&gt;By Sergio De Simone&lt;/i&gt;</description>
      <category>Search Engine</category>
      <category>Cloud</category>
      <category>Large language models</category>
      <category>Cloudflare</category>
      <category>Agents</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>news</category>
      <pubDate>Sun, 30 Aug 2026 19:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/cloudflare-ai-search/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Sergio De Simone</dc:creator>
      <dc:date>2026-08-30T19:00:00Z</dc:date>
      <dc:identifier>/news/2026/08/cloudflare-ai-search/en</dc:identifier>
    </item>
    <item>
      <title>Presentation: Architecting the Data Layer for AI Agents: from Transactional Systems to MCP and Semantic Models</title>
      <link>https://www.infoq.com/presentations/enterprise-data-architecture-ai-agents/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://res.infoq.com/presentations/enterprise-data-architecture-ai-agents/en/mediumimage/fabiane-nardon-medium-1787218382028.jpeg"/&gt;&lt;p&gt;Fabiane Nardon shares how TOTVS prepares enterprise data for token-hungry AI agents. She discusses balancing deterministic logic and non-deterministic LLMs across precision, security, and cost. Nardon details using data mesh, low-latency database architectures, semantic ontologies, and dynamic MCP tool selection to optimize context windows and reduce token overhead in transactional systems.&lt;/p&gt; &lt;i&gt;By Fabiane Nardon&lt;/i&gt;</description>
      <category>AI Cost Optimisation</category>
      <category>Transcripts</category>
      <category>AI Security</category>
      <category>AI Architecture</category>
      <category>Semantic Web</category>
      <category>QCon AI Boston 2026</category>
      <category>Agentic AI Architecture</category>
      <category>Large language models</category>
      <category>Model Context Protocol (MCP)</category>
      <category>Agents</category>
      <category>Data Mesh</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Sat, 29 Aug 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/enterprise-data-architecture-ai-agents/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Fabiane Nardon</dc:creator>
      <dc:date>2026-08-29T11:00:00Z</dc:date>
      <dc:identifier>/presentations/enterprise-data-architecture-ai-agents/en</dc:identifier>
    </item>
    <item>
      <title>FreeToken Unlocks Frontier MoE Inference on Consumer Hardware via Dynamic Co-Execution</title>
      <link>https://www.infoq.com/news/2026/08/freetoken-local-inference/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Researchers from UC Berkeley and MIT have developed FreeToken, an open-source inference engine that enhances the utility of Mixture-of-Experts models on consumer hardware. By implementing a dynamic scheduling policy and optimising weight management, FreeToken improves decoding speeds and execution efficiency in edge AI applications, fostering self-hosted reasoning systems.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>Model Inference</category>
      <category>Local First</category>
      <category>Large language models</category>
      <category>Local Inference</category>
      <category>Frontier Model</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>DevOps</category>
      <category>news</category>
      <pubDate>Sat, 29 Aug 2026 05:05:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/freetoken-local-inference/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Large+language+models</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-08-29T05:05:00Z</dc:date>
      <dc:identifier>/news/2026/08/freetoken-local-inference/en</dc:identifier>
    </item>
  </channel>
</rss>
