<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Local Inference</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Local Inference feed</description>
    <item>
      <title>Presentation: Running AI at the Edge: Running Real Workloads Directly in the Browser</title>
      <link>https://www.infoq.com/presentations/local-ai-browser-inference-privacy/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</link>
      <description>&lt;img src="https://res.infoq.com/presentations/local-ai-browser-inference-privacy/en/mediumimage/james-hall-medium-1787813225372.jpeg"/&gt;&lt;p&gt;James Hall discusses the strategic and technical imperative of moving AI workloads from cloud providers to local edge devices. He shares practical approaches using WebGPU, Transformers.js, and DuckDB to achieve near-native performance in JavaScript. Through real-world case studies, he explains how to minimize data privacy risks, optimize browser inference, and build rigorous evaluation suites.&lt;/p&gt; &lt;i&gt;By James Hall&lt;/i&gt;</description>
      <category>Transcripts</category>
      <category>Edge Computing</category>
      <category>QCon London 2026</category>
      <category>AI Security</category>
      <category>Cloud Computing</category>
      <category>GPU</category>
      <category>Web Browser</category>
      <category>Machine Learning</category>
      <category>Local Inference</category>
      <category>Web Development</category>
      <category>Privacy</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Mon, 31 Aug 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/local-ai-browser-inference-privacy/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</guid>
      <dc:creator>James Hall</dc:creator>
      <dc:date>2026-08-31T11:00:00Z</dc:date>
      <dc:identifier>/presentations/local-ai-browser-inference-privacy/en</dc:identifier>
    </item>
    <item>
      <title>FreeToken Unlocks Frontier MoE Inference on Consumer Hardware via Dynamic Co-Execution</title>
      <link>https://www.infoq.com/news/2026/08/freetoken-local-inference/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Researchers from UC Berkeley and MIT have developed FreeToken, an open-source inference engine that enhances the utility of Mixture-of-Experts models on consumer hardware. By implementing a dynamic scheduling policy and optimising weight management, FreeToken improves decoding speeds and execution efficiency in edge AI applications, fostering self-hosted reasoning systems.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>Model Inference</category>
      <category>Local First</category>
      <category>Large language models</category>
      <category>Local Inference</category>
      <category>Frontier Model</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>DevOps</category>
      <category>news</category>
      <pubDate>Sat, 29 Aug 2026 05:05:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/freetoken-local-inference/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-08-29T05:05:00Z</dc:date>
      <dc:identifier>/news/2026/08/freetoken-local-inference/en</dc:identifier>
    </item>
  </channel>
</rss>
