<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Local Inference</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Local Inference feed</description>
    <item>
      <title>Presentation: Running AI at the Edge: Running Real Workloads Directly in the Browser</title>
      <link>https://www.infoq.com/presentations/local-ai-browser-inference-privacy/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</link>
      <description>&lt;img src="https://res.infoq.com/presentations/local-ai-browser-inference-privacy/en/mediumimage/james-hall-medium-1787813225372.jpeg"/&gt;&lt;p&gt;James Hall discusses the strategic and technical imperative of moving AI workloads from cloud providers to local edge devices. He shares practical approaches using WebGPU, Transformers.js, and DuckDB to achieve near-native performance in JavaScript. Through real-world case studies, he explains how to minimize data privacy risks, optimize browser inference, and build rigorous evaluation suites.&lt;/p&gt; &lt;i&gt;By James Hall&lt;/i&gt;</description>
      <category>AI Security</category>
      <category>Cloud Computing</category>
      <category>GPU</category>
      <category>QCon London 2026</category>
      <category>Privacy</category>
      <category>Web Browser</category>
      <category>Edge Computing</category>
      <category>Web Development</category>
      <category>Machine Learning</category>
      <category>Local Inference</category>
      <category>Transcripts</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Mon, 31 Aug 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/local-ai-browser-inference-privacy/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</guid>
      <dc:creator>James Hall</dc:creator>
      <dc:date>2026-08-31T11:00:00Z</dc:date>
      <dc:identifier>/presentations/local-ai-browser-inference-privacy/en</dc:identifier>
    </item>
    <item>
      <title>FreeToken Unlocks Frontier MoE Inference on Consumer Hardware via Dynamic Co-Execution</title>
      <link>https://www.infoq.com/news/2026/08/freetoken-local-inference/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Researchers from UC Berkeley and MIT have developed FreeToken, an open-source inference engine that enhances the utility of Mixture-of-Experts models on consumer hardware. By implementing a dynamic scheduling policy and optimising weight management, FreeToken improves decoding speeds and execution efficiency in edge AI applications, fostering self-hosted reasoning systems.&lt;/p&gt; &lt;i&gt;By Olimpiu Pop&lt;/i&gt;</description>
      <category>Local First</category>
      <category>Large language models</category>
      <category>Model Inference</category>
      <category>Frontier Model</category>
      <category>Local Inference</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>DevOps</category>
      <category>news</category>
      <pubDate>Sat, 29 Aug 2026 05:05:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/freetoken-local-inference/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Local+Inference</guid>
      <dc:creator>Olimpiu Pop</dc:creator>
      <dc:date>2026-08-29T05:05:00Z</dc:date>
      <dc:identifier>/news/2026/08/freetoken-local-inference/en</dc:identifier>
    </item>
  </channel>
</rss>
