<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Data Lake</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Data Lake feed</description>
    <item>
      <title>Harper Argues Against the Multi-System Stack and Releases 5.2</title>
      <link>https://www.infoq.com/news/2026/08/harper-vercel-benchmark/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</link>
      <description>&lt;img src="https://res.infoq.com/news/2026/08/harper-vercel-benchmark/en/headerimage/generatedHeaderImage-1786778815218.jpg"/&gt;&lt;p&gt;The database platform Harper advocates for a single-runtime architecture that keeps application code and data together, with its benchmark against a Vercel-based stack reporting significantly better performance on live, personalized-data workloads. Harper recently released version 5.2, with a new record cache and more throughput per node.&lt;/p&gt; &lt;i&gt;By Renato Losio&lt;/i&gt;</description>
      <category>Big Data</category>
      <category>Database</category>
      <category>Data Lake</category>
      <category>Serverless</category>
      <category>Distributed Systems</category>
      <category>Distributed Data</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>news</category>
      <pubDate>Thu, 20 Aug 2026 06:20:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/harper-vercel-benchmark/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</guid>
      <dc:creator>Renato Losio</dc:creator>
      <dc:date>2026-08-20T06:20:00Z</dc:date>
      <dc:identifier>/news/2026/08/harper-vercel-benchmark/en</dc:identifier>
    </item>
    <item>
      <title>Spotify Builds External Index to Enable Low Latency Point Queries on its Data Lake</title>
      <link>https://www.infoq.com/news/2026/08/spotify-data-lake-point-queries/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</link>
      <description>&lt;img src="https://www.infoq.com/styles/static/images/logo/logo_bigger.jpg"/&gt;&lt;p&gt;Spotify introduced  external indexing architecture for Apache Parquet data lakes that enables low-latency point queries without replicating datasets into operational databases. The approach maps lookup keys to Parquet files and row locations, allowing targeted reads from cloud object storage while supporting analytics, machine learning, AI applications, and online services from the same datasets.&lt;/p&gt; &lt;i&gt;By Leela Kumili&lt;/i&gt;</description>
      <category>Google Cloud</category>
      <category>AI Architecture</category>
      <category>Big Data</category>
      <category>IndexedDB</category>
      <category>Apache Iceberg</category>
      <category>Data Lake</category>
      <category>Software Engineering</category>
      <category>Data</category>
      <category>Distributed Systems</category>
      <category>Apache</category>
      <category>Query</category>
      <category>Low Latency</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Development</category>
      <category>Architecture &amp; Design</category>
      <category>news</category>
      <pubDate>Wed, 12 Aug 2026 14:26:00 GMT</pubDate>
      <guid>https://www.infoq.com/news/2026/08/spotify-data-lake-point-queries/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</guid>
      <dc:creator>Leela Kumili</dc:creator>
      <dc:date>2026-08-12T14:26:00Z</dc:date>
      <dc:identifier>/news/2026/08/spotify-data-lake-point-queries/en</dc:identifier>
    </item>
  </channel>
</rss>
