<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>InfoQ - Data Lake</title>
    <link>https://www.infoq.com</link>
    <description>InfoQ Data Lake feed</description>
    <item>
      <title>Presentation: From S3 to GPU in One Copy: Rethinking Data Loading for ML Training</title>
      <link>https://www.infoq.com/presentations/vortex-columnar-file-format-gpu-streaming/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</link>
      <description>&lt;img src="https://res.infoq.com/presentations/vortex-columnar-file-format-gpu-streaming/en/mediumimage/onur-satici-medium-1787813397611.jpeg"/&gt;&lt;p&gt;Onur Satici explains how Vortex, an open-source columnar file format under the Linux Foundation, revolutionizes high-throughput data loading. He details how cascading lightweight encodings, layout-based segment pruning, and zero-copy memory pipelines eliminate CPU/NVMe bottlenecks to stream S3 data straight to GPUs at speeds up to 60 Gbps without requiring upfront data reprocessing.&lt;/p&gt; &lt;i&gt;By Onur Satici&lt;/i&gt;</description>
      <category>Architecture</category>
      <category>Columnar Databases</category>
      <category>Data Lake</category>
      <category>GPU</category>
      <category>Machine Learning</category>
      <category>S3</category>
      <category>Streaming</category>
      <category>Transcripts</category>
      <category>QCon London 2026</category>
      <category>Data Pipelines</category>
      <category>CUDA</category>
      <category>Rust</category>
      <category>Performance</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>Architecture &amp; Design</category>
      <category>Development</category>
      <category>presentation</category>
      <pubDate>Fri, 04 Sep 2026 11:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/presentations/vortex-columnar-file-format-gpu-streaming/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</guid>
      <dc:creator>Onur Satici</dc:creator>
      <dc:date>2026-09-04T11:00:00Z</dc:date>
      <dc:identifier>/presentations/vortex-columnar-file-format-gpu-streaming/en</dc:identifier>
    </item>
    <item>
      <title>Article: Beyond Offset Lag: Computing Time in Queue for Apache Hudi Data Lake Pipelines at Petabyte Scale</title>
      <link>https://www.infoq.com/articles/beyond-offset-lag-kafka-apache-hudi/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</link>
      <description>&lt;img src="https://res.infoq.com/articles/beyond-offset-lag-kafka-apache-hudi/en/headerimage/beyond-offset-lag-kafka-apache-hudi-header-1787577734265.jpg"/&gt;&lt;p&gt;In this article, author Srikanth Mamidala discusses the data lake architecture used for analytics, reporting, and machine learning and shows how to manage the consumer lag metrics when using Kafka and Apache Hudi.&lt;/p&gt; &lt;i&gt;By Srikanth Mamidala&lt;/i&gt;</description>
      <category>Streaming</category>
      <category>Data Pipelines</category>
      <category>Data Lake</category>
      <category>Messaging</category>
      <category>Apache Kafka</category>
      <category>AI, ML &amp; Data Engineering</category>
      <category>article</category>
      <pubDate>Wed, 26 Aug 2026 09:00:00 GMT</pubDate>
      <guid>https://www.infoq.com/articles/beyond-offset-lag-kafka-apache-hudi/?utm_campaign=infoq_content&amp;utm_source=infoq&amp;utm_medium=feed&amp;utm_term=Data+Lake</guid>
      <dc:creator>Srikanth Mamidala</dc:creator>
      <dc:date>2026-08-26T09:00:00Z</dc:date>
      <dc:identifier>/articles/beyond-offset-lag-kafka-apache-hudi/en</dc:identifier>
    </item>
  </channel>
</rss>
