<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Data Engineering on App Coding</title>
    <link>https://appcoding.com/tags/data-engineering/</link>
    <description>Recent content in Data Engineering on App Coding</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Mon, 05 Oct 2026 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://appcoding.com/tags/data-engineering/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>DuckDB Already Queries Every File in a Folder, So Build the Part That Infers the Joins</title>
      <link>https://appcoding.com/duckdb-already-queries-every-file-in-a-folder-so-build-the-part-that-infers-the-joins/</link>
      <pubDate>Mon, 05 Oct 2026 00:00:00 +0000</pubDate>
      <guid>https://appcoding.com/duckdb-already-queries-every-file-in-a-folder-so-build-the-part-that-infers-the-joins/</guid>
      <description>&lt;p&gt;Someone hands you a zip: &lt;code&gt;customers.csv&lt;/code&gt; from the CRM, a JSON dump from the billing API, and a SQLite file from an internal app nobody has touched in two years. DuckDB will make all three queryable in a few lines of SQL, with no import step (&lt;code&gt;SELECT * FROM &#39;customers.csv&#39;&lt;/code&gt; works as written). Then you spend the afternoon working out which column joins to which. The CRM says &lt;code&gt;customer_id&lt;/code&gt; and stores &lt;code&gt;00123&lt;/code&gt;. The billing API says &lt;code&gt;customerId&lt;/code&gt; and stores &lt;code&gt;123&lt;/code&gt;. The SQLite table has a column called &lt;code&gt;customer&lt;/code&gt; holding the text &lt;code&gt;123&lt;/code&gt;. Same customer, three spellings.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
