<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>Blog</title>
    <link>https://www.starrocks.io/blog</link>
    <description>Discover what's new with the world's fastest, freshest, and most flexible open-source data warehouse and lakehouse query engine.</description>
    <language>en</language>
    <pubDate>Thu, 30 Apr 2026 02:08:13 GMT</pubDate>
    <dc:date>2026-04-30T02:08:13Z</dc:date>
    <dc:language>en</dc:language>
    <item>
      <title>StarRocks 4.1: Built for Production, Designed to Simplify</title>
      <link>https://www.starrocks.io/blog/starrocks-4.1-now-available-built-for-production-designed-to-simplify</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-4.1-now-available-built-for-production-designed-to-simplify" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813896.png" alt="StarRocks 4.1: Built for Production, Designed to Simplify" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-4.1-now-available-built-for-production-designed-to-simplify" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813896.png" alt="StarRocks 4.1: Built for Production, Designed to Simplify" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fstarrocks-4.1-now-available-built-for-production-designed-to-simplify&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <category>Release</category>
      <pubDate>Wed, 29 Apr 2026 07:17:54 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/starrocks-4.1-now-available-built-for-production-designed-to-simplify</guid>
      <dc:date>2026-04-29T07:17:54Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>StarRocks Monitor &amp; Alert Guide_Part 3: Application Availability</title>
      <link>https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-3-application-availability</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-3-application-availability" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813895.png" alt="StarRocks Monitor &amp;amp; Alert Guide_Part 3: Application Availability" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="line-height: 1.2;"&gt;&lt;span style="white-space-collapse: preserve;"&gt;&lt;span style="width: 624px; height: 352px;"&gt;&lt;br&gt;&lt;br&gt;&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-3-application-availability" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813895.png" alt="StarRocks Monitor &amp;amp; Alert Guide_Part 3: Application Availability" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="line-height: 1.2;"&gt;&lt;span style="white-space-collapse: preserve;"&gt;&lt;span style="width: 624px; height: 352px;"&gt;&lt;br&gt;&lt;br&gt;&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fstarrocks-monitor-alert-guide_part-3-application-availability&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Tue, 21 Apr 2026 22:00:00 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-3-application-availability</guid>
      <dc:date>2026-04-21T22:00:00Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>Designing an Analytics Engine for AI Agent Workloads</title>
      <link>https://www.starrocks.io/blog/designing-an-analytics-engine-for-ai-agent-workloads</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/designing-an-analytics-engine-for-ai-agent-workloads" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813894.png" alt="Designing an Analytics Engine for AI Agent Workloads" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About the Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;em&gt;&lt;a href="https://kangkaisen.com/"&gt;Kaisen Kang&lt;/a&gt;&lt;/em&gt; 
          &lt;em&gt;, StarRocks TSC Member, &lt;/em&gt; 
          &lt;em&gt;Query&lt;/em&gt; 
          &lt;em&gt; Engine &amp;amp; AI Agent Team Lead at CelerData&lt;/em&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/designing-an-analytics-engine-for-ai-agent-workloads" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813894.png" alt="Designing an Analytics Engine for AI Agent Workloads" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About the Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;em&gt;&lt;a href="https://kangkaisen.com/"&gt;Kaisen Kang&lt;/a&gt;&lt;/em&gt; 
          &lt;em&gt;, StarRocks TSC Member, &lt;/em&gt; 
          &lt;em&gt;Query&lt;/em&gt; 
          &lt;em&gt; Engine &amp;amp; AI Agent Team Lead at CelerData&lt;/em&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fdesigning-an-analytics-engine-for-ai-agent-workloads&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Tue, 07 Apr 2026 08:11:29 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/designing-an-analytics-engine-for-ai-agent-workloads</guid>
      <dc:date>2026-04-07T08:11:29Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>How to Manage Schema Migrations in StarRocks with SQLAlchemy and Alembic</title>
      <link>https://www.starrocks.io/blog/how-to-manage-schema-migrations-in-starrocks-with-sqlalchemy-and-alembic</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/how-to-manage-schema-migrations-in-starrocks-with-sqlalchemy-and-alembic" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813893.png" alt="How to Manage Schema Migrations in StarRocks with SQLAlchemy and Alembic" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="line-height: 1.2;"&gt;&lt;span style="color: #2d2d2d; white-space-collapse: preserve;"&gt;&lt;span style="width: 624px; height: 348px;"&gt;&lt;br&gt;&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/how-to-manage-schema-migrations-in-starrocks-with-sqlalchemy-and-alembic" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813893.png" alt="How to Manage Schema Migrations in StarRocks with SQLAlchemy and Alembic" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="line-height: 1.2;"&gt;&lt;span style="color: #2d2d2d; white-space-collapse: preserve;"&gt;&lt;span style="width: 624px; height: 348px;"&gt;&lt;br&gt;&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fhow-to-manage-schema-migrations-in-starrocks-with-sqlalchemy-and-alembic&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Tue, 31 Mar 2026 06:33:09 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/how-to-manage-schema-migrations-in-starrocks-with-sqlalchemy-and-alembic</guid>
      <dc:date>2026-03-31T06:33:09Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>StarRocks Monitor &amp; Alert Guide_Part 2: Cluster Service Health</title>
      <link>https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-2-cluster-service-health</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-2-cluster-service-health" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813892.png" alt="StarRocks Monitor &amp;amp; Alert Guide_Part 2: Cluster Service Health" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="line-height: 1.2;"&gt;&lt;span style="white-space-collapse: preserve;"&gt;&lt;span style="width: 624px; height: 352px;"&gt;&lt;br&gt;&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-2-cluster-service-health" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813892.png" alt="StarRocks Monitor &amp;amp; Alert Guide_Part 2: Cluster Service Health" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="line-height: 1.2;"&gt;&lt;span style="white-space-collapse: preserve;"&gt;&lt;span style="width: 624px; height: 352px;"&gt;&lt;br&gt;&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fstarrocks-monitor-alert-guide_part-2-cluster-service-health&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Tue, 24 Mar 2026 07:09:50 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-2-cluster-service-health</guid>
      <dc:date>2026-03-24T07:09:50Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>StarRocks Monitor &amp; Alert Guide_Part 1: Resource Saturation</title>
      <link>https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-1-resource-saturation</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-1-resource-saturation" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%20132.png" alt="StarRocks Monitor &amp;amp; Alert Guide_Part 1: Resource Saturation" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;div&gt; 
      &lt;p&gt;&lt;span style="background-color: transparent;"&gt;I&lt;/span&gt;&lt;span style="background-color: transparent;"&gt;n&lt;/span&gt;&lt;span style="background-color: transparent;"&gt; d&lt;/span&gt;&lt;span style="background-color: transparent;"&gt;ay-to-day operations, unexpected issues pop up all the time. One moment everything is running smoothly, and the next you're scrambling to figure out why queries are timing out or nodes are suddenly under heavy load. Many teams end up stuck in a constant cycle of reacting to incidents and “putting out fires.”&lt;/span&gt;&lt;/p&gt; 
      &lt;p&gt;&amp;nbsp;&lt;/p&gt; 
     &lt;/div&gt; 
     &lt;div&gt; 
      &lt;div&gt; 
       &lt;span style="background-color: transparent;"&gt;&lt;/span&gt; 
       &lt;span style="background-color: transparent;"&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-1-resource-saturation" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%20132.png" alt="StarRocks Monitor &amp;amp; Alert Guide_Part 1: Resource Saturation" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;div&gt; 
      &lt;p&gt;&lt;span style="background-color: transparent;"&gt;I&lt;/span&gt;&lt;span style="background-color: transparent;"&gt;n&lt;/span&gt;&lt;span style="background-color: transparent;"&gt; d&lt;/span&gt;&lt;span style="background-color: transparent;"&gt;ay-to-day operations, unexpected issues pop up all the time. One moment everything is running smoothly, and the next you're scrambling to figure out why queries are timing out or nodes are suddenly under heavy load. Many teams end up stuck in a constant cycle of reacting to incidents and “putting out fires.”&lt;/span&gt;&lt;/p&gt; 
      &lt;p&gt;&amp;nbsp;&lt;/p&gt; 
     &lt;/div&gt; 
     &lt;div&gt; 
      &lt;div&gt; 
       &lt;span style="background-color: transparent;"&gt;&lt;/span&gt; 
       &lt;span style="background-color: transparent;"&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fstarrocks-monitor-alert-guide_part-1-resource-saturation&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Mon, 16 Mar 2026 15:30:00 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/starrocks-monitor-alert-guide_part-1-resource-saturation</guid>
      <dc:date>2026-03-16T15:30:00Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>Deep Dive: How StarRocks Built a High-Performance Vectorized Engine</title>
      <link>https://www.starrocks.io/blog/deep-dive-how-starrocks-built-a-high-performance-vectorized-engine</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/deep-dive-how-starrocks-built-a-high-performance-vectorized-engine" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%20135.png" alt="Deep Dive: How StarRocks Built a High-Performance Vectorized Engine" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About the Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;em&gt;&lt;a href="https://kangkaisen.com/"&gt;Kaisen Kang&lt;/a&gt;&lt;/em&gt; 
          &lt;em&gt;, StarRocks TSC Member, &lt;/em&gt; 
          &lt;em&gt;Query&lt;/em&gt; 
          &lt;em&gt; Engine &amp;amp; AI Agent Team Lead at CelerData&lt;/em&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/deep-dive-how-starrocks-built-a-high-performance-vectorized-engine" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%20135.png" alt="Deep Dive: How StarRocks Built a High-Performance Vectorized Engine" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About the Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;em&gt;&lt;a href="https://kangkaisen.com/"&gt;Kaisen Kang&lt;/a&gt;&lt;/em&gt; 
          &lt;em&gt;, StarRocks TSC Member, &lt;/em&gt; 
          &lt;em&gt;Query&lt;/em&gt; 
          &lt;em&gt; Engine &amp;amp; AI Agent Team Lead at CelerData&lt;/em&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fdeep-dive-how-starrocks-built-a-high-performance-vectorized-engine&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Tue, 24 Feb 2026 19:00:00 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/deep-dive-how-starrocks-built-a-high-performance-vectorized-engine</guid>
      <dc:date>2026-02-24T19:00:00Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>DataOps-Driven Governance and Analytics with dbt and StarRocks</title>
      <link>https://www.starrocks.io/blog/dataops-driven-governance-and-analytics-with-dbt-and-starrocks</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/dataops-driven-governance-and-analytics-with-dbt-and-starrocks" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813891%20(1).png" alt="DataOps-Driven Governance and Analytics with dbt and StarRocks" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About The Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;div&gt;
            Author: 
           &lt;em&gt;Jacky Wu&lt;/em&gt; is a 
           &lt;a href="https://github.com/StarRocks/dbt-starrocks"&gt;dbt-starrocks&lt;/a&gt; 
           &lt;strong&gt; contributor&lt;/strong&gt; and 
           &lt;strong&gt;Senior Enterprise Solution Manager at SJM Resorts&lt;/strong&gt;. He specializes in enterprise data architecture, DataOps practices, real-time analytics, and works closely with engineering and business teams to deliver scalable, governance-ready data platforms. 
          &lt;/div&gt; 
          &lt;span&gt;&lt;/span&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;
        As enterprises increasingly rely on data to drive real-time decision-making, traditional data architectures and development workflows are struggling to keep pace with growing demands for speed, reliability, and governance. This article explores how a modern, integrated data architecture, built on 
       &lt;strong&gt;dbt, StarRocks, and DataOps practices, &lt;/strong&gt;addresses these challenges by unifying data modeling, automation, and analytics into a single, cohesive framework. 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;
        Through a combination of engineering best practices and platform innovation, this “three-in-one” approach enables faster iteration, stronger governance, and more reliable analytics across both real-time and batch scenarios. The discussion unfolds across four key dimensions: the role of dbt in data modeling and governance automation, the impact of DataOps on agility and control, StarRocks' technical breakthroughs for hybrid analytics, and real-world case studies that demonstrate these capabilities in practice. 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
       &lt;h2&gt;The Core Role of dbt in Data Modeling and Governance Automation&lt;/h2&gt; 
       &lt;h3&gt;Key Capabilities of dbt&lt;/h3&gt; At its core, dbt is a framework for building and governing analytics data through code. Raw data processed with dbt is transformed according to the principle of 
       &lt;strong&gt;“data models as code,”&lt;/strong&gt; allowing teams to define transformations, logic, and dependencies in a structured, version-controlled way. Beyond generating curated data models, dbt also produces essential governance artifacts, including data dictionaries, lineage graphs, and automated data quality tests. 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;
        With these foundations in place, dbt serves as the backbone for developing a wide range of data products—from analytical dashboards to data-driven applications that directly support business workflows. 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;&lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;At the methodological level, dbt introduces a core concept that closely aligns with DevOps. Most engineering teams are already familiar with DevOps, which focuses on managing and collaborating on code through standardized, engineering-driven practices. dbt extends this philosophy into the data domain, applying the same engineering rigor to data development and governance.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Data Models as Code&lt;/h3&gt; 
&lt;p&gt;In practice, teams typically work across multiple feature branches, with automated tests triggered during the merge process. Once changes are validated in a staging environment, CI/CD pipelines are used to promote them into production. This workflow allows data models to be versioned, reviewed, and managed with the same discipline and rigor as application code.To make this model effective in real-world analytics systems, it needs to be paired with an execution engine that can support both high-performance queries and frequent model changes. This is where the analytical database layer becomes critical.&lt;/p&gt; 
&lt;h4&gt;&amp;nbsp;&lt;/h4&gt; 
&lt;h4&gt;StarRocks as the Analytical Execution Layer&lt;/h4&gt; 
&lt;p&gt;StarRocks is a high-performance analytical database designed for modern, real-time analytics workloads. It follows a lakehouse-oriented architecture, enabling unified analytics across both real-time and batch data without maintaining separate systems for streaming and offline processing.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;StarRocks supports high-concurrency, low-latency queries directly on fresh data, making it well suited for dashboards, operational analytics, and data-driven applications. At the same time, it integrates naturally with data lakes and ELT pipelines, allowing teams to build scalable analytics platforms without sacrificing performance or consistency.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In this architecture, StarRocks serves as the analytical execution layer where raw data is continuously ingested and stored, while dbt operates on top of it to define transformations, models, and governance logic. Together, they form a clean separation of responsibilities: StarRocks focuses on efficient data storage and query execution, while dbt manages modeling, version control, and testing.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Within this setup, dbt integrates seamlessly with ELT workflows, improving the efficiency, reliability, and overall controllability of data modeling and governance. In day-to-day development, when issues are identified in a specific data model, teams can quickly roll back changes at the branch level. With Git as the system of record, every change goes through code review and is deployed via automated CI/CD pipelines.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;As a result, SQL models, materialized views, and other data objects are promoted to production through standard pull request (PR) workflows, ensuring consistency, traceability, and operational safety across the entire lifecycle.&lt;/p&gt; 
&lt;p&gt;dbt also integrates seamlessly with the native StarRocks ecosystem, enabling version control across multiple types of data objects, including tables, views, materialized views (MVs), and tasks. In dbt, a &lt;em&gt;model&lt;/em&gt; is essentially a SQL template. A typical pattern is to first create a staging model for customer data, then reuse it as a dependency for downstream custom business models. dbt automatically resolves dependencies between models and manages execution order, eliminating the need for external scheduling tools—running dbt alone is sufficient to orchestrate the entire pipeline.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Automated Data Dictionary Generation&lt;/h3&gt; 
&lt;p&gt;From a documentation and asset management perspective, dbt can automatically generate data dictionaries and other documentation artifacts. Through dbt-generated HTML documentation, teams can easily explore field definitions, business meanings, underlying SQL logic, and upstream/downstream dependencies. The documentation interface can also be customized with enterprise branding, including logos and visual styles.&lt;/p&gt; 
&lt;h3&gt;Automated Data Lineage&lt;/h3&gt; 
&lt;div style="text-align: center;"&gt;
  &amp;nbsp; 
&lt;/div&gt; 
&lt;p&gt;Data lineage is a foundational component of any data governance framework. Large enterprises often manage thousands of tables and a vast portfolio of data products. In industries such as hospitality, where organizations operate across hotels, restaurants, and other business lines, it is common to build a unified &lt;strong&gt;Customer 360&lt;/strong&gt; view to consolidate data assets across domains.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In these environments, even small changes to upstream raw data can have far-reaching effects. A core challenge is quickly identifying which downstream models, reports, or data products may be impacted. Data lineage addresses this need by enabling precise impact analysis, giving teams clear visibility into dependencies and helping them assess the scope and risk of upstream changes or data quality issues.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Automated Data Quality Testing&lt;/h3&gt; 
&lt;p&gt;Beyond data lineage, automated data testing is a core pillar of dbt best practices. Teams can define a wide range of automated tests for their data models, such as scheduled daily validations to ensure that datasets continue to meet expected conditions. When anomalies are detected, alerts can be triggered immediately, enabling faster detection and resolution of data issues.&lt;/p&gt; 
&lt;p&gt;From an implementation standpoint, dbt model configurations are typically defined in YAML files. Each model includes metadata such as its name and description, which capture the model's purpose and business context, followed by field-level definitions. dbt provides a comprehensive set of built-in tests, such as &lt;code&gt;unique&lt;/code&gt; and &lt;code&gt;not_null&lt;/code&gt; to enforce common data quality constraints, including uniqueness and null checks.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In most OLAP databases, foreign key constraints are not enforced at the database level. To address this gap, dbt offers &lt;code&gt;ref&lt;/code&gt;-based and relationship testing capabilities, allowing teams to validate that models are correctly referenced by downstream tables or models. These tests are also centrally managed through YAML configuration files.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In practice, some teams further enhance this workflow by using AI-assisted tools to automatically generate and batch-maintain YAML configurations, significantly reducing manual effort while improving consistency.&lt;/p&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;How DataOps Workflows Improve Agility and Control in Data Projects&lt;/h2&gt; 
&lt;h3&gt;Key Components of DataOps&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Building on the principles of DevOps, DataOps extends the same engineering mindset into the data domain. In application development, DevOps has become the standard approach for managing code development, integration, and deployment. DataOps applies these proven practices to the full data lifecycle, from development, testing, deployment, and ongoing operations to improve both speed and reliability.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In a typical DataOps architecture, the areas covered by dbt are often clearly highlighted. These include data modeling, automated data quality testing, data lineage analysis, and documentation-driven validation—all of which are native capabilities of dbt. For stages beyond dbt's scope, teams typically integrate other well-established tools to complete the workflow. Project management platforms such as Jira is used to track model changes and defects; scheduling tools manage periodic model execution and data checks (hourly or daily); CI/CD tools like Jenkins handle automated integration and deployment; and downstream analytics and BI tools support data consumption, visualization, and decision-making.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Version Control Standards: Conventional Commits&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;At the version control layer, experienced teams typically adopt a standardized commit convention. One widely used approach is &lt;strong&gt;Conventional Commits&lt;/strong&gt;, which defines a structured format for commit messages. Its core idea is to clearly distinguish between different types of changes, such as feature enhancements and bug fixes, so that versioning and release management can be automated.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;For example, when a change introduces new functionality, such as adding a new dimension to an &lt;code&gt;order&lt;/code&gt; model, the &lt;strong&gt;minor version&lt;/strong&gt; is incremented (e.g., from 2.0 to 2.1). These changes are generally backward compatible. In contrast, bug fixes only trigger a &lt;strong&gt;patch version&lt;/strong&gt; increment (e.g., from 2.1.0 to 2.1.1).&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Conventional Commits also make it possible to generate release notes automatically. In the past, release managers often had to manually confirm changes with developers and compile release notes by hand—a process that was both time-consuming and error-prone. With Conventional Commits in place, this workflow can be fully automated.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;When commit messages follow the defined convention (for example, using the &lt;code&gt;fix&lt;/code&gt; prefix), they can be automatically parsed and aggregated into structured release notes. Developers focus on writing clear, standardized commit messages, while the tooling handles release note generation. As part of the CI/CD pipeline, release notes are created and updated automatically with each release, requiring no additional manual effort from the team.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;A CI/CD Automation Example in DataOps&lt;/h3&gt; 
&lt;p&gt;In a DataOps framework, the CI/CD workflow typically starts with a &lt;strong&gt;Pull Request (PR)&lt;/strong&gt;. Every change is introduced through a PR, which triggers a standardized validation and deployment pipeline.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;The pipeline begins with automated code quality checks, commonly referred to as &lt;em&gt;linting&lt;/em&gt;. Similar to style and syntax validation in application development, linting tools automatically analyze SQL models and related configurations to ensure they conform to predefined standards. Once these checks pass, the changes are deployed to a staging environment.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Using dbt as an example, individual models can be deployed to specific environments. This makes it possible to deploy a single model to staging and run targeted unit tests and data quality checks. After the automated tests pass, the workflow moves to a manual review stage, where reviewers examine the PR to validate the logic and assess its potential impact.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Once the changes are approved and merged into the main branch, the system automatically packages a new version, updates the change history, and promotes the release to QA or production environments. This end-to-end process ensures that data changes are delivered with both speed and control—key characteristics of a mature DataOps practice.&lt;/p&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;Technical Breakthroughs with StarRocks for Real-Time and Batch Analytics&lt;/h2&gt; 
&lt;h3&gt;Traditional Siloed ETL Architectures in Lake–Warehouse Separation&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In early-stage implementations, many teams relied on siloed ETL architectures. Using a hospitality industry scenario as an example, the initial setup typically consisted of multiple independent operational databases at the source layer. ETL jobs ran on fixed schedules (for example, every 15 minutes) to extract data from these systems and load it into multiple data warehouses. In parallel, separate data pipelines were built to support mobile applications, reporting systems, and other analytical workloads.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;This architecture came with several fundamental limitations. Data models lacked version control, making the system fragile and easy to break as changes accumulated. Testing was largely manual, which made it difficult to establish a consistent and repeatable quality assurance process. Documentation was also highly fragmented, often maintained as standalone Word files across different teams, resulting in poor consistency, limited traceability, and high maintenance overhead.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;The StarRocks ELT Framework&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;With the introduction of StarRocks as an integrated platform, the overall architecture was redesigned into a unified lakehouse model that supports both real-time and batch processing. Using real-time CDC, data from multiple operational systems is continuously ingested into the data lake. On top of the lakehouse, an ELT-based framework enables rapid construction of application-facing data products.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;At the same time, data governance capabilities are implemented end-to-end across the pipeline. Version control is established around data models, data dictionaries are centrally maintained, and data lineage views are generated through tooling, providing full visibility into dependencies and impact.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Results from StarRocks + dbt + DataOps in Practice&lt;/h3&gt; 
&lt;p&gt;In the re-architected system, near-real-time data simultaneously supports mobile applications, reporting dashboards, and behavioral analytics workloads. On top of this foundation, a unified “three-in-one” DataOps framework delivers several immediate and measurable benefits:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;strong&gt;Faster iteration and recovery.&lt;/strong&gt; Data models built with dbt can be updated and rolled back quickly, significantly improving development velocity and incident recovery time.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;More efficient delivery through standardized pipelines.&lt;/strong&gt; DataOps manages business requirements and data product delivery through well-defined pipelines, introducing Agile-style practices that standardize project management workflows and automated release cycles. This substantially shortens the end-to-end lead time from requirement definition to production.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Strong consistency between models and documentation.&lt;/strong&gt; Model definitions and their corresponding YAML files are version-controlled together in Git, creating a single source of truth. Any model change requires the related documentation to be updated; otherwise, the release pipeline will fail validation.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Higher analytical accuracy and reliability.&lt;/strong&gt; For enterprises operating dozens of hotel systems, data lineage analysis makes it possible to understand where a system sits within the overall data pipeline and to assess downstream impact before upgrades. In parallel, automated data testing performs continuous health checks on models, validating data correctness and alignment with expectations on a daily basis。&lt;br&gt;&lt;br&gt;&lt;br&gt;&lt;/li&gt; 
&lt;/ul&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/dataops-driven-governance-and-analytics-with-dbt-and-starrocks" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813891%20(1).png" alt="DataOps-Driven Governance and Analytics with dbt and StarRocks" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About The Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;div&gt;
            Author: 
           &lt;em&gt;Jacky Wu&lt;/em&gt; is a 
           &lt;a href="https://github.com/StarRocks/dbt-starrocks"&gt;dbt-starrocks&lt;/a&gt; 
           &lt;strong&gt; contributor&lt;/strong&gt; and 
           &lt;strong&gt;Senior Enterprise Solution Manager at SJM Resorts&lt;/strong&gt;. He specializes in enterprise data architecture, DataOps practices, real-time analytics, and works closely with engineering and business teams to deliver scalable, governance-ready data platforms. 
          &lt;/div&gt; 
          &lt;span&gt;&lt;/span&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;
        As enterprises increasingly rely on data to drive real-time decision-making, traditional data architectures and development workflows are struggling to keep pace with growing demands for speed, reliability, and governance. This article explores how a modern, integrated data architecture, built on 
       &lt;strong&gt;dbt, StarRocks, and DataOps practices, &lt;/strong&gt;addresses these challenges by unifying data modeling, automation, and analytics into a single, cohesive framework. 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;
        Through a combination of engineering best practices and platform innovation, this “three-in-one” approach enables faster iteration, stronger governance, and more reliable analytics across both real-time and batch scenarios. The discussion unfolds across four key dimensions: the role of dbt in data modeling and governance automation, the impact of DataOps on agility and control, StarRocks' technical breakthroughs for hybrid analytics, and real-world case studies that demonstrate these capabilities in practice. 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
       &lt;h2&gt;The Core Role of dbt in Data Modeling and Governance Automation&lt;/h2&gt; 
       &lt;h3&gt;Key Capabilities of dbt&lt;/h3&gt; At its core, dbt is a framework for building and governing analytics data through code. Raw data processed with dbt is transformed according to the principle of 
       &lt;strong&gt;“data models as code,”&lt;/strong&gt; allowing teams to define transformations, logic, and dependencies in a structured, version-controlled way. Beyond generating curated data models, dbt also produces essential governance artifacts, including data dictionaries, lineage graphs, and automated data quality tests. 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;
        With these foundations in place, dbt serves as the backbone for developing a wide range of data products—from analytical dashboards to data-driven applications that directly support business workflows. 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;&lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;At the methodological level, dbt introduces a core concept that closely aligns with DevOps. Most engineering teams are already familiar with DevOps, which focuses on managing and collaborating on code through standardized, engineering-driven practices. dbt extends this philosophy into the data domain, applying the same engineering rigor to data development and governance.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Data Models as Code&lt;/h3&gt; 
&lt;p&gt;In practice, teams typically work across multiple feature branches, with automated tests triggered during the merge process. Once changes are validated in a staging environment, CI/CD pipelines are used to promote them into production. This workflow allows data models to be versioned, reviewed, and managed with the same discipline and rigor as application code.To make this model effective in real-world analytics systems, it needs to be paired with an execution engine that can support both high-performance queries and frequent model changes. This is where the analytical database layer becomes critical.&lt;/p&gt; 
&lt;h4&gt;&amp;nbsp;&lt;/h4&gt; 
&lt;h4&gt;StarRocks as the Analytical Execution Layer&lt;/h4&gt; 
&lt;p&gt;StarRocks is a high-performance analytical database designed for modern, real-time analytics workloads. It follows a lakehouse-oriented architecture, enabling unified analytics across both real-time and batch data without maintaining separate systems for streaming and offline processing.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;StarRocks supports high-concurrency, low-latency queries directly on fresh data, making it well suited for dashboards, operational analytics, and data-driven applications. At the same time, it integrates naturally with data lakes and ELT pipelines, allowing teams to build scalable analytics platforms without sacrificing performance or consistency.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In this architecture, StarRocks serves as the analytical execution layer where raw data is continuously ingested and stored, while dbt operates on top of it to define transformations, models, and governance logic. Together, they form a clean separation of responsibilities: StarRocks focuses on efficient data storage and query execution, while dbt manages modeling, version control, and testing.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Within this setup, dbt integrates seamlessly with ELT workflows, improving the efficiency, reliability, and overall controllability of data modeling and governance. In day-to-day development, when issues are identified in a specific data model, teams can quickly roll back changes at the branch level. With Git as the system of record, every change goes through code review and is deployed via automated CI/CD pipelines.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;As a result, SQL models, materialized views, and other data objects are promoted to production through standard pull request (PR) workflows, ensuring consistency, traceability, and operational safety across the entire lifecycle.&lt;/p&gt; 
&lt;p&gt;dbt also integrates seamlessly with the native StarRocks ecosystem, enabling version control across multiple types of data objects, including tables, views, materialized views (MVs), and tasks. In dbt, a &lt;em&gt;model&lt;/em&gt; is essentially a SQL template. A typical pattern is to first create a staging model for customer data, then reuse it as a dependency for downstream custom business models. dbt automatically resolves dependencies between models and manages execution order, eliminating the need for external scheduling tools—running dbt alone is sufficient to orchestrate the entire pipeline.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Automated Data Dictionary Generation&lt;/h3&gt; 
&lt;p&gt;From a documentation and asset management perspective, dbt can automatically generate data dictionaries and other documentation artifacts. Through dbt-generated HTML documentation, teams can easily explore field definitions, business meanings, underlying SQL logic, and upstream/downstream dependencies. The documentation interface can also be customized with enterprise branding, including logos and visual styles.&lt;/p&gt; 
&lt;h3&gt;Automated Data Lineage&lt;/h3&gt; 
&lt;div style="text-align: center;"&gt;
  &amp;nbsp; 
&lt;/div&gt; 
&lt;p&gt;Data lineage is a foundational component of any data governance framework. Large enterprises often manage thousands of tables and a vast portfolio of data products. In industries such as hospitality, where organizations operate across hotels, restaurants, and other business lines, it is common to build a unified &lt;strong&gt;Customer 360&lt;/strong&gt; view to consolidate data assets across domains.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In these environments, even small changes to upstream raw data can have far-reaching effects. A core challenge is quickly identifying which downstream models, reports, or data products may be impacted. Data lineage addresses this need by enabling precise impact analysis, giving teams clear visibility into dependencies and helping them assess the scope and risk of upstream changes or data quality issues.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Automated Data Quality Testing&lt;/h3&gt; 
&lt;p&gt;Beyond data lineage, automated data testing is a core pillar of dbt best practices. Teams can define a wide range of automated tests for their data models, such as scheduled daily validations to ensure that datasets continue to meet expected conditions. When anomalies are detected, alerts can be triggered immediately, enabling faster detection and resolution of data issues.&lt;/p&gt; 
&lt;p&gt;From an implementation standpoint, dbt model configurations are typically defined in YAML files. Each model includes metadata such as its name and description, which capture the model's purpose and business context, followed by field-level definitions. dbt provides a comprehensive set of built-in tests, such as &lt;code&gt;unique&lt;/code&gt; and &lt;code&gt;not_null&lt;/code&gt; to enforce common data quality constraints, including uniqueness and null checks.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In most OLAP databases, foreign key constraints are not enforced at the database level. To address this gap, dbt offers &lt;code&gt;ref&lt;/code&gt;-based and relationship testing capabilities, allowing teams to validate that models are correctly referenced by downstream tables or models. These tests are also centrally managed through YAML configuration files.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In practice, some teams further enhance this workflow by using AI-assisted tools to automatically generate and batch-maintain YAML configurations, significantly reducing manual effort while improving consistency.&lt;/p&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;How DataOps Workflows Improve Agility and Control in Data Projects&lt;/h2&gt; 
&lt;h3&gt;Key Components of DataOps&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Building on the principles of DevOps, DataOps extends the same engineering mindset into the data domain. In application development, DevOps has become the standard approach for managing code development, integration, and deployment. DataOps applies these proven practices to the full data lifecycle, from development, testing, deployment, and ongoing operations to improve both speed and reliability.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In a typical DataOps architecture, the areas covered by dbt are often clearly highlighted. These include data modeling, automated data quality testing, data lineage analysis, and documentation-driven validation—all of which are native capabilities of dbt. For stages beyond dbt's scope, teams typically integrate other well-established tools to complete the workflow. Project management platforms such as Jira is used to track model changes and defects; scheduling tools manage periodic model execution and data checks (hourly or daily); CI/CD tools like Jenkins handle automated integration and deployment; and downstream analytics and BI tools support data consumption, visualization, and decision-making.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Version Control Standards: Conventional Commits&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;At the version control layer, experienced teams typically adopt a standardized commit convention. One widely used approach is &lt;strong&gt;Conventional Commits&lt;/strong&gt;, which defines a structured format for commit messages. Its core idea is to clearly distinguish between different types of changes, such as feature enhancements and bug fixes, so that versioning and release management can be automated.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;For example, when a change introduces new functionality, such as adding a new dimension to an &lt;code&gt;order&lt;/code&gt; model, the &lt;strong&gt;minor version&lt;/strong&gt; is incremented (e.g., from 2.0 to 2.1). These changes are generally backward compatible. In contrast, bug fixes only trigger a &lt;strong&gt;patch version&lt;/strong&gt; increment (e.g., from 2.1.0 to 2.1.1).&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Conventional Commits also make it possible to generate release notes automatically. In the past, release managers often had to manually confirm changes with developers and compile release notes by hand—a process that was both time-consuming and error-prone. With Conventional Commits in place, this workflow can be fully automated.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;When commit messages follow the defined convention (for example, using the &lt;code&gt;fix&lt;/code&gt; prefix), they can be automatically parsed and aggregated into structured release notes. Developers focus on writing clear, standardized commit messages, while the tooling handles release note generation. As part of the CI/CD pipeline, release notes are created and updated automatically with each release, requiring no additional manual effort from the team.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;A CI/CD Automation Example in DataOps&lt;/h3&gt; 
&lt;p&gt;In a DataOps framework, the CI/CD workflow typically starts with a &lt;strong&gt;Pull Request (PR)&lt;/strong&gt;. Every change is introduced through a PR, which triggers a standardized validation and deployment pipeline.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;The pipeline begins with automated code quality checks, commonly referred to as &lt;em&gt;linting&lt;/em&gt;. Similar to style and syntax validation in application development, linting tools automatically analyze SQL models and related configurations to ensure they conform to predefined standards. Once these checks pass, the changes are deployed to a staging environment.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Using dbt as an example, individual models can be deployed to specific environments. This makes it possible to deploy a single model to staging and run targeted unit tests and data quality checks. After the automated tests pass, the workflow moves to a manual review stage, where reviewers examine the PR to validate the logic and assess its potential impact.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;Once the changes are approved and merged into the main branch, the system automatically packages a new version, updates the change history, and promotes the release to QA or production environments. This end-to-end process ensures that data changes are delivered with both speed and control—key characteristics of a mature DataOps practice.&lt;/p&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;Technical Breakthroughs with StarRocks for Real-Time and Batch Analytics&lt;/h2&gt; 
&lt;h3&gt;Traditional Siloed ETL Architectures in Lake–Warehouse Separation&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;In early-stage implementations, many teams relied on siloed ETL architectures. Using a hospitality industry scenario as an example, the initial setup typically consisted of multiple independent operational databases at the source layer. ETL jobs ran on fixed schedules (for example, every 15 minutes) to extract data from these systems and load it into multiple data warehouses. In parallel, separate data pipelines were built to support mobile applications, reporting systems, and other analytical workloads.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;This architecture came with several fundamental limitations. Data models lacked version control, making the system fragile and easy to break as changes accumulated. Testing was largely manual, which made it difficult to establish a consistent and repeatable quality assurance process. Documentation was also highly fragmented, often maintained as standalone Word files across different teams, resulting in poor consistency, limited traceability, and high maintenance overhead.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;The StarRocks ELT Framework&lt;/h3&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;With the introduction of StarRocks as an integrated platform, the overall architecture was redesigned into a unified lakehouse model that supports both real-time and batch processing. Using real-time CDC, data from multiple operational systems is continuously ingested into the data lake. On top of the lakehouse, an ELT-based framework enables rapid construction of application-facing data products.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;At the same time, data governance capabilities are implemented end-to-end across the pipeline. Version control is established around data models, data dictionaries are centrally maintained, and data lineage views are generated through tooling, providing full visibility into dependencies and impact.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;Results from StarRocks + dbt + DataOps in Practice&lt;/h3&gt; 
&lt;p&gt;In the re-architected system, near-real-time data simultaneously supports mobile applications, reporting dashboards, and behavioral analytics workloads. On top of this foundation, a unified “three-in-one” DataOps framework delivers several immediate and measurable benefits:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;strong&gt;Faster iteration and recovery.&lt;/strong&gt; Data models built with dbt can be updated and rolled back quickly, significantly improving development velocity and incident recovery time.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;More efficient delivery through standardized pipelines.&lt;/strong&gt; DataOps manages business requirements and data product delivery through well-defined pipelines, introducing Agile-style practices that standardize project management workflows and automated release cycles. This substantially shortens the end-to-end lead time from requirement definition to production.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Strong consistency between models and documentation.&lt;/strong&gt; Model definitions and their corresponding YAML files are version-controlled together in Git, creating a single source of truth. Any model change requires the related documentation to be updated; otherwise, the release pipeline will fail validation.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Higher analytical accuracy and reliability.&lt;/strong&gt; For enterprises operating dozens of hotel systems, data lineage analysis makes it possible to understand where a system sits within the overall data pipeline and to assess downstream impact before upgrades. In parallel, automated data testing performs continuous health checks on models, validating data correctness and alignment with expectations on a daily basis。&lt;br&gt;&lt;br&gt;&lt;br&gt;&lt;/li&gt; 
&lt;/ul&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fdataops-driven-governance-and-analytics-with-dbt-and-starrocks&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>User Stories</category>
      <category>Technology</category>
      <category>Partners</category>
      <pubDate>Thu, 12 Feb 2026 11:50:29 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/dataops-driven-governance-and-analytics-with-dbt-and-starrocks</guid>
      <dc:date>2026-02-12T11:50:29Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>Escaping the Small-File Trap: How StarRocks Optimizes Bulk Ingestion</title>
      <link>https://www.starrocks.io/blog/escaping-the-small-file-trap-how-starrocks-optimizes-bulk-ingestion</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/escaping-the-small-file-trap-how-starrocks-optimizes-bulk-ingestion" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%20133.png" alt="Escaping the Small-File Trap: How StarRocks Optimizes Bulk Ingestion" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About The Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;div&gt; 
           &lt;div&gt;
             Casey Luo, StarRocks Committer &amp;amp; Engineer at Celerdata 
           &lt;/div&gt; 
          &lt;/div&gt; 
          &lt;span&gt;&lt;/span&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;h2&gt;&lt;strong&gt;TL&lt;/strong&gt;&lt;strong&gt;;&lt;/strong&gt;&lt;strong&gt;DR&lt;/strong&gt;&lt;/h2&gt; 
        &lt;div&gt;
          Under the storage–compute separation (shared-data) architecture, one-time ingestion of massive historical datasets has become an amplified but often overlooked risk. This article explains how StarRocks rethinks the large-scale ingestion path at its source. By redesigning the write pipeline, from 
         &lt;strong&gt;memory → local disk spill → centralized merge → &lt;/strong&gt; 
         &lt;strong&gt;object storage&lt;/strong&gt;, StarRocks minimizes remote writes and redundant overhead, significantly reduces the number of S3 write operations, and improves write throughput by fully utilizing local I/O capacity. This approach addresses the small-file problem at its root, enabling higher efficiency and stability at a lower overall cost. 
        &lt;/div&gt; 
        &lt;blockquote&gt; 
         &lt;div&gt; 
          &lt;strong&gt;Note:&lt;/strong&gt; This optimization is available starting with StarRocks 3.5 and does not apply to earlier versions. 
         &lt;/div&gt; 
        &lt;/blockquote&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;span&gt;&amp;nbsp;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;h2&gt;Large-Scale Ingestion Becomes an "Amplified Problem" in Shared-Data Architecture&lt;/h2&gt; 
        &lt;div&gt;
          As more users migrate large volumes of historical data into StarRocks, one-time bulk ingestion has become a common operational pattern. On the surface, this looks like a straightforward offline data load. In practice, however, under a shared-data architecture backed by object storage, improper handling can easily trigger a chain reaction: degraded ingestion performance, explosive growth in small files at the storage layer, and ultimately impaired query performance. 
        &lt;/div&gt; 
        &lt;div&gt;
          As a distributed columnar database, StarRocks adopts an LSM-tree–like storage model. Newly ingested data is first written into in-memory memtables. After sorting and other processing, background threads flush these memtables to persistent storage, and subsequent compaction merges multiple small files into larger, ordered ones. Under normal incremental write workloads, this design balances write efficiency with query performance effectively. But when ingesting massive volumes of historical data in bulk, the same mechanisms can become a bottleneck—and the issues are significantly magnified: 
        &lt;/div&gt; 
        &lt;ul&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Huge data volumes and many tablets.&lt;/strong&gt; Historical datasets often span a large number of tablets. Each tablet maintains its own memtable, and under high-concurrency ingestion, memtables are flushed frequently, generating a large number of small files in a short time. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Limited compute resources during ingestion.&lt;/strong&gt; In shared-data architecture deployments, users often start with a small number of compute nodes (CNs), sometimes even a single CN, with modest CPU and memory. These constraints further exacerbate the pattern of small memtables, frequent flushes, and rapid accumulation of small files. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Early scale-down after ingestion.&lt;/strong&gt; One advantage of storage–compute separation is the ability to scale down or release compute resources immediately after bulk ingestion, retaining only the data in object storage to reduce costs. However, this also means that the large number of small files generated during ingestion may never be sufficiently compacted, leaving long-term fragmentation in the underlying storage. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Query&lt;/strong&gt; 
           &lt;strong&gt; performance degradation later on.&lt;/strong&gt; When the cluster is scaled back up and queries are run against these historical datasets, the need to scan and process a vast number of small files can significantly degrade query performance. 
          &lt;/div&gt; &lt;/li&gt; 
        &lt;/ul&gt; 
        &lt;div&gt;
          In short, these issues are more pronounced in storage–compute separation environments because users naturally favor completing large historical ingestions with minimal, lower-spec compute resources. This choice amplifies the small-file problem, which then cascades into long-term performance penalties during query execution. 
        &lt;/div&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;h2&gt;Reworking Bulk Ingestion at the Ingestion Entry Point&lt;/h2&gt; 
        &lt;div&gt;
          To truly address the small-file problem caused by large-scale ingestion, relying on downstream compaction alone is far from sufficient. A closer examination of the entire write pipeline reveals that the root causes are concentrated in several key areas: 
        &lt;/div&gt; 
        &lt;ul&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Memory constraints force premature flushes.&lt;/strong&gt; On compute nodes (CNs), memtables are often flushed before they are filled due to limited memory, resulting in relatively small files per flush. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;High-latency remote writes in storage–compute separation.&lt;/strong&gt; Under a storage–compute separation architecture, every flush writes directly to object storage. The combination of high-latency remote I/O and frequent write operations significantly degrades ingestion throughput. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Repeated heavy write work.&lt;/strong&gt; Each flush triggers the full write pipeline: sorting, encoding, compression, and index construction. Repeating these CPU-intensive steps for small batches wastes substantial compute resources. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Redundant work during &lt;/strong&gt; 
           &lt;strong&gt;compaction&lt;/strong&gt; 
           &lt;strong&gt;.&lt;/strong&gt; The excessive number of small files must later be read again and merged during compaction. Much of the earlier sorting and encoding effort becomes partially redundant, further amplifying resource waste. 
          &lt;/div&gt; &lt;/li&gt; 
        &lt;/ul&gt; 
        &lt;div&gt;
          Based on this analysis, StarRocks redesigns the bulk ingestion write path for storage–compute separation scenarios, optimizing it at the entry point of the ingestion pipeline. 
        &lt;/div&gt; 
        &lt;h3&gt;1. Write Phase: Spill to Local Disk First&lt;/h3&gt; 
        &lt;div&gt;
          When a memtable is full, data is no longer written directly to object storage. Instead, StarRocks spills intermediate data to local disks on the CN. This approach avoids high-latency object storage writes and prevents repeated execution of heavy operations, such as sorting and encoding, before the data has stabilized. If local disk space becomes constrained, intermediate data can be selectively spilled to object storage (for example, S3) to ensure overall system robustness. 
        &lt;/div&gt; 
        &lt;h3&gt;2. Consolidation Phase: Merge First, Then Write to Object Storage&lt;/h3&gt; 
        &lt;div&gt;
          Once the bulk ingestion task completes, StarRocks performs a centralized merge on the temporary spill files. These files are consolidated into well-structured, appropriately sized data files, which are then written to object storage in a single, efficient step. 
        &lt;/div&gt; 
        &lt;div&gt;
          In summary, the redesigned bulk ingestion pipeline can be described as: 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;strong&gt;Memory → Local Disk Spill → Centralized Merge → &lt;/strong&gt; 
         &lt;strong&gt;Object Storage&lt;/strong&gt; 
        &lt;/div&gt; 
        &lt;div&gt;
          By restructuring the write path in this way, StarRocks tackles the small-file problem at its origin, delivering higher ingestion throughput, better resource utilization, and more stable performance in shared-data architecture deployments. 
        &lt;/div&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;div&gt;&lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;div&gt;
            &amp;nbsp; 
          &lt;/div&gt; 
          &lt;div&gt;
            This optimized bulk ingestion path delivers clear benefits across three key dimensions: 
          &lt;/div&gt; 
          &lt;ol start="1"&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Higher write &lt;/strong&gt; 
             &lt;strong&gt;throughput&lt;/strong&gt; 
             &lt;strong&gt; during ingestion.&lt;/strong&gt; When a memtable fills up, StarRocks spills intermediate results only to local disk instead of writing directly to backend object storage. By avoiding high-latency remote writes at this stage, the system significantly improves ingestion performance. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Lower CPU and resource overhead.&lt;/strong&gt; During the spill phase, the system simply persists memtable data to disk without triggering the full write pipeline, such as global sorting, encoding, or index construction. Skipping these expensive operations at intermediate stages reduces unnecessary CPU and memory consumption. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Fewer, larger files with more predictable &lt;/strong&gt; 
             &lt;strong&gt;query&lt;/strong&gt; 
             &lt;strong&gt; performance.&lt;/strong&gt; All temporary spill files are merged in a single, centralized step before being written to object storage, producing a much smaller number of well-sized data files. This dramatically reduces small-file proliferation and largely eliminates reliance on background compaction. As a result, even if queries are issued immediately after ingestion completes, they can run with stable and predictable performance. 
            &lt;/div&gt; &lt;/li&gt; 
          &lt;/ol&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
         &lt;div&gt; 
          &lt;h2&gt;Performance Comparison&lt;/h2&gt; 
          &lt;div&gt;
            To evaluate the real-world impact of the bulk ingestion optimizations described above, we designed two comparative tests on a shared-data architecture cluster: 
          &lt;/div&gt; 
          &lt;ul&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Single-concurrency scenario:&lt;/strong&gt; A single ingestion job loading 
             &lt;strong&gt;1 TB of data&lt;/strong&gt;, comparing ingestion time and post-ingestion query performance before and after the optimization. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;High-concurrency scenario:&lt;/strong&gt; 
             &lt;strong&gt;10 &lt;/strong&gt; 
             &lt;strong&gt;concurrent&lt;/strong&gt; 
             &lt;strong&gt; ingestion jobs&lt;/strong&gt;, each loading 
             &lt;strong&gt;100 &lt;/strong&gt; 
             &lt;strong&gt;GB&lt;/strong&gt; (still 
             &lt;strong&gt;1 &lt;/strong&gt; 
             &lt;strong&gt;TB&lt;/strong&gt; in total), comparing ingestion throughput and query performance after ingestion, before and after the optimization. 
            &lt;/div&gt; &lt;/li&gt; 
          &lt;/ul&gt; 
          &lt;h3&gt;Test 1: Single-Concurrency Bulk Ingestion&lt;/h3&gt; 
          &lt;div&gt;
            In this test, we used 
           &lt;strong&gt;Broker Load&lt;/strong&gt; to ingest a 
           &lt;strong&gt;1 &lt;/strong&gt; 
           &lt;strong&gt;TB&lt;/strong&gt; 
           &lt;strong&gt; dataset&lt;/strong&gt; (approximately 
           &lt;strong&gt;270 million rows&lt;/strong&gt;) in a single concurrent job. 
          &lt;/div&gt; 
          &lt;div&gt;
            Before the optimization: 
          &lt;/div&gt; 
          &lt;ul&gt; 
           &lt;li&gt; 
            &lt;div&gt;
              The ingestion phase itself took approximately 2 hours and 15 minutes. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt;
              After ingestion was completed, the system spent an additional 34 minutes performing background compaction. 
            &lt;/div&gt; &lt;/li&gt; 
          &lt;/ul&gt; 
          &lt;div&gt;
            From a user’s perspective, the total time from submitting the ingestion job to the system returning to a stable, query-ready state was approximately 2 hours and 50 minutes. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 3. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 10409
State: FINISHED
Type: BROKER
SinkRows: 270000000
LoadStartTime: 2024-12-27 10:59:12
LoadFinishTime: 2024-12-27 13:14:04&lt;/pre&gt; 
          &lt;div&gt;
            After the ingestion was completed, the compaction score for the partition was: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-yaml"&gt;&lt;span class="hljs-attr"&gt;AvgCS: 358.06 P50CS: 299.00 MaxCS:&lt;/span&gt; &lt;span class="hljs-number"&gt;1056.00&lt;/span&gt;&lt;/pre&gt; 
          &lt;div&gt;
            When the ingestion finished, the following query was executed immediately: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-sql"&gt;mysql&lt;span class="hljs-operator"&gt;&amp;gt;&lt;/span&gt; &lt;span class="hljs-keyword"&gt;select&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-keyword"&gt;from&lt;/span&gt; duplicate_21_0;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-number"&gt;270000000&lt;/span&gt; &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-number"&gt;1&lt;/span&gt; &lt;span class="hljs-type"&gt;row&lt;/span&gt; &lt;span class="hljs-keyword"&gt;in&lt;/span&gt; &lt;span class="hljs-keyword"&gt;set&lt;/span&gt; (&lt;span class="hljs-number"&gt;56.25&lt;/span&gt; sec)&lt;/pre&gt; 
          &lt;div&gt;
            After the optimization, the total ingestion time was approximately 
           &lt;strong&gt;2 hours and 42 minutes&lt;/strong&gt;. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 2. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 10642
State: FINISHED
Type: BROKER
SinkRows: 270000000
LoadStartTime: 2024-12-27 16:14:08
LoadFinishTime: 2024-12-27 18:56:00&lt;/pre&gt; 
          &lt;div&gt;
            After ingestion was completed, the compaction score was already at the optimal level, requiring no background compaction. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-makefile"&gt;&lt;span class="hljs-section"&gt;AvgCS: 2.39 P50CS: 2.00 MaxCS: 5.00&lt;/span&gt;&lt;/pre&gt; 
          &lt;div&gt;
            Immediately after the ingestion was completed, the query was executed: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-sql"&gt;mysql&lt;span class="hljs-operator"&gt;&amp;gt;&lt;/span&gt; &lt;span class="hljs-keyword"&gt;select&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-keyword"&gt;from&lt;/span&gt; duplicate_21_0;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-number"&gt;270000000&lt;/span&gt; &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-number"&gt;1&lt;/span&gt; &lt;span class="hljs-type"&gt;row&lt;/span&gt; &lt;span class="hljs-keyword"&gt;in&lt;/span&gt; &lt;span class="hljs-keyword"&gt;set&lt;/span&gt; (&lt;span class="hljs-number"&gt;0.72&lt;/span&gt; sec)&lt;/pre&gt; 
          &lt;h3&gt;Test 2: High-Concurrency Bulk Ingestion Stress Test&lt;/h3&gt; 
          &lt;div&gt;
            In this test, we ran a high-concurrency ingestion workload on a total dataset of 
           &lt;strong&gt;1 &lt;/strong&gt; 
           &lt;strong&gt;TB&lt;/strong&gt;. The target table contained 
           &lt;strong&gt;28 partitions&lt;/strong&gt;, with 
           &lt;strong&gt;256 tablets per partition&lt;/strong&gt;. 
          &lt;/div&gt; 
          &lt;div&gt;
            Before the optimization, ingestion was constrained by the CPU and memory resources of the compute nodes. As a result, the workload failed to complete within the 
           &lt;strong&gt;4-hour timeout window&lt;/strong&gt; and was eventually 
           &lt;strong&gt;automatically canceled by the system&lt;/strong&gt;. The final job state is shown below: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 10. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 11458
State: CANCELLED
Type: BROKER
Priority: NORMAL
ScanRows: 21905408
LoadStartTime: 2025-01-06 17:11:46
LoadFinishTime: 2025-01-06 21:11:44&lt;/pre&gt; 
          &lt;div&gt;
            After the optimization: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 20. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 28336
State: FINISHED
Type: BROKER
Priority: NORMAL
ScanRows: 30000000
LoadStartTime: 2025-01-06 20:10:49
LoadFinishTime: 2025-01-06 20:27:59&lt;/pre&gt; 
          &lt;div&gt;
            Under the same conditions, the 
           &lt;strong&gt;10 &lt;/strong&gt; 
           &lt;strong&gt;concurrent&lt;/strong&gt; 
           &lt;strong&gt; ingestion jobs&lt;/strong&gt; started at 2025-01-06 20:10:49 and all completed by 2025-01-06 20:36:10, for a total runtime of approximately 
           &lt;strong&gt;25 minutes&lt;/strong&gt;. 
          &lt;/div&gt; 
          &lt;div&gt;
            Although these jobs did trigger the compaction threshold, the system’s compaction score remained within a healthy and stable range throughout the ingestion and at completion. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-makefile"&gt;&lt;span class="hljs-section"&gt;vgCS: 10.00 P50CS: 10.00 MaxCS: 10.00&lt;/span&gt;&lt;/pre&gt; 
          &lt;div&gt;
            In addition, we can compare several key backend object storage metrics before and after the optimization: 
          &lt;/div&gt; 
          &lt;div&gt;
            &amp;nbsp; 
          &lt;/div&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt;&lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div style="text-align: center;"&gt;
          (Key S3 Metrics Before Optimization) 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div style="text-align: center;"&gt;
         (Key S3 Metrics After Optimization) 
       &lt;/div&gt; 
       &lt;div style="text-align: center;"&gt;
         &amp;nbsp; 
       &lt;/div&gt; 
       &lt;div style="text-align: left;"&gt;&lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
       &lt;div&gt; 
        &lt;div style="text-align: center;"&gt;
          (Comparison of Local Disk I/O Utilization Before and After Optimization) 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;div&gt;
          The results clearly show that after this optimization is enabled: 
        &lt;/div&gt; 
        &lt;ol start="1"&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Significantly fewer writes to S3, with much higher &lt;/strong&gt; 
           &lt;strong&gt;throughput&lt;/strong&gt; 
           &lt;strong&gt;.&lt;/strong&gt; The number of write operations to S3 is drastically reduced, while overall write throughput increases substantially. At the same time, the average object size grows significantly, helping lower storage costs and improving both read and write efficiency. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;More effective utilization of local disk &lt;/strong&gt; 
           &lt;strong&gt;I/O&lt;/strong&gt; 
           &lt;strong&gt; during ingestion.&lt;/strong&gt; The ingestion process is able to fully leverage local disk I/O capacity, leading to a noticeable improvement in overall ingestion performance. 
          &lt;/div&gt; &lt;/li&gt; 
        &lt;/ol&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;h2&gt;Summary&lt;/h2&gt; 
        &lt;div&gt;
          By optimizing bulk data ingestion at the engine level, StarRocks effectively avoids the proliferation of small files that commonly occurs under resource constraints, especially limited memory, during historical data backfill. This enables users in shared-data architecture environments to achieve higher efficiency and more stable performance with lower infrastructure investment. 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;&lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/escaping-the-small-file-trap-how-starrocks-optimizes-bulk-ingestion" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%20133.png" alt="Escaping the Small-File Trap: How StarRocks Optimizes Bulk Ingestion" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About The Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;div&gt; 
           &lt;div&gt;
             Casey Luo, StarRocks Committer &amp;amp; Engineer at Celerdata 
           &lt;/div&gt; 
          &lt;/div&gt; 
          &lt;span&gt;&lt;/span&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;h2&gt;&lt;strong&gt;TL&lt;/strong&gt;&lt;strong&gt;;&lt;/strong&gt;&lt;strong&gt;DR&lt;/strong&gt;&lt;/h2&gt; 
        &lt;div&gt;
          Under the storage–compute separation (shared-data) architecture, one-time ingestion of massive historical datasets has become an amplified but often overlooked risk. This article explains how StarRocks rethinks the large-scale ingestion path at its source. By redesigning the write pipeline, from 
         &lt;strong&gt;memory → local disk spill → centralized merge → &lt;/strong&gt; 
         &lt;strong&gt;object storage&lt;/strong&gt;, StarRocks minimizes remote writes and redundant overhead, significantly reduces the number of S3 write operations, and improves write throughput by fully utilizing local I/O capacity. This approach addresses the small-file problem at its root, enabling higher efficiency and stability at a lower overall cost. 
        &lt;/div&gt; 
        &lt;blockquote&gt; 
         &lt;div&gt; 
          &lt;strong&gt;Note:&lt;/strong&gt; This optimization is available starting with StarRocks 3.5 and does not apply to earlier versions. 
         &lt;/div&gt; 
        &lt;/blockquote&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;span&gt;&amp;nbsp;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;h2&gt;Large-Scale Ingestion Becomes an "Amplified Problem" in Shared-Data Architecture&lt;/h2&gt; 
        &lt;div&gt;
          As more users migrate large volumes of historical data into StarRocks, one-time bulk ingestion has become a common operational pattern. On the surface, this looks like a straightforward offline data load. In practice, however, under a shared-data architecture backed by object storage, improper handling can easily trigger a chain reaction: degraded ingestion performance, explosive growth in small files at the storage layer, and ultimately impaired query performance. 
        &lt;/div&gt; 
        &lt;div&gt;
          As a distributed columnar database, StarRocks adopts an LSM-tree–like storage model. Newly ingested data is first written into in-memory memtables. After sorting and other processing, background threads flush these memtables to persistent storage, and subsequent compaction merges multiple small files into larger, ordered ones. Under normal incremental write workloads, this design balances write efficiency with query performance effectively. But when ingesting massive volumes of historical data in bulk, the same mechanisms can become a bottleneck—and the issues are significantly magnified: 
        &lt;/div&gt; 
        &lt;ul&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Huge data volumes and many tablets.&lt;/strong&gt; Historical datasets often span a large number of tablets. Each tablet maintains its own memtable, and under high-concurrency ingestion, memtables are flushed frequently, generating a large number of small files in a short time. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Limited compute resources during ingestion.&lt;/strong&gt; In shared-data architecture deployments, users often start with a small number of compute nodes (CNs), sometimes even a single CN, with modest CPU and memory. These constraints further exacerbate the pattern of small memtables, frequent flushes, and rapid accumulation of small files. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Early scale-down after ingestion.&lt;/strong&gt; One advantage of storage–compute separation is the ability to scale down or release compute resources immediately after bulk ingestion, retaining only the data in object storage to reduce costs. However, this also means that the large number of small files generated during ingestion may never be sufficiently compacted, leaving long-term fragmentation in the underlying storage. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Query&lt;/strong&gt; 
           &lt;strong&gt; performance degradation later on.&lt;/strong&gt; When the cluster is scaled back up and queries are run against these historical datasets, the need to scan and process a vast number of small files can significantly degrade query performance. 
          &lt;/div&gt; &lt;/li&gt; 
        &lt;/ul&gt; 
        &lt;div&gt;
          In short, these issues are more pronounced in storage–compute separation environments because users naturally favor completing large historical ingestions with minimal, lower-spec compute resources. This choice amplifies the small-file problem, which then cascades into long-term performance penalties during query execution. 
        &lt;/div&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;h2&gt;Reworking Bulk Ingestion at the Ingestion Entry Point&lt;/h2&gt; 
        &lt;div&gt;
          To truly address the small-file problem caused by large-scale ingestion, relying on downstream compaction alone is far from sufficient. A closer examination of the entire write pipeline reveals that the root causes are concentrated in several key areas: 
        &lt;/div&gt; 
        &lt;ul&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Memory constraints force premature flushes.&lt;/strong&gt; On compute nodes (CNs), memtables are often flushed before they are filled due to limited memory, resulting in relatively small files per flush. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;High-latency remote writes in storage–compute separation.&lt;/strong&gt; Under a storage–compute separation architecture, every flush writes directly to object storage. The combination of high-latency remote I/O and frequent write operations significantly degrades ingestion throughput. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Repeated heavy write work.&lt;/strong&gt; Each flush triggers the full write pipeline: sorting, encoding, compression, and index construction. Repeating these CPU-intensive steps for small batches wastes substantial compute resources. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Redundant work during &lt;/strong&gt; 
           &lt;strong&gt;compaction&lt;/strong&gt; 
           &lt;strong&gt;.&lt;/strong&gt; The excessive number of small files must later be read again and merged during compaction. Much of the earlier sorting and encoding effort becomes partially redundant, further amplifying resource waste. 
          &lt;/div&gt; &lt;/li&gt; 
        &lt;/ul&gt; 
        &lt;div&gt;
          Based on this analysis, StarRocks redesigns the bulk ingestion write path for storage–compute separation scenarios, optimizing it at the entry point of the ingestion pipeline. 
        &lt;/div&gt; 
        &lt;h3&gt;1. Write Phase: Spill to Local Disk First&lt;/h3&gt; 
        &lt;div&gt;
          When a memtable is full, data is no longer written directly to object storage. Instead, StarRocks spills intermediate data to local disks on the CN. This approach avoids high-latency object storage writes and prevents repeated execution of heavy operations, such as sorting and encoding, before the data has stabilized. If local disk space becomes constrained, intermediate data can be selectively spilled to object storage (for example, S3) to ensure overall system robustness. 
        &lt;/div&gt; 
        &lt;h3&gt;2. Consolidation Phase: Merge First, Then Write to Object Storage&lt;/h3&gt; 
        &lt;div&gt;
          Once the bulk ingestion task completes, StarRocks performs a centralized merge on the temporary spill files. These files are consolidated into well-structured, appropriately sized data files, which are then written to object storage in a single, efficient step. 
        &lt;/div&gt; 
        &lt;div&gt;
          In summary, the redesigned bulk ingestion pipeline can be described as: 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;strong&gt;Memory → Local Disk Spill → Centralized Merge → &lt;/strong&gt; 
         &lt;strong&gt;Object Storage&lt;/strong&gt; 
        &lt;/div&gt; 
        &lt;div&gt;
          By restructuring the write path in this way, StarRocks tackles the small-file problem at its origin, delivering higher ingestion throughput, better resource utilization, and more stable performance in shared-data architecture deployments. 
        &lt;/div&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;div&gt;&lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;div&gt;
            &amp;nbsp; 
          &lt;/div&gt; 
          &lt;div&gt;
            This optimized bulk ingestion path delivers clear benefits across three key dimensions: 
          &lt;/div&gt; 
          &lt;ol start="1"&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Higher write &lt;/strong&gt; 
             &lt;strong&gt;throughput&lt;/strong&gt; 
             &lt;strong&gt; during ingestion.&lt;/strong&gt; When a memtable fills up, StarRocks spills intermediate results only to local disk instead of writing directly to backend object storage. By avoiding high-latency remote writes at this stage, the system significantly improves ingestion performance. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Lower CPU and resource overhead.&lt;/strong&gt; During the spill phase, the system simply persists memtable data to disk without triggering the full write pipeline, such as global sorting, encoding, or index construction. Skipping these expensive operations at intermediate stages reduces unnecessary CPU and memory consumption. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Fewer, larger files with more predictable &lt;/strong&gt; 
             &lt;strong&gt;query&lt;/strong&gt; 
             &lt;strong&gt; performance.&lt;/strong&gt; All temporary spill files are merged in a single, centralized step before being written to object storage, producing a much smaller number of well-sized data files. This dramatically reduces small-file proliferation and largely eliminates reliance on background compaction. As a result, even if queries are issued immediately after ingestion completes, they can run with stable and predictable performance. 
            &lt;/div&gt; &lt;/li&gt; 
          &lt;/ol&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
         &lt;div&gt; 
          &lt;h2&gt;Performance Comparison&lt;/h2&gt; 
          &lt;div&gt;
            To evaluate the real-world impact of the bulk ingestion optimizations described above, we designed two comparative tests on a shared-data architecture cluster: 
          &lt;/div&gt; 
          &lt;ul&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;Single-concurrency scenario:&lt;/strong&gt; A single ingestion job loading 
             &lt;strong&gt;1 TB of data&lt;/strong&gt;, comparing ingestion time and post-ingestion query performance before and after the optimization. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt; 
             &lt;strong&gt;High-concurrency scenario:&lt;/strong&gt; 
             &lt;strong&gt;10 &lt;/strong&gt; 
             &lt;strong&gt;concurrent&lt;/strong&gt; 
             &lt;strong&gt; ingestion jobs&lt;/strong&gt;, each loading 
             &lt;strong&gt;100 &lt;/strong&gt; 
             &lt;strong&gt;GB&lt;/strong&gt; (still 
             &lt;strong&gt;1 &lt;/strong&gt; 
             &lt;strong&gt;TB&lt;/strong&gt; in total), comparing ingestion throughput and query performance after ingestion, before and after the optimization. 
            &lt;/div&gt; &lt;/li&gt; 
          &lt;/ul&gt; 
          &lt;h3&gt;Test 1: Single-Concurrency Bulk Ingestion&lt;/h3&gt; 
          &lt;div&gt;
            In this test, we used 
           &lt;strong&gt;Broker Load&lt;/strong&gt; to ingest a 
           &lt;strong&gt;1 &lt;/strong&gt; 
           &lt;strong&gt;TB&lt;/strong&gt; 
           &lt;strong&gt; dataset&lt;/strong&gt; (approximately 
           &lt;strong&gt;270 million rows&lt;/strong&gt;) in a single concurrent job. 
          &lt;/div&gt; 
          &lt;div&gt;
            Before the optimization: 
          &lt;/div&gt; 
          &lt;ul&gt; 
           &lt;li&gt; 
            &lt;div&gt;
              The ingestion phase itself took approximately 2 hours and 15 minutes. 
            &lt;/div&gt; &lt;/li&gt; 
           &lt;li&gt; 
            &lt;div&gt;
              After ingestion was completed, the system spent an additional 34 minutes performing background compaction. 
            &lt;/div&gt; &lt;/li&gt; 
          &lt;/ul&gt; 
          &lt;div&gt;
            From a user’s perspective, the total time from submitting the ingestion job to the system returning to a stable, query-ready state was approximately 2 hours and 50 minutes. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 3. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 10409
State: FINISHED
Type: BROKER
SinkRows: 270000000
LoadStartTime: 2024-12-27 10:59:12
LoadFinishTime: 2024-12-27 13:14:04&lt;/pre&gt; 
          &lt;div&gt;
            After the ingestion was completed, the compaction score for the partition was: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-yaml"&gt;&lt;span class="hljs-attr"&gt;AvgCS: 358.06 P50CS: 299.00 MaxCS:&lt;/span&gt; &lt;span class="hljs-number"&gt;1056.00&lt;/span&gt;&lt;/pre&gt; 
          &lt;div&gt;
            When the ingestion finished, the following query was executed immediately: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-sql"&gt;mysql&lt;span class="hljs-operator"&gt;&amp;gt;&lt;/span&gt; &lt;span class="hljs-keyword"&gt;select&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-keyword"&gt;from&lt;/span&gt; duplicate_21_0;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-number"&gt;270000000&lt;/span&gt; &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-number"&gt;1&lt;/span&gt; &lt;span class="hljs-type"&gt;row&lt;/span&gt; &lt;span class="hljs-keyword"&gt;in&lt;/span&gt; &lt;span class="hljs-keyword"&gt;set&lt;/span&gt; (&lt;span class="hljs-number"&gt;56.25&lt;/span&gt; sec)&lt;/pre&gt; 
          &lt;div&gt;
            After the optimization, the total ingestion time was approximately 
           &lt;strong&gt;2 hours and 42 minutes&lt;/strong&gt;. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 2. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 10642
State: FINISHED
Type: BROKER
SinkRows: 270000000
LoadStartTime: 2024-12-27 16:14:08
LoadFinishTime: 2024-12-27 18:56:00&lt;/pre&gt; 
          &lt;div&gt;
            After ingestion was completed, the compaction score was already at the optimal level, requiring no background compaction. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-makefile"&gt;&lt;span class="hljs-section"&gt;AvgCS: 2.39 P50CS: 2.00 MaxCS: 5.00&lt;/span&gt;&lt;/pre&gt; 
          &lt;div&gt;
            Immediately after the ingestion was completed, the query was executed: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-sql"&gt;mysql&lt;span class="hljs-operator"&gt;&amp;gt;&lt;/span&gt; &lt;span class="hljs-keyword"&gt;select&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-keyword"&gt;from&lt;/span&gt; duplicate_21_0;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-built_in"&gt;count&lt;/span&gt;(&lt;span class="hljs-operator"&gt;*&lt;/span&gt;) &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-operator"&gt;|&lt;/span&gt; &lt;span class="hljs-number"&gt;270000000&lt;/span&gt; &lt;span class="hljs-operator"&gt;|&lt;/span&gt;
&lt;span class="hljs-operator"&gt;+&lt;/span&gt;&lt;span class="hljs-comment"&gt;-----------+&lt;/span&gt;
&lt;span class="hljs-number"&gt;1&lt;/span&gt; &lt;span class="hljs-type"&gt;row&lt;/span&gt; &lt;span class="hljs-keyword"&gt;in&lt;/span&gt; &lt;span class="hljs-keyword"&gt;set&lt;/span&gt; (&lt;span class="hljs-number"&gt;0.72&lt;/span&gt; sec)&lt;/pre&gt; 
          &lt;h3&gt;Test 2: High-Concurrency Bulk Ingestion Stress Test&lt;/h3&gt; 
          &lt;div&gt;
            In this test, we ran a high-concurrency ingestion workload on a total dataset of 
           &lt;strong&gt;1 &lt;/strong&gt; 
           &lt;strong&gt;TB&lt;/strong&gt;. The target table contained 
           &lt;strong&gt;28 partitions&lt;/strong&gt;, with 
           &lt;strong&gt;256 tablets per partition&lt;/strong&gt;. 
          &lt;/div&gt; 
          &lt;div&gt;
            Before the optimization, ingestion was constrained by the CPU and memory resources of the compute nodes. As a result, the workload failed to complete within the 
           &lt;strong&gt;4-hour timeout window&lt;/strong&gt; and was eventually 
           &lt;strong&gt;automatically canceled by the system&lt;/strong&gt;. The final job state is shown below: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 10. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 11458
State: CANCELLED
Type: BROKER
Priority: NORMAL
ScanRows: 21905408
LoadStartTime: 2025-01-06 17:11:46
LoadFinishTime: 2025-01-06 21:11:44&lt;/pre&gt; 
          &lt;div&gt;
            After the optimization: 
          &lt;/div&gt; 
          &lt;pre class="hljs language-markdown"&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;*** 20. row **&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;&lt;span class="hljs-strong"&gt;****&lt;/span&gt;*
JobId: 28336
State: FINISHED
Type: BROKER
Priority: NORMAL
ScanRows: 30000000
LoadStartTime: 2025-01-06 20:10:49
LoadFinishTime: 2025-01-06 20:27:59&lt;/pre&gt; 
          &lt;div&gt;
            Under the same conditions, the 
           &lt;strong&gt;10 &lt;/strong&gt; 
           &lt;strong&gt;concurrent&lt;/strong&gt; 
           &lt;strong&gt; ingestion jobs&lt;/strong&gt; started at 2025-01-06 20:10:49 and all completed by 2025-01-06 20:36:10, for a total runtime of approximately 
           &lt;strong&gt;25 minutes&lt;/strong&gt;. 
          &lt;/div&gt; 
          &lt;div&gt;
            Although these jobs did trigger the compaction threshold, the system’s compaction score remained within a healthy and stable range throughout the ingestion and at completion. 
          &lt;/div&gt; 
          &lt;pre class="hljs language-makefile"&gt;&lt;span class="hljs-section"&gt;vgCS: 10.00 P50CS: 10.00 MaxCS: 10.00&lt;/span&gt;&lt;/pre&gt; 
          &lt;div&gt;
            In addition, we can compare several key backend object storage metrics before and after the optimization: 
          &lt;/div&gt; 
          &lt;div&gt;
            &amp;nbsp; 
          &lt;/div&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt;&lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div style="text-align: center;"&gt;
          (Key S3 Metrics Before Optimization) 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div style="text-align: center;"&gt;
         (Key S3 Metrics After Optimization) 
       &lt;/div&gt; 
       &lt;div style="text-align: center;"&gt;
         &amp;nbsp; 
       &lt;/div&gt; 
       &lt;div style="text-align: left;"&gt;&lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
       &lt;div&gt; 
        &lt;div style="text-align: center;"&gt;
          (Comparison of Local Disk I/O Utilization Before and After Optimization) 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;div&gt;
          The results clearly show that after this optimization is enabled: 
        &lt;/div&gt; 
        &lt;ol start="1"&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;Significantly fewer writes to S3, with much higher &lt;/strong&gt; 
           &lt;strong&gt;throughput&lt;/strong&gt; 
           &lt;strong&gt;.&lt;/strong&gt; The number of write operations to S3 is drastically reduced, while overall write throughput increases substantially. At the same time, the average object size grows significantly, helping lower storage costs and improving both read and write efficiency. 
          &lt;/div&gt; &lt;/li&gt; 
         &lt;li&gt; 
          &lt;div&gt; 
           &lt;strong&gt;More effective utilization of local disk &lt;/strong&gt; 
           &lt;strong&gt;I/O&lt;/strong&gt; 
           &lt;strong&gt; during ingestion.&lt;/strong&gt; The ingestion process is able to fully leverage local disk I/O capacity, leading to a noticeable improvement in overall ingestion performance. 
          &lt;/div&gt; &lt;/li&gt; 
        &lt;/ol&gt; 
        &lt;div&gt;
          &amp;nbsp; 
        &lt;/div&gt; 
        &lt;h2&gt;Summary&lt;/h2&gt; 
        &lt;div&gt;
          By optimizing bulk data ingestion at the engine level, StarRocks effectively avoids the proliferation of small files that commonly occurs under resource constraints, especially limited memory, during historical data backfill. This enables users in shared-data architecture environments to achieve higher efficiency and more stable performance with lower infrastructure investment. 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
      &lt;div&gt;&lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Fescaping-the-small-file-trap-how-starrocks-optimizes-bulk-ingestion&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Thu, 29 Jan 2026 06:36:44 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/escaping-the-small-file-trap-how-starrocks-optimizes-bulk-ingestion</guid>
      <dc:date>2026-01-29T06:36:44Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
    <item>
      <title>Inside StarRocks: Why Joins Are Faster Than You’d Expect</title>
      <link>https://www.starrocks.io/blog/inside-starrocks-why-joins-are-faster-than-youd-expect</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/inside-starrocks-why-joins-are-faster-than-youd-expect" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813890.png" alt="Inside StarRocks: Why Joins Are Faster Than You’d Expect" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About The Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;em&gt;&lt;span style="background-color: transparent;"&gt;Seaven He, StarRocks Committer, Engineer at Celerdata&lt;/span&gt;&lt;/em&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
   &lt;div&gt; 
    &lt;span style="background-color: transparent;"&gt;Joins are the hardest part of OLAP. Many systems can’t run them efficiently at scale, so teams denormalize into wide tables instead, 10× their storage, dealing with complex stream processing pipelines, and painfully slow and expensive schema evolution that triggers large backfills.&lt;/span&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt; 
&lt;p&gt;StarRocks takes the opposite approach: keep data normalized and make joins fast enough to run on the fly. The challenge is the plan. In a distributed system, the join search space is huge, and a good plan can be orders of magnitude faster.&lt;/p&gt; 
&lt;p&gt;This deep dive explains how StarRocks’ cost-based optimizer makes that possible, in four parts: join fundamentals and optimization challenges, logical join optimizations, join reordering, and distributed join planning. Finally, we examine real-world case studies from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;NAVER, Demandbase, and Shopee&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to illustrate how efficient join execution delivers tangible business value.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;h2&gt;Join Fundamentals and Optimization Challenges&lt;/h2&gt; 
&lt;h3&gt;&lt;strong&gt;1.1 Join Types&lt;/strong&gt;&lt;/h3&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;The diagram above illustrates several common join types:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;strong&gt;Cross Join&lt;/strong&gt;: Produces a Cartesian product between the left and right tables.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Full / Left / Right Outer Join&lt;/strong&gt;: For rows that do not find a match, outer joins return results with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;values filled in according to the join semantics—on both tables (full), the left table (left), or the right table (right).&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Anti Join&lt;/strong&gt;: Returns rows that do&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;not&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;have a matching counterpart in the join relationship. Anti-joins typically appear in query plans for&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NOT IN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NOT EXISTS&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;subqueries.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Semi Join&lt;/strong&gt;: The opposite of an anti-join, it returns only rows that&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;do&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;have a match in the join relationship, without producing duplicate result rows from the matching side.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Inner Join&lt;/strong&gt;: Returns the intersection of the left and right tables. Based on the join condition, it may generate one-to-many result rows.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;1.2 Challenges in Join Optimization&lt;/h3&gt; 
&lt;p&gt;Join performance optimization generally falls into two areas:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;improving the efficiency of join operators on a single node, and&lt;/li&gt; 
 &lt;li&gt;designing a reasonable join plan that minimizes input size and execution cost.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;This article focuses on the second aspect. To set the stage, we begin by examining the key challenges in join optimization.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 1: Multiple Join Implementation Strategies&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;As shown above, different join algorithms perform very differently depending on the scenario. For example, Sort-Merge Join can be significantly more efficient than Hash Join when operating on already sorted data. However, in distributed databases where data is typically hash-partitioned, Hash Join often outperforms Sort-Merge Join by a wide margin. As a result,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;the database must choose the most appropriate join strategy based on the specific workload and data characteristics.&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 2: Join Order Selection in Multi-Table Joins&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;In multi-table join scenarios, executing highly selective joins first can significantly improve overall query performance. However, determining the optimal join order is far from trivial.&lt;/p&gt; 
&lt;p&gt;As illustrated above, under a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep join tree&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;model, the number of possible join orders for&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;N&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;tables is on the order of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;2^n-1&lt;/code&gt;. Under a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;bushy join tree&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;model, the number of possible combinations grows even more dramatically, reaching&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;2^(n-1) * C(n-1)&lt;/code&gt;. For a database optimizer, the time and cost required to search for the optimal join order therefore increases&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;exponentially&lt;/strong&gt;, making join ordering one of the most challenging problems in query optimization.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 3: Difficulty in Estimating Join Effectiveness&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;Before query execution, it is extremely difficult for the database to accurately predict the real execution behavior of a join. A common assumption is that joining a small table with a large table is more selective than joining two large tables, but this is not always true.&lt;/p&gt; 
&lt;p&gt;In practice, one-to-many relationships are common, and in more complex queries, joins are often combined with filters, aggregations, and other operators. After data flows through multiple transformations, the optimizer’s ability to accurately estimate join input sizes and selectivity degrades significantly.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 4: A Single-Node Optimal Plan Is Not Necessarily Optimal in Distributed Systems&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;In distributed systems, data often needs to be&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;reshuffled&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;broadcast&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;across nodes so that the required records can participate in join computation. Distributed joins are no exception.&lt;/p&gt; 
&lt;p&gt;This introduces a key complication: an execution plan that is optimal in a single-node database may perform poorly in a distributed environment because it ignores data distribution and network transfer costs.&lt;/p&gt; 
&lt;p&gt;Therefore, when planning join execution strategies in distributed databases, the optimizer must explicitly account for data placement and communication overhead in addition to local execution efficiency.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;1.3 SQL Optimization Workflow&lt;/h3&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;In StarRocks, SQL optimization is primarily handled by the query optimizer and is mainly concentrated in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Rewrite&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Optimize&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;phases.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;1.4 Principles of Join Optimization&lt;/h3&gt; 
&lt;p&gt;At present, StarRocks primarily uses&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Hash Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;as its join algorithm. By default, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;right-hand table&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is used to build the hash table. Based on this design choice, we summarize five key optimization principles:&lt;/p&gt; 
&lt;ol&gt; 
 &lt;li&gt;Different join types have very different performance characteristics. Whenever possible, prefer higher-performance join types and avoid expensive ones. Based on the typical size of join outputs, the rough performance ranking is:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Semi Join / Anti Join &amp;gt; Inner Join &amp;gt; Outer Join &amp;gt; Full Outer Join &amp;gt; Cross Join&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;When using Hash Join, building the hash table on a smaller input is significantly more efficient than building it on a large table.&lt;/li&gt; 
 &lt;li&gt;In multi-table joins, prioritize joins with high selectivity.&lt;/li&gt; 
 &lt;li&gt;Minimize the amount of data participating in joins whenever possible.&lt;/li&gt; 
 &lt;li&gt;Minimize network overhead introduced by distributed joins.&lt;/li&gt; 
&lt;/ol&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;Join Logical Optimization&lt;/h2&gt; 
&lt;p&gt;This section introduces a set of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;heuristic rules&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;used by StarRocks to optimize joins at the logical level.&lt;/p&gt; 
&lt;h3&gt;2.1 Type Transformations&lt;/h3&gt; 
&lt;p&gt;The first group of optimizations directly follows the first join optimization principle discussed earlier:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;transform low-efficiency join types into more efficient ones whenever the semantics allow it&lt;/strong&gt;.&lt;/p&gt; 
&lt;p&gt;StarRocks currently applies three major transformation rules.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Rule 1: Converting a Cross Join into an Inner Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;A Cross Join can be rewritten as an Inner Join when it satisfies the following condition:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There exists at least one predicate that expresses a join relationship between the two tables.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1, t2 &lt;span&gt;WHERE&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;-- WHERE t1.v1 = t2.v1 is a join predicate&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;INNER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Rule 2: Converting an Outer Join into an Inner Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;A Left / Right Outer Join can be rewritten as an Inner Join when the following conditions are met:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There exists a predicate referencing the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;nullable side&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;of the outer join (Right table for a Left Outer Join, or Left table for a Right Outer Join).&lt;/li&gt; 
 &lt;li&gt;The predicate is a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;strict (null-rejecting) predicate&lt;/strong&gt;.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;-- t2.v1 &amp;gt; 0 is a strict predicate on t2&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;INNER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;⚠️&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Important note:&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;In an outer join,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause predicates participate in&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;null extension&lt;/strong&gt;, not filtering. Therefore, this rule does&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;not&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;apply to join predicates inside the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;AND&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;1&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;This query is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;not semantically equivalent&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;INNER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;AND&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;1&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;This introduces the concept of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;strict (null-rejecting) predicates&lt;/strong&gt;. In StarRocks, a predicate that filters out&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;values is considered a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;strict predicate&lt;/em&gt;, for example&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;a &amp;gt; 0&lt;/code&gt;. Predicates that do not eliminate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;values are classified as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;non-strict predicates&lt;/em&gt;, such as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;a IS NULL&lt;/code&gt;. Most predicates fall into the strict category; non-strict predicates are primarily those involving&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;IS NULL&lt;/code&gt;,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;IF&lt;/code&gt;,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;CASE WHEN&lt;/code&gt;, or certain function-based expressions.&lt;/p&gt; 
&lt;p&gt;To determine whether a predicate is strict, StarRocks uses a simple yet effective approach: all referenced columns are replaced with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;, and the expression is then simplified. If the result evaluates to&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;TRUE&lt;/code&gt;, it means the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause does not filter out rows with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;inputs, and the predicate is therefore non-strict. Conversely, if the result evaluates to&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;FALSE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;, the predicate is considered strict.&lt;/p&gt;  
&lt;div&gt; 
 &lt;br&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Rule 3: Converting a Full Outer Join into a Left / Right Outer Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;A Full Outer Join can be rewritten as a Left Outer Join or Right Outer Join when the following condition is satisfied:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There exists a strict predicate that can be bound exclusively to the left or right table.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;FULL&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t1.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;-- t1.v1 &amp;gt; 0 is a strict predicate on the left table&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t1.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;2.2 Predicate Pushdown&lt;/h3&gt; 
&lt;p&gt;Predicate pushdown is one of the most important and commonly used join optimization techniques. Its primary purpose is to&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;filter join inputs as early as possible&lt;/strong&gt;, thereby reducing the amount of data involved in the join and improving overall performance.&lt;/p&gt; 
&lt;p&gt;For predicates in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause, predicate pushdown can be applied—and may enable join type transformations—when the following conditions are satisfied:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;The join can be of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;any type&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;The&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicate can be&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;bound to one of the join inputs&lt;/strong&gt;.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;  &lt;br&gt;&lt;span&gt;From&lt;/span&gt; t1 &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t2 &lt;span&gt;On&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;br&gt;        &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t3 &lt;span&gt;On&lt;/span&gt; t2.v2 &lt;span&gt;=&lt;/span&gt; t3.v2 &lt;br&gt;&lt;span&gt;Where&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;1&lt;/span&gt; &lt;span&gt;And&lt;/span&gt; t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;span&gt;And&lt;/span&gt; t3.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;The predicate pushdown process proceeds as follows.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Step 1&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;:&lt;/p&gt; 
&lt;p&gt;Push down&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = 1 AND t2.v1 = 2)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t3.v2 = 3)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;separately. Since the join type transformation rules are satisfied,&lt;code&gt;(t1 LEFT OUTER JOIN t2) LEFT OUTER JOIN t3&lt;/code&gt;can be rewritten as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1 LEFT OUTER JOIN t2) INNER JOIN t3&lt;/code&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;  
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Step 2:&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;Continue pushing down&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = 1)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2)&lt;/code&gt;. At this point,&lt;code&gt;t1 LEFT OUTER JOIN t2&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be further transformed into&lt;code&gt;t1 INNER JOIN t2&lt;/code&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;It is important to note that&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;predicate pushdown rules for join predicates in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;ON&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause differ from those for the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;WHERE&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause&lt;/strong&gt;. We distinguish between two cases:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Inner Joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;other join types&lt;/strong&gt;.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Case 1: Inner Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Inner Joins&lt;/strong&gt;, pushing down join predicates in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause follows the same rules as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause predicate pushdown. This has already been discussed above and will not be repeated here.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Case 2: Outer / Semi / Anti Joins&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;For Outer, Semi, and Anti Joins, predicate pushdown on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause join predicates must satisfy the following constraints, and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;no join type transformation is allowed during the pushdown process&lt;/strong&gt;:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;The join must be a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left or Right Outer / Semi / Anti Join&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;The join predicate must be bindable&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;only to the nullable side&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;(the right input for a Left Join, or the left input for a Right Join).&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Consider the following example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;  &lt;br&gt;&lt;span&gt;From&lt;/span&gt; t1 &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t2 &lt;span&gt;On&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;And&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;1&lt;/span&gt; &lt;span&gt;And&lt;/span&gt; t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;br&gt;        &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t3 &lt;span&gt;On&lt;/span&gt; t2.v2 &lt;span&gt;=&lt;/span&gt; t3.v2 &lt;span&gt;And&lt;/span&gt; t3.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;The predicate pushdown proceeds as follows.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Step 1:&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;Push down the join predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t3.v2 = 3)&lt;/code&gt;, which can be bound to the right input of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t1 LEFT JOIN t2 LEFT JOIN t3&lt;/code&gt;. At this stage, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LEFT OUTER JOIN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;cannot&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;be converted into an&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;INNER JOIN&lt;/code&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;&lt;strong&gt;Step 2:&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;Push down the join predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2)&lt;/code&gt;, which can be bound to the right input of&lt;code&gt;t1 LEFT JOIN t2&lt;/code&gt;.&lt;/p&gt; 
&lt;p&gt;However, the predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = 1)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is bound to the left input. Pushing it down would filter rows from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t1&lt;/code&gt;, violating the semantics of a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LEFT OUTER JOIN&lt;/code&gt;. Therefore, this predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;cannot be pushed down&lt;/strong&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;h3&gt;2.3 Predicate Extraction&lt;/h3&gt; 
&lt;p&gt;In the predicate pushdown rules discussed earlier, only predicates with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;conjunctive semantics&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down. For example, in&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t1.v1 = 1 AND t2.v1 = 2 AND t3.v2 = 3&lt;/code&gt;, each sub-predicate is connected by conjunction, making pushdown straightforward. However, predicates with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;disjunctive semantics&lt;/strong&gt;, such as&lt;code&gt;t1.v1 = 1 OR t2.v1 = 2 OR t3.v2 = 3&lt;/code&gt;, cannot be pushed down directly.&lt;/p&gt; 
&lt;p&gt;In real-world queries, disjunctive predicates are quite common. To address this, StarRocks introduces an optimization called&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;predicate extraction (column value derivation)&lt;/strong&gt;. This technique derives conjunctive predicates from disjunctive ones by performing a series of union and intersection operations on column value ranges. The derived conjunctive predicates can then be pushed down to reduce join input size.&lt;/p&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before predicate extraction&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;)&lt;span&gt;OR&lt;/span&gt; (t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;Using column value derivation on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2 AND t1.v2 = 3) OR (t2.v1 &amp;gt; 5 AND t1.v2 = 4)&lt;/code&gt;, the optimizer can extract the following predicates:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;code&gt;t2.v1 &amp;gt;= 2&lt;/code&gt;&lt;/li&gt; 
 &lt;li&gt;&lt;code&gt;t1.v2 IN (3, 4)&lt;/code&gt;&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;The query can then be rewritten as:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; *&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 = t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 = &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;3&lt;/span&gt;)&lt;span&gt;OR&lt;/span&gt; (t2.v1 &amp;gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;4&lt;/span&gt;)&lt;span&gt;AND&lt;/span&gt; t2.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;&lt;br&gt;&lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;IN&lt;/span&gt; (&lt;span&gt;3&lt;/span&gt;, &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;It is important to note that the extracted predicates may form a superset of the original predicate ranges. As a result, they cannot safely replace the original predicates and must instead be applied in addition to them.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;2.4 Equivalence Derivation&lt;/h3&gt; 
&lt;p&gt;In addition to predicate extraction, another important predicate-level optimization is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;equivalence derivation&lt;/strong&gt;. This technique leverages&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;join equality conditions&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to infer value constraints on one side of the join from predicates applied to the other side.&lt;/p&gt; 
&lt;p&gt;Specifically, based on the join condition, value ranges on columns from the left table can be used to derive corresponding value ranges on columns from the right table, and vice versa.&lt;/p&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Original SQL&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;)&lt;br&gt;   &lt;span&gt;OR&lt;/span&gt; (t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;Using predicate extraction on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2 AND t1.v2 = 3) OR (t2.v1 &amp;gt; 5 AND t1.v2 = 4)&lt;/code&gt;, the optimizer can derive the following predicates:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;code&gt;t2.v1 &amp;gt;= 2&lt;/code&gt;&lt;/li&gt; 
 &lt;li&gt;&lt;code&gt;t1.v2 IN (3, 4)&lt;/code&gt;&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Resulting in:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; *&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 = t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 = &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;3&lt;/span&gt;)&lt;br&gt;   &lt;span&gt;OR&lt;/span&gt; (t2.v1 &amp;gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;4&lt;/span&gt;)&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t2.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;IN&lt;/span&gt; (&lt;span&gt;3&lt;/span&gt;, &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;Next, using the join predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = t2.v1)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;together with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t2.v1 &amp;gt;= 2&lt;/code&gt;, equivalence derivation can infer an additional predicate:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;code&gt;t1.v1 &amp;gt;= 2&lt;/code&gt;&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;The query can therefore be further rewritten as:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; *&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 = t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 = &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;3&lt;/span&gt;)&lt;br&gt;   &lt;span&gt;OR&lt;/span&gt; (t2.v1 &amp;gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;4&lt;/span&gt;)&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t2.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;IN&lt;/span&gt; (&lt;span&gt;3&lt;/span&gt;, &lt;span&gt;4&lt;/span&gt;)&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t1.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;&lt;strong&gt;Applicability and Constraints&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;The scope of equivalence derivation is more limited than predicate extraction. Predicate extraction can be applied to arbitrary predicates, whereas equivalence derivation, like predicate pushdown, has different constraints depending on the join type. As before, we distinguish between&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicates and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause join predicates.&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;WHERE&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicates:&lt;/strong&gt;&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There are almost no restrictions. Predicates can be derived from the left table to the right table and vice versa.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;&lt;strong&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;ON&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause join predicates:&lt;/strong&gt;&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Inner Joins&lt;/strong&gt;, the rules are the same as for&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicates—no additional constraints apply.&lt;/li&gt; 
 &lt;li&gt;For join types other than Inner Join, only&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Semi Joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Outer Joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;are supported, and derivation is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;one-directional only&lt;/strong&gt;, opposite to the join direction:&lt;/li&gt; 
 &lt;li&gt;For a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left Outer Join&lt;/strong&gt;, predicates can be derived from the left table to the right table.&lt;/li&gt; 
 &lt;li&gt;For a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Right Outer Join&lt;/strong&gt;, predicates can be derived from the right table to the left table.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Why Is Equivalence Derivation One-Directional for Outer / Semi Joins?&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;The reason is straightforward. Consider a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left Outer Join&lt;/strong&gt;. As discussed in predicate pushdown rules, only predicates on the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;right table&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down; predicates on the left table cannot, as doing so would violate the semantics of a left outer join.&lt;/p&gt; 
&lt;p&gt;For the same reason, predicates derived&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;from the right table&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and applied to the left table must also respect this constraint. In practice, such derived predicates on the preserved side do not help filter data early and instead introduce additional evaluation overhead. Therefore, equivalence derivation for Outer and Semi Joins is intentionally restricted to a single direction.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Implementation Details&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;StarRocks implements equivalence derivation by maintaining two internal maps:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;One map tracks&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;column-to-column equivalence relationships&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;The other map tracks&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;column-to-value or column-to-expression equivalences&lt;/strong&gt;.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;By performing lookups and inference across these two maps, the optimizer derives additional equivalent predicates. The overall mechanism is illustrated below:&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;h3&gt;2.5 Limit Pushdown&lt;/h3&gt; 
&lt;p&gt;In addition to predicates,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;&lt;strong&gt;LIMIT&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clauses can also be pushed down through joins&lt;/strong&gt;. When a query involves an&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Outer Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Cross Join&lt;/strong&gt;, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LIMIT&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down to child operators whose output row count is guaranteed to be stable.&lt;/p&gt; 
&lt;p&gt;For example, in a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left Outer Join&lt;/strong&gt;, the output row count is at least the same as that of the left input. Therefore, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LIMIT&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down to the left table (and symmetrically for a Right Outer Join).&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; (&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 LIMIT &lt;span&gt;100&lt;/span&gt;) t&lt;br&gt;&lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;h3&gt;Special Cases: Cross Join and Full Outer Join&lt;/h3&gt; 
&lt;p&gt;A Cross Join produces a Cartesian product, with output cardinality equal to&lt;code&gt;rows(left) × rows(right)&lt;/code&gt;. A Full Outer Join produces at least&lt;code&gt;rows(left) + rows(right)&lt;/code&gt;.&lt;/p&gt; 
&lt;p&gt;For these join types, a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LIMIT&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down to both inputs independently:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2&lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; (&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 LIMIT &lt;span&gt;100&lt;/span&gt;) x1&lt;br&gt;&lt;span&gt;JOIN&lt;/span&gt; (&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t2 LIMIT &lt;span&gt;100&lt;/span&gt;) &lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;Join Reordering&lt;/h2&gt; 
&lt;p&gt;Join reordering is used to determine the execution order of multi-table joins. The optimizer aims to execute&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;high-selectivity joins as early as possible&lt;/strong&gt;, thereby reducing the size of intermediate results and improving overall query performance.&lt;/p&gt; 
&lt;p&gt;In StarRocks, join reordering primarily operates on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;continuous sequences of Inner Joins or Cross Joins&lt;/strong&gt;. As illustrated below, StarRocks groups a sequence of consecutive Inner / Cross Joins into a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Multi Join Node&lt;/strong&gt;. A Multi Join Node is the basic unit for join reordering: if a query plan contains multiple such nodes, StarRocks performs join reordering independently for each one.&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;There are many join reordering algorithms in the industry, often based on different optimization models, including:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;strong&gt;Heuristic-based approaches:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;Rely on predefined rules, such as those used in MemSQL, where join order is determined around dimension tables and fact tables.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Left-Deep Trees:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;Restrict plans to left-deep trees, significantly reducing the search space, though the resulting plan is not always optimal.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Bushy Trees:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;Allow fully bushy join trees, resulting in a much larger search space that includes the optimal plan. Common reordering algorithms under this model include:&lt;/li&gt; 
 &lt;li&gt;Exhaustive search (based on commutativity and associativity)&lt;/li&gt; 
 &lt;li&gt;Greedy algorithms&lt;/li&gt; 
 &lt;li&gt;Simulated annealing&lt;/li&gt; 
 &lt;li&gt;Dynamic programming (e.g., DPsize, DPsub, DPccp)&lt;/li&gt; 
 &lt;li&gt;Genetic algorithms (e.g., Greenplum)&lt;/li&gt; 
 &lt;li&gt;……&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;StarRocks currently implements several join reordering strategies, including Left-Deep, Exhaustive, Greedy, and DPsub. In the following sections, we focus on the implementation details of Exhaustive and Greedy join reordering in StarRocks.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;div&gt; 
 &lt;h3&gt;3.1 Exhaustive&lt;/h3&gt; 
 &lt;p&gt;The exhaustive join reordering algorithm is based on systematically enumerating all possible join orders. In practice, this is achieved through two fundamental rules, which together cover nearly the entire space of join permutations.&lt;/p&gt; 
 &lt;p&gt;&amp;nbsp;&lt;/p&gt; 
 &lt;p&gt;&lt;strong&gt;Rule 1: Join Commutativity&lt;/strong&gt;&lt;/p&gt; 
 &lt;p&gt;A join between two relations can be reordered by swapping its inputs:&lt;code&gt;A JOIN B → B JOIN A&lt;/code&gt;&lt;/p&gt; 
 &lt;p&gt;During this transformation, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;join type must be adjusted accordingly&lt;/strong&gt;. For example, a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LEFT OUTER JOIN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;becomes a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;RIGHT OUTER JOIN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;after swapping the join operands.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;&lt;strong&gt;Rule 2: Join Associativity&lt;/strong&gt;&lt;/p&gt; 
 &lt;p&gt;Join associativity allows the join order among three relations to be rearranged:&lt;code&gt;(A JOIN B) JOIN C → A JOIN (B JOIN C)&lt;/code&gt;&lt;/p&gt; 
 &lt;p&gt;In StarRocks, associativity is handled differently depending on the join type. Specifically, StarRocks distinguishes between:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;Associativity for Inner / Cross Joins&lt;/li&gt; 
  &lt;li&gt;Associativity for Semi Joins&lt;br&gt;&lt;br&gt;&lt;/li&gt; 
 &lt;/ul&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;&lt;strong&gt;3.2 Greedy&lt;/strong&gt;&lt;/h3&gt; 
 &lt;p&gt;For its greedy join reordering strategy, StarRocks primarily draws inspiration from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;multi-sequence greedy algorithms&lt;/strong&gt;, with a small but important enhancement: at each iteration level, instead of keeping only a single best result, StarRocks&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;retains the top 10 candidate plans&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;(which may not be globally optimal). These candidates are then carried forward into the next iteration, ultimately producing&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;10 greedy-optimized plans&lt;/strong&gt;.&lt;/p&gt; 
 &lt;p&gt;Due to the inherent limitations of greedy algorithms, this approach does not guarantee a globally optimal plan. However, by preserving multiple high-quality candidates at each step, it significantly&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;increases the likelihood of finding a near-optimal or optimal solution&lt;/strong&gt;.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;3.3 Cost Model&lt;/h3&gt; 
 &lt;p&gt;StarRocks uses these join reordering algorithms to generate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;N&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;candidate plans. It then evaluates them with a cost model that estimates the cost of each join. The overall cost is computed as:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Join Cost = CPU × (Row(L) + Row(R)) + Memory × Row(R)&lt;/strong&gt;&lt;/p&gt; 
 &lt;p&gt;Here,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;Row(L)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;Row(R)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;are the estimated output row counts of the join’s left and right children, respectively. This formula primarily accounts for the CPU cost of processing both inputs, as well as the memory cost of building the hash table on the right side of a hash join. The figure below shows how StarRocks estimates join output row counts in more detail.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;  
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;Because different join reordering algorithms explore search spaces of varying sizes and have different time complexities, StarRocks benchmarks their execution time and complexity characteristics, as shown below.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;Based on the observed execution costs, StarRocks applies practical limits to how different join reordering algorithms are used:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;For joins involving up to 4 tables&lt;/strong&gt;, StarRocks uses the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;exhaustive&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;algorithm.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;For joins with 4–10 tables&lt;/strong&gt;, StarRocks generates:&lt;/li&gt; 
  &lt;li&gt;1 plan using the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;strategy,&lt;/li&gt; 
  &lt;li&gt;10 plans using the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;greedy&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;algorithm,&lt;/li&gt; 
  &lt;li&gt;1 plan using&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;dynamic programming&lt;/strong&gt;.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;p&gt;On top of these, StarRocks further explores additional plans using&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;join commutativity&lt;/strong&gt;.&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;For joins with more than 10 tables&lt;/strong&gt;, StarRocks relies only on the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;greedy&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;strategies, producing a total of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;11 candidate plans&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;as the basis for reordering.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;When statistics are unavailable&lt;/strong&gt;, cost-based greedy and dynamic programming approaches become unreliable. In this case, StarRocks falls back to using a single&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;plan as the basis for join reordering.&lt;/li&gt; 
 &lt;/ul&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h2&gt;Distributed Join Planning&lt;/h2&gt; 
 &lt;p&gt;After covering the logical optimizations involved in join queries, we now turn to join execution in a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;distributed environment&lt;/strong&gt;, focusing on how StarRocks optimizes distributed join planning as a distributed database.&lt;/p&gt; 
 &lt;h3&gt;4.1 MPP Parallel Execution&lt;/h3&gt; 
 &lt;p&gt;StarRocks is built on an&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;MPP (Massively Parallel Processing)&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;execution framework. The overall architecture is illustrated below. Using a simple join query as an example, the execution of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;A JOIN B&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;in StarRocks typically proceeds as follows:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;Data from tables&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;A&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;B&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is read in parallel from different nodes, based on their respective data distributions.&lt;/li&gt; 
  &lt;li&gt;According to the join predicate, data from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;A&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;B&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;reshuffled&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;so that matching rows are sent to the same set of nodes.&lt;/li&gt; 
  &lt;li&gt;The join is executed locally on each node, and the partial results are produced.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;p&gt;As shown, query execution usually involves multiple sets of machines: the nodes reading table A, the nodes reading table B, and the nodes performing the join are not necessarily the same. As a result, execution inevitably involves network transfers and data exchanges.&lt;/p&gt; 
 &lt;p&gt;These network operations introduce significant overhead. Therefore, a key goal in optimizing distributed join execution in StarRocks is to minimize network cost, while more intelligently partitioning and distributing the query plan to fully leverage the benefits of parallel execution.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.2 Distributed Join Optimization&lt;/h3&gt; 
 &lt;p&gt;We begin by introducing the distributed execution plans that StarRocks can generate. Using a simple join query as an example:&lt;/p&gt; 
 &lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; * &lt;span&gt;From&lt;/span&gt; A &lt;span&gt;Join&lt;/span&gt; B &lt;span&gt;on&lt;/span&gt; A.a = B.b&lt;/span&gt;&lt;/pre&gt;  
 &lt;div&gt; 
  &lt;span&gt;&lt;/span&gt;  
 &lt;/div&gt;  
 &lt;p&gt;In practice, StarRocks can generate five basic types of distributed join plans:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;Shuffle Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;Data from both tables A and B is shuffled based on the join key so that matching rows are sent to the same set of nodes, where the join is then executed.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Broadcast Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;The entire table B is broadcast to all nodes that hold table A, and the join is performed locally on those nodes. Compared to a shuffle join, this avoids shuffling table A, but requires broadcasting all of table B. This strategy is suitable when B is a small table.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Bucket Shuffle Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;An optimization over broadcast join. Instead of broadcasting table B to all nodes, B is shuffled according to A’s data distribution and sent only to the corresponding nodes that hold matching buckets of A. Globally, the shuffled data from B exists only once, significantly reducing network traffic compared to broadcast join. This strategy has an important constraint: the join key must be consistent with A’s distribution key.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Colocate Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;When tables A and B are created within the same colocate group, their data distributions are guaranteed to be identical. If the join key matches the distribution key, StarRocks can execute the join directly on the local nodes holding A and B, without any data shuffle.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Replicate Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;An experimental feature in StarRocks. If every node holding table A also contains a full copy of table B, the join can be executed locally. This approach has very strict requirements — essentially requiring the replication factor of table B to match the total number of nodes in the cluster — making it impractical in most real-world scenarios.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.3 Exploring Distributed Join Plans&lt;/h3&gt; 
 &lt;p&gt;StarRocks derives distributed join plans through&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;distribution property inference&lt;/strong&gt;. Using a shuffle join as an example:&lt;code&gt;SELECT * FROM A JOIN B ON A.a = B.b&lt;/code&gt;, the join operator propagates shuffle requirements top-down to tables A and B. If a scan node cannot satisfy the required distribution, StarRocks inserts an Enforce operator to introduce a shuffle. In the final execution plan, this shuffle is translated into an Exchange node responsible for network data transfer.&lt;/p&gt; 
 &lt;p&gt;Other distributed join strategies are derived in the same way: the join operator requests different distribution properties from its input operators, and the optimizer generates the corresponding distributed execution plans accordingly.&lt;/p&gt; 
 &lt;p&gt;&amp;nbsp;&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.4 Complex Distributed Joins&lt;/h3&gt; 
 &lt;p&gt;In real-world workloads, user queries are far more complex than a simple&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;A JOIN B&lt;/code&gt;. They often involve three or more tables. For such queries, StarRocks generates a richer set of distributed execution plans, all derived from the same fundamental join strategies described earlier.&lt;/p&gt; 
 &lt;p&gt;For example:&lt;/p&gt; 
 &lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; * &lt;span&gt;From&lt;/span&gt; A &lt;span&gt;Join&lt;/span&gt; B &lt;span&gt;on&lt;/span&gt; A.a = B.b &lt;span&gt;Join&lt;/span&gt; C &lt;span&gt;on&lt;/span&gt; A.a = C.c&lt;/span&gt;&lt;/pre&gt; 
 &lt;p&gt;Using combinations of Shuffle Join and Broadcast Join, StarRocks can derive multiple distributed plans, as illustrated below.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;br&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;If Colocate Join and Bucket Shuffle Join are also considered, even more execution plans become possible:&lt;/p&gt;  
 &lt;div&gt; 
  &lt;br&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;Despite their increased complexity, the underlying derivation logic remains the same. Distribution properties are propagated downward through the plan tree, allowing the optimizer to infer different combinations of distributed join strategies.&lt;/p&gt; 
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.5 Global Runtime Filters&lt;/h3&gt; 
 &lt;p&gt;Beyond exploring distributed execution plans, StarRocks further optimizes join performance by leveraging the execution characteristics of join operators to build&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Global Runtime Filters&lt;/strong&gt;.&lt;/p&gt; 
 &lt;p&gt;The execution flow of a Hash Join in StarRocks is as follows:&lt;/p&gt; 
 &lt;ol&gt; 
  &lt;li&gt;Retrieve the complete data set from the right table.&lt;/li&gt; 
  &lt;li&gt;Build a hash table from the right table.&lt;/li&gt; 
  &lt;li&gt;Fetch data from the left table.&lt;/li&gt; 
  &lt;li&gt;Probe the hash table to evaluate join conditions.&lt;/li&gt; 
  &lt;li&gt;Produce the join results.&lt;/li&gt; 
 &lt;/ol&gt; 
 &lt;p&gt;Global Runtime Filters are applied between Step 2 and Step 3. After constructing the hash table on the right side, StarRocks derives runtime filter predicates from the observed data and pushes these filters down to the scan nodes of the left table&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;before&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;left-side data is read. This allows the left table to filter out irrelevant rows early, significantly reducing join input size.&lt;/p&gt; 
 &lt;p&gt;At present, Global Runtime Filters in StarRocks support the following filtering techniques: Min/Max filters, IN predicates, and Bloom filters. The diagram below illustrates how these filters work in practice.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;br&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h2&gt;Summary&lt;/h2&gt; 
 &lt;p&gt;This article has explored StarRocks’ practical experience and ongoing work in join query optimization. All of the techniques discussed are closely aligned with the core optimization principles outlined throughout the article. When optimizing SQL queries in practice, users can also apply the following guidelines together with the features provided by StarRocks to achieve better performance:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;Join operators vary significantly in performance.&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;Prefer high-performance join types whenever possible and avoid expensive ones. Based on typical join output sizes, the rough performance ranking is:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;Semi Join / Anti Join &amp;gt; Inner Join &amp;gt; Outer Join &amp;gt; Full Outer Join &amp;gt; Cross Join&lt;/em&gt;.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;For hash joins, building the hash table on a smaller input is far more efficient&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;than building it on a large table.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;In multi-table joins, execute highly selective joins first&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to substantially reduce the cost of subsequent joins.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Minimize the amount of data participating in joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;through early filtering and pruning.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Reduce network overhead in distributed joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;as much as possible to fully benefit from parallel execution.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
 &lt;h2&gt;Case Studies&lt;/h2&gt; 
 &lt;h3&gt;Demandbase&lt;/h3&gt; 
 &lt;p&gt;By leveraging StarRocks’ On-the-Fly JOIN capabilities, Demandbase successfully replaced its existing ClickHouse clusters, optimizing performance while significantly reducing costs across multiple areas.&lt;/p&gt; 
 &lt;p&gt;Read the case study:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;a href="https://medium.com/starrocks-engineering/demandbase-ditches-denormalization-by-switching-off-clickhouse-44195d795a83"&gt;Demandbase Ditches Denormalization By Switching off ClickHouse&lt;/a&gt;&lt;/p&gt; 
 &lt;h3&gt;Naver&lt;/h3&gt; 
 &lt;p&gt;NAVER modernized its data infrastructure with StarRocks by enabling scalable, real-time analytics over multi-table joins without denormalization. The case study highlights the critical role of efficient, on-the-fly join execution in supporting production-scale analytical workloads.&lt;/p&gt; 
 &lt;p&gt;Read the case study:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;a href="https://celerdata.com/blog/how-join-changed-how-we-approach-data-infra-at-naver"&gt;How JOIN Changed How We Approach Data Infra At NAVER&lt;/a&gt;&lt;/p&gt; 
 &lt;h3&gt;Shopee&lt;/h3&gt; 
 &lt;p&gt;Data Go is a no-code query platform where Shopee business users build queries from multiple tables. Presto struggled with complex join performance and high resource usage. When Shopee switched to StarRocks for multi-table joins, they observed&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;3×–10× performance improvements&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and a ~60% reduction in CPU usage compared with Presto on external Hive data.&lt;/p&gt; 
 &lt;p&gt;Read the case study:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;a href="https://www.starrocks.io/blog/how-shopee-3xed-their-query-performance-with-starrocks"&gt;How Shopee 3xed Their Query Performance With StarRocks&lt;/a&gt;&lt;/p&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;span&gt;&lt;/span&gt; 
&lt;/div&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.starrocks.io/blog/inside-starrocks-why-joins-are-faster-than-youd-expect" title="" class="hs-featured-image-link"&gt; &lt;img src="https://21782839.fs1.hubspotusercontent-na1.net/hubfs/21782839/Group%201142813890.png" alt="Inside StarRocks: Why Joins Are Faster Than You’d Expect" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;div&gt; 
  &lt;div&gt; 
   &lt;div&gt; 
    &lt;div&gt; 
     &lt;blockquote&gt; 
      &lt;div&gt; 
       &lt;div&gt; 
        &lt;div&gt; 
         &lt;em&gt;✍&#x1f3fc; About The Author:&lt;/em&gt; 
        &lt;/div&gt; 
        &lt;div&gt; 
         &lt;div&gt; 
          &lt;em&gt;&lt;span style="background-color: transparent;"&gt;Seaven He, StarRocks Committer, Engineer at Celerdata&lt;/span&gt;&lt;/em&gt; 
         &lt;/div&gt; 
         &lt;span&gt;&lt;/span&gt; 
        &lt;/div&gt; 
       &lt;/div&gt; 
       &lt;span&gt;&lt;/span&gt; 
      &lt;/div&gt; 
     &lt;/blockquote&gt; 
     &lt;div&gt; 
      &lt;div&gt;
        &amp;nbsp; 
      &lt;/div&gt; 
     &lt;/div&gt; 
    &lt;/div&gt; 
   &lt;/div&gt; 
   &lt;div&gt; 
    &lt;span style="background-color: transparent;"&gt;Joins are the hardest part of OLAP. Many systems can’t run them efficiently at scale, so teams denormalize into wide tables instead, 10× their storage, dealing with complex stream processing pipelines, and painfully slow and expensive schema evolution that triggers large backfills.&lt;/span&gt; 
   &lt;/div&gt; 
  &lt;/div&gt; 
 &lt;/div&gt; 
&lt;/div&gt; 
&lt;p&gt;StarRocks takes the opposite approach: keep data normalized and make joins fast enough to run on the fly. The challenge is the plan. In a distributed system, the join search space is huge, and a good plan can be orders of magnitude faster.&lt;/p&gt; 
&lt;p&gt;This deep dive explains how StarRocks’ cost-based optimizer makes that possible, in four parts: join fundamentals and optimization challenges, logical join optimizations, join reordering, and distributed join planning. Finally, we examine real-world case studies from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;NAVER, Demandbase, and Shopee&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to illustrate how efficient join execution delivers tangible business value.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;h2&gt;Join Fundamentals and Optimization Challenges&lt;/h2&gt; 
&lt;h3&gt;&lt;strong&gt;1.1 Join Types&lt;/strong&gt;&lt;/h3&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;The diagram above illustrates several common join types:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;strong&gt;Cross Join&lt;/strong&gt;: Produces a Cartesian product between the left and right tables.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Full / Left / Right Outer Join&lt;/strong&gt;: For rows that do not find a match, outer joins return results with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;values filled in according to the join semantics—on both tables (full), the left table (left), or the right table (right).&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Anti Join&lt;/strong&gt;: Returns rows that do&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;not&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;have a matching counterpart in the join relationship. Anti-joins typically appear in query plans for&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NOT IN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NOT EXISTS&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;subqueries.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Semi Join&lt;/strong&gt;: The opposite of an anti-join, it returns only rows that&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;do&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;have a match in the join relationship, without producing duplicate result rows from the matching side.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Inner Join&lt;/strong&gt;: Returns the intersection of the left and right tables. Based on the join condition, it may generate one-to-many result rows.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;1.2 Challenges in Join Optimization&lt;/h3&gt; 
&lt;p&gt;Join performance optimization generally falls into two areas:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;improving the efficiency of join operators on a single node, and&lt;/li&gt; 
 &lt;li&gt;designing a reasonable join plan that minimizes input size and execution cost.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;This article focuses on the second aspect. To set the stage, we begin by examining the key challenges in join optimization.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 1: Multiple Join Implementation Strategies&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;As shown above, different join algorithms perform very differently depending on the scenario. For example, Sort-Merge Join can be significantly more efficient than Hash Join when operating on already sorted data. However, in distributed databases where data is typically hash-partitioned, Hash Join often outperforms Sort-Merge Join by a wide margin. As a result,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;the database must choose the most appropriate join strategy based on the specific workload and data characteristics.&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 2: Join Order Selection in Multi-Table Joins&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;In multi-table join scenarios, executing highly selective joins first can significantly improve overall query performance. However, determining the optimal join order is far from trivial.&lt;/p&gt; 
&lt;p&gt;As illustrated above, under a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep join tree&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;model, the number of possible join orders for&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;N&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;tables is on the order of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;2^n-1&lt;/code&gt;. Under a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;bushy join tree&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;model, the number of possible combinations grows even more dramatically, reaching&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;2^(n-1) * C(n-1)&lt;/code&gt;. For a database optimizer, the time and cost required to search for the optimal join order therefore increases&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;exponentially&lt;/strong&gt;, making join ordering one of the most challenging problems in query optimization.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 3: Difficulty in Estimating Join Effectiveness&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;Before query execution, it is extremely difficult for the database to accurately predict the real execution behavior of a join. A common assumption is that joining a small table with a large table is more selective than joining two large tables, but this is not always true.&lt;/p&gt; 
&lt;p&gt;In practice, one-to-many relationships are common, and in more complex queries, joins are often combined with filters, aggregations, and other operators. After data flows through multiple transformations, the optimizer’s ability to accurately estimate join input sizes and selectivity degrades significantly.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Challenge 4: A Single-Node Optimal Plan Is Not Necessarily Optimal in Distributed Systems&lt;/strong&gt;&lt;/p&gt;  
&lt;div&gt; 
 &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;In distributed systems, data often needs to be&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;reshuffled&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;broadcast&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;across nodes so that the required records can participate in join computation. Distributed joins are no exception.&lt;/p&gt; 
&lt;p&gt;This introduces a key complication: an execution plan that is optimal in a single-node database may perform poorly in a distributed environment because it ignores data distribution and network transfer costs.&lt;/p&gt; 
&lt;p&gt;Therefore, when planning join execution strategies in distributed databases, the optimizer must explicitly account for data placement and communication overhead in addition to local execution efficiency.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;1.3 SQL Optimization Workflow&lt;/h3&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;In StarRocks, SQL optimization is primarily handled by the query optimizer and is mainly concentrated in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Rewrite&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Optimize&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;phases.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;1.4 Principles of Join Optimization&lt;/h3&gt; 
&lt;p&gt;At present, StarRocks primarily uses&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Hash Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;as its join algorithm. By default, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;right-hand table&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is used to build the hash table. Based on this design choice, we summarize five key optimization principles:&lt;/p&gt; 
&lt;ol&gt; 
 &lt;li&gt;Different join types have very different performance characteristics. Whenever possible, prefer higher-performance join types and avoid expensive ones. Based on the typical size of join outputs, the rough performance ranking is:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Semi Join / Anti Join &amp;gt; Inner Join &amp;gt; Outer Join &amp;gt; Full Outer Join &amp;gt; Cross Join&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;When using Hash Join, building the hash table on a smaller input is significantly more efficient than building it on a large table.&lt;/li&gt; 
 &lt;li&gt;In multi-table joins, prioritize joins with high selectivity.&lt;/li&gt; 
 &lt;li&gt;Minimize the amount of data participating in joins whenever possible.&lt;/li&gt; 
 &lt;li&gt;Minimize network overhead introduced by distributed joins.&lt;/li&gt; 
&lt;/ol&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;Join Logical Optimization&lt;/h2&gt; 
&lt;p&gt;This section introduces a set of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;heuristic rules&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;used by StarRocks to optimize joins at the logical level.&lt;/p&gt; 
&lt;h3&gt;2.1 Type Transformations&lt;/h3&gt; 
&lt;p&gt;The first group of optimizations directly follows the first join optimization principle discussed earlier:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;transform low-efficiency join types into more efficient ones whenever the semantics allow it&lt;/strong&gt;.&lt;/p&gt; 
&lt;p&gt;StarRocks currently applies three major transformation rules.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Rule 1: Converting a Cross Join into an Inner Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;A Cross Join can be rewritten as an Inner Join when it satisfies the following condition:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There exists at least one predicate that expresses a join relationship between the two tables.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1, t2 &lt;span&gt;WHERE&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;-- WHERE t1.v1 = t2.v1 is a join predicate&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;INNER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Rule 2: Converting an Outer Join into an Inner Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;A Left / Right Outer Join can be rewritten as an Inner Join when the following conditions are met:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There exists a predicate referencing the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;nullable side&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;of the outer join (Right table for a Left Outer Join, or Left table for a Right Outer Join).&lt;/li&gt; 
 &lt;li&gt;The predicate is a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;strict (null-rejecting) predicate&lt;/strong&gt;.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;-- t2.v1 &amp;gt; 0 is a strict predicate on t2&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;INNER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;⚠️&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Important note:&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;In an outer join,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause predicates participate in&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;null extension&lt;/strong&gt;, not filtering. Therefore, this rule does&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;not&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;apply to join predicates inside the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;AND&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;1&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;This query is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;not semantically equivalent&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;INNER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;AND&lt;/span&gt; t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;1&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;This introduces the concept of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;strict (null-rejecting) predicates&lt;/strong&gt;. In StarRocks, a predicate that filters out&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;values is considered a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;strict predicate&lt;/em&gt;, for example&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;a &amp;gt; 0&lt;/code&gt;. Predicates that do not eliminate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;values are classified as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;non-strict predicates&lt;/em&gt;, such as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;a IS NULL&lt;/code&gt;. Most predicates fall into the strict category; non-strict predicates are primarily those involving&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;IS NULL&lt;/code&gt;,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;IF&lt;/code&gt;,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;CASE WHEN&lt;/code&gt;, or certain function-based expressions.&lt;/p&gt; 
&lt;p&gt;To determine whether a predicate is strict, StarRocks uses a simple yet effective approach: all referenced columns are replaced with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;, and the expression is then simplified. If the result evaluates to&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;TRUE&lt;/code&gt;, it means the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause does not filter out rows with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;inputs, and the predicate is therefore non-strict. Conversely, if the result evaluates to&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;FALSE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;NULL&lt;/code&gt;, the predicate is considered strict.&lt;/p&gt;  
&lt;div&gt; 
 &lt;br&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Rule 3: Converting a Full Outer Join into a Left / Right Outer Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;A Full Outer Join can be rewritten as a Left Outer Join or Right Outer Join when the following condition is satisfied:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There exists a strict predicate that can be bound exclusively to the left or right table.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;FULL&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t1.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After transformation&lt;/span&gt;&lt;br&gt;&lt;span&gt;-- t1.v1 &amp;gt; 0 is a strict predicate on the left table&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;WHERE&lt;/span&gt; t1.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;0&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;2.2 Predicate Pushdown&lt;/h3&gt; 
&lt;p&gt;Predicate pushdown is one of the most important and commonly used join optimization techniques. Its primary purpose is to&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;filter join inputs as early as possible&lt;/strong&gt;, thereby reducing the amount of data involved in the join and improving overall performance.&lt;/p&gt; 
&lt;p&gt;For predicates in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause, predicate pushdown can be applied—and may enable join type transformations—when the following conditions are satisfied:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;The join can be of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;any type&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;The&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicate can be&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;bound to one of the join inputs&lt;/strong&gt;.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;  &lt;br&gt;&lt;span&gt;From&lt;/span&gt; t1 &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t2 &lt;span&gt;On&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;br&gt;        &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t3 &lt;span&gt;On&lt;/span&gt; t2.v2 &lt;span&gt;=&lt;/span&gt; t3.v2 &lt;br&gt;&lt;span&gt;Where&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;1&lt;/span&gt; &lt;span&gt;And&lt;/span&gt; t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;span&gt;And&lt;/span&gt; t3.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;The predicate pushdown process proceeds as follows.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Step 1&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;:&lt;/p&gt; 
&lt;p&gt;Push down&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = 1 AND t2.v1 = 2)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t3.v2 = 3)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;separately. Since the join type transformation rules are satisfied,&lt;code&gt;(t1 LEFT OUTER JOIN t2) LEFT OUTER JOIN t3&lt;/code&gt;can be rewritten as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1 LEFT OUTER JOIN t2) INNER JOIN t3&lt;/code&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;  
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Step 2:&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;Continue pushing down&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = 1)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2)&lt;/code&gt;. At this point,&lt;code&gt;t1 LEFT OUTER JOIN t2&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be further transformed into&lt;code&gt;t1 INNER JOIN t2&lt;/code&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;It is important to note that&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;predicate pushdown rules for join predicates in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;ON&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause differ from those for the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;WHERE&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause&lt;/strong&gt;. We distinguish between two cases:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Inner Joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;other join types&lt;/strong&gt;.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Case 1: Inner Join&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Inner Joins&lt;/strong&gt;, pushing down join predicates in the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause follows the same rules as&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause predicate pushdown. This has already been discussed above and will not be repeated here.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Case 2: Outer / Semi / Anti Joins&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;For Outer, Semi, and Anti Joins, predicate pushdown on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause join predicates must satisfy the following constraints, and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;no join type transformation is allowed during the pushdown process&lt;/strong&gt;:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;The join must be a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left or Right Outer / Semi / Anti Join&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;The join predicate must be bindable&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;only to the nullable side&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;(the right input for a Left Join, or the left input for a Right Join).&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Consider the following example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;  &lt;br&gt;&lt;span&gt;From&lt;/span&gt; t1 &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t2 &lt;span&gt;On&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1 &lt;span&gt;And&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;1&lt;/span&gt; &lt;span&gt;And&lt;/span&gt; t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;br&gt;        &lt;span&gt;Left&lt;/span&gt; &lt;span&gt;Outer&lt;/span&gt; &lt;span&gt;Join&lt;/span&gt; t3 &lt;span&gt;On&lt;/span&gt; t2.v2 &lt;span&gt;=&lt;/span&gt; t3.v2 &lt;span&gt;And&lt;/span&gt; t3.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;The predicate pushdown proceeds as follows.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Step 1:&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;Push down the join predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t3.v2 = 3)&lt;/code&gt;, which can be bound to the right input of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t1 LEFT JOIN t2 LEFT JOIN t3&lt;/code&gt;. At this stage, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LEFT OUTER JOIN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;cannot&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;be converted into an&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;INNER JOIN&lt;/code&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;&lt;strong&gt;Step 2:&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;Push down the join predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2)&lt;/code&gt;, which can be bound to the right input of&lt;code&gt;t1 LEFT JOIN t2&lt;/code&gt;.&lt;/p&gt; 
&lt;p&gt;However, the predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = 1)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is bound to the left input. Pushing it down would filter rows from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t1&lt;/code&gt;, violating the semantics of a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LEFT OUTER JOIN&lt;/code&gt;. Therefore, this predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;cannot be pushed down&lt;/strong&gt;.&lt;/p&gt;  
&lt;div&gt; 
 &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;h3&gt;2.3 Predicate Extraction&lt;/h3&gt; 
&lt;p&gt;In the predicate pushdown rules discussed earlier, only predicates with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;conjunctive semantics&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down. For example, in&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t1.v1 = 1 AND t2.v1 = 2 AND t3.v2 = 3&lt;/code&gt;, each sub-predicate is connected by conjunction, making pushdown straightforward. However, predicates with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;disjunctive semantics&lt;/strong&gt;, such as&lt;code&gt;t1.v1 = 1 OR t2.v1 = 2 OR t3.v2 = 3&lt;/code&gt;, cannot be pushed down directly.&lt;/p&gt; 
&lt;p&gt;In real-world queries, disjunctive predicates are quite common. To address this, StarRocks introduces an optimization called&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;predicate extraction (column value derivation)&lt;/strong&gt;. This technique derives conjunctive predicates from disjunctive ones by performing a series of union and intersection operations on column value ranges. The derived conjunctive predicates can then be pushed down to reduce join input size.&lt;/p&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before predicate extraction&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;)&lt;span&gt;OR&lt;/span&gt; (t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;Using column value derivation on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2 AND t1.v2 = 3) OR (t2.v1 &amp;gt; 5 AND t1.v2 = 4)&lt;/code&gt;, the optimizer can extract the following predicates:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;code&gt;t2.v1 &amp;gt;= 2&lt;/code&gt;&lt;/li&gt; 
 &lt;li&gt;&lt;code&gt;t1.v2 IN (3, 4)&lt;/code&gt;&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;The query can then be rewritten as:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; *&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 = t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 = &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;3&lt;/span&gt;)&lt;span&gt;OR&lt;/span&gt; (t2.v1 &amp;gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;4&lt;/span&gt;)&lt;span&gt;AND&lt;/span&gt; t2.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;&lt;br&gt;&lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;IN&lt;/span&gt; (&lt;span&gt;3&lt;/span&gt;, &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;It is important to note that the extracted predicates may form a superset of the original predicate ranges. As a result, they cannot safely replace the original predicates and must instead be applied in addition to them.&lt;/p&gt; 
&lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
&lt;h3&gt;2.4 Equivalence Derivation&lt;/h3&gt; 
&lt;p&gt;In addition to predicate extraction, another important predicate-level optimization is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;equivalence derivation&lt;/strong&gt;. This technique leverages&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;join equality conditions&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to infer value constraints on one side of the join from predicates applied to the other side.&lt;/p&gt; 
&lt;p&gt;Specifically, based on the join condition, value ranges on columns from the left table can be used to derive corresponding value ranges on columns from the right table, and vice versa.&lt;/p&gt; 
&lt;p&gt;For example:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Original SQL&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;3&lt;/span&gt;)&lt;br&gt;   &lt;span&gt;OR&lt;/span&gt; (t2.v1 &lt;span&gt;&amp;gt;&lt;/span&gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;=&lt;/span&gt; &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;Using predicate extraction on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t2.v1 = 2 AND t1.v2 = 3) OR (t2.v1 &amp;gt; 5 AND t1.v2 = 4)&lt;/code&gt;, the optimizer can derive the following predicates:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;code&gt;t2.v1 &amp;gt;= 2&lt;/code&gt;&lt;/li&gt; 
 &lt;li&gt;&lt;code&gt;t1.v2 IN (3, 4)&lt;/code&gt;&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;Resulting in:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; *&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 = t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 = &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;3&lt;/span&gt;)&lt;br&gt;   &lt;span&gt;OR&lt;/span&gt; (t2.v1 &amp;gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;4&lt;/span&gt;)&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t2.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;IN&lt;/span&gt; (&lt;span&gt;3&lt;/span&gt;, &lt;span&gt;4&lt;/span&gt;);&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;Next, using the join predicate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;(t1.v1 = t2.v1)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;together with&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;t2.v1 &amp;gt;= 2&lt;/code&gt;, equivalence derivation can infer an additional predicate:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;code&gt;t1.v1 &amp;gt;= 2&lt;/code&gt;&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;The query can therefore be further rewritten as:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;SELECT&lt;/span&gt; *&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 = t2.v1&lt;br&gt;&lt;span&gt;WHERE&lt;/span&gt; (t2.v1 = &lt;span&gt;2&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;3&lt;/span&gt;)&lt;br&gt;   &lt;span&gt;OR&lt;/span&gt; (t2.v1 &amp;gt; &lt;span&gt;5&lt;/span&gt; &lt;span&gt;AND&lt;/span&gt; t1.v2 = &lt;span&gt;4&lt;/span&gt;)&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t2.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t1.v2 &lt;span&gt;IN&lt;/span&gt; (&lt;span&gt;3&lt;/span&gt;, &lt;span&gt;4&lt;/span&gt;)&lt;br&gt;  &lt;span&gt;AND&lt;/span&gt; t1.v1 &amp;gt;= &lt;span&gt;2&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;p&gt;&lt;strong&gt;Applicability and Constraints&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;The scope of equivalence derivation is more limited than predicate extraction. Predicate extraction can be applied to arbitrary predicates, whereas equivalence derivation, like predicate pushdown, has different constraints depending on the join type. As before, we distinguish between&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicates and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;ON&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause join predicates.&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;WHERE&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicates:&lt;/strong&gt;&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;There are almost no restrictions. Predicates can be derived from the left table to the right table and vice versa.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;&lt;strong&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;&lt;code&gt;&lt;strong&gt;ON&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clause join predicates:&lt;/strong&gt;&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;For&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Inner Joins&lt;/strong&gt;, the rules are the same as for&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;WHERE&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;predicates—no additional constraints apply.&lt;/li&gt; 
 &lt;li&gt;For join types other than Inner Join, only&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Semi Joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Outer Joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;are supported, and derivation is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;one-directional only&lt;/strong&gt;, opposite to the join direction:&lt;/li&gt; 
 &lt;li&gt;For a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left Outer Join&lt;/strong&gt;, predicates can be derived from the left table to the right table.&lt;/li&gt; 
 &lt;li&gt;For a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Right Outer Join&lt;/strong&gt;, predicates can be derived from the right table to the left table.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Why Is Equivalence Derivation One-Directional for Outer / Semi Joins?&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;The reason is straightforward. Consider a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left Outer Join&lt;/strong&gt;. As discussed in predicate pushdown rules, only predicates on the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;right table&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down; predicates on the left table cannot, as doing so would violate the semantics of a left outer join.&lt;/p&gt; 
&lt;p&gt;For the same reason, predicates derived&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;from the right table&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and applied to the left table must also respect this constraint. In practice, such derived predicates on the preserved side do not help filter data early and instead introduce additional evaluation overhead. Therefore, equivalence derivation for Outer and Semi Joins is intentionally restricted to a single direction.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;p&gt;&lt;strong&gt;Implementation Details&lt;/strong&gt;&lt;/p&gt; 
&lt;p&gt;StarRocks implements equivalence derivation by maintaining two internal maps:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;One map tracks&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;column-to-column equivalence relationships&lt;/strong&gt;.&lt;/li&gt; 
 &lt;li&gt;The other map tracks&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;column-to-value or column-to-expression equivalences&lt;/strong&gt;.&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;By performing lookups and inference across these two maps, the optimizer derives additional equivalent predicates. The overall mechanism is illustrated below:&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;h3&gt;2.5 Limit Pushdown&lt;/h3&gt; 
&lt;p&gt;In addition to predicates,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;&lt;strong&gt;LIMIT&lt;/strong&gt;&lt;/code&gt;&lt;strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;clauses can also be pushed down through joins&lt;/strong&gt;. When a query involves an&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Outer Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;or a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Cross Join&lt;/strong&gt;, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LIMIT&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down to child operators whose output row count is guaranteed to be stable.&lt;/p&gt; 
&lt;p&gt;For example, in a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Left Outer Join&lt;/strong&gt;, the output row count is at least the same as that of the left input. Therefore, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LIMIT&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down to the left table (and symmetrically for a Right Outer Join).&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t1.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; (&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 LIMIT &lt;span&gt;100&lt;/span&gt;) t&lt;br&gt;&lt;span&gt;LEFT&lt;/span&gt; &lt;span&gt;OUTER&lt;/span&gt; &lt;span&gt;JOIN&lt;/span&gt; t2 &lt;span&gt;ON&lt;/span&gt; t.v1 &lt;span&gt;=&lt;/span&gt; t2.v1&lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;h3&gt;Special Cases: Cross Join and Full Outer Join&lt;/h3&gt; 
&lt;p&gt;A Cross Join produces a Cartesian product, with output cardinality equal to&lt;code&gt;rows(left) × rows(right)&lt;/code&gt;. A Full Outer Join produces at least&lt;code&gt;rows(left) + rows(right)&lt;/code&gt;.&lt;/p&gt; 
&lt;p&gt;For these join types, a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LIMIT&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;can be pushed down to both inputs independently:&lt;/p&gt; 
&lt;pre&gt;&lt;span&gt;&lt;span&gt;-- Before pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; t1 &lt;span&gt;JOIN&lt;/span&gt; t2&lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;br&gt;&lt;br&gt;&lt;span&gt;-- After pushdown&lt;/span&gt;&lt;br&gt;&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt;&lt;br&gt;&lt;span&gt;FROM&lt;/span&gt; (&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t1 LIMIT &lt;span&gt;100&lt;/span&gt;) x1&lt;br&gt;&lt;span&gt;JOIN&lt;/span&gt; (&lt;span&gt;SELECT&lt;/span&gt; &lt;span&gt;*&lt;/span&gt; &lt;span&gt;FROM&lt;/span&gt; t2 LIMIT &lt;span&gt;100&lt;/span&gt;) &lt;br&gt;LIMIT &lt;span&gt;100&lt;/span&gt;;&lt;/span&gt;&lt;/pre&gt; 
&lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
&lt;h2&gt;Join Reordering&lt;/h2&gt; 
&lt;p&gt;Join reordering is used to determine the execution order of multi-table joins. The optimizer aims to execute&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;high-selectivity joins as early as possible&lt;/strong&gt;, thereby reducing the size of intermediate results and improving overall query performance.&lt;/p&gt; 
&lt;p&gt;In StarRocks, join reordering primarily operates on&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;continuous sequences of Inner Joins or Cross Joins&lt;/strong&gt;. As illustrated below, StarRocks groups a sequence of consecutive Inner / Cross Joins into a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Multi Join Node&lt;/strong&gt;. A Multi Join Node is the basic unit for join reordering: if a query plan contains multiple such nodes, StarRocks performs join reordering independently for each one.&lt;/p&gt;  
&lt;div&gt; 
 &lt;div&gt;     
 &lt;/div&gt; 
&lt;/div&gt;  
&lt;p&gt;There are many join reordering algorithms in the industry, often based on different optimization models, including:&lt;/p&gt; 
&lt;ul&gt; 
 &lt;li&gt;&lt;strong&gt;Heuristic-based approaches:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;Rely on predefined rules, such as those used in MemSQL, where join order is determined around dimension tables and fact tables.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Left-Deep Trees:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;Restrict plans to left-deep trees, significantly reducing the search space, though the resulting plan is not always optimal.&lt;/li&gt; 
 &lt;li&gt;&lt;strong&gt;Bushy Trees:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;/strong&gt;Allow fully bushy join trees, resulting in a much larger search space that includes the optimal plan. Common reordering algorithms under this model include:&lt;/li&gt; 
 &lt;li&gt;Exhaustive search (based on commutativity and associativity)&lt;/li&gt; 
 &lt;li&gt;Greedy algorithms&lt;/li&gt; 
 &lt;li&gt;Simulated annealing&lt;/li&gt; 
 &lt;li&gt;Dynamic programming (e.g., DPsize, DPsub, DPccp)&lt;/li&gt; 
 &lt;li&gt;Genetic algorithms (e.g., Greenplum)&lt;/li&gt; 
 &lt;li&gt;……&lt;/li&gt; 
&lt;/ul&gt; 
&lt;p&gt;StarRocks currently implements several join reordering strategies, including Left-Deep, Exhaustive, Greedy, and DPsub. In the following sections, we focus on the implementation details of Exhaustive and Greedy join reordering in StarRocks.&lt;/p&gt; 
&lt;p&gt;&amp;nbsp;&lt;/p&gt; 
&lt;div&gt; 
 &lt;h3&gt;3.1 Exhaustive&lt;/h3&gt; 
 &lt;p&gt;The exhaustive join reordering algorithm is based on systematically enumerating all possible join orders. In practice, this is achieved through two fundamental rules, which together cover nearly the entire space of join permutations.&lt;/p&gt; 
 &lt;p&gt;&amp;nbsp;&lt;/p&gt; 
 &lt;p&gt;&lt;strong&gt;Rule 1: Join Commutativity&lt;/strong&gt;&lt;/p&gt; 
 &lt;p&gt;A join between two relations can be reordered by swapping its inputs:&lt;code&gt;A JOIN B → B JOIN A&lt;/code&gt;&lt;/p&gt; 
 &lt;p&gt;During this transformation, the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;join type must be adjusted accordingly&lt;/strong&gt;. For example, a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;LEFT OUTER JOIN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;becomes a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;RIGHT OUTER JOIN&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;after swapping the join operands.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;&lt;strong&gt;Rule 2: Join Associativity&lt;/strong&gt;&lt;/p&gt; 
 &lt;p&gt;Join associativity allows the join order among three relations to be rearranged:&lt;code&gt;(A JOIN B) JOIN C → A JOIN (B JOIN C)&lt;/code&gt;&lt;/p&gt; 
 &lt;p&gt;In StarRocks, associativity is handled differently depending on the join type. Specifically, StarRocks distinguishes between:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;Associativity for Inner / Cross Joins&lt;/li&gt; 
  &lt;li&gt;Associativity for Semi Joins&lt;br&gt;&lt;br&gt;&lt;/li&gt; 
 &lt;/ul&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;&lt;strong&gt;3.2 Greedy&lt;/strong&gt;&lt;/h3&gt; 
 &lt;p&gt;For its greedy join reordering strategy, StarRocks primarily draws inspiration from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;multi-sequence greedy algorithms&lt;/strong&gt;, with a small but important enhancement: at each iteration level, instead of keeping only a single best result, StarRocks&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;retains the top 10 candidate plans&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;(which may not be globally optimal). These candidates are then carried forward into the next iteration, ultimately producing&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;10 greedy-optimized plans&lt;/strong&gt;.&lt;/p&gt; 
 &lt;p&gt;Due to the inherent limitations of greedy algorithms, this approach does not guarantee a globally optimal plan. However, by preserving multiple high-quality candidates at each step, it significantly&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;increases the likelihood of finding a near-optimal or optimal solution&lt;/strong&gt;.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;span&gt;Press enter or click to view image in full size&lt;/span&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;3.3 Cost Model&lt;/h3&gt; 
 &lt;p&gt;StarRocks uses these join reordering algorithms to generate&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;N&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;candidate plans. It then evaluates them with a cost model that estimates the cost of each join. The overall cost is computed as:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Join Cost = CPU × (Row(L) + Row(R)) + Memory × Row(R)&lt;/strong&gt;&lt;/p&gt; 
 &lt;p&gt;Here,&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;Row(L)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;Row(R)&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;are the estimated output row counts of the join’s left and right children, respectively. This formula primarily accounts for the CPU cost of processing both inputs, as well as the memory cost of building the hash table on the right side of a hash join. The figure below shows how StarRocks estimates join output row counts in more detail.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;  
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;Because different join reordering algorithms explore search spaces of varying sizes and have different time complexities, StarRocks benchmarks their execution time and complexity characteristics, as shown below.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;Based on the observed execution costs, StarRocks applies practical limits to how different join reordering algorithms are used:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;For joins involving up to 4 tables&lt;/strong&gt;, StarRocks uses the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;exhaustive&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;algorithm.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;For joins with 4–10 tables&lt;/strong&gt;, StarRocks generates:&lt;/li&gt; 
  &lt;li&gt;1 plan using the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;strategy,&lt;/li&gt; 
  &lt;li&gt;10 plans using the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;greedy&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;algorithm,&lt;/li&gt; 
  &lt;li&gt;1 plan using&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;dynamic programming&lt;/strong&gt;.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;p&gt;On top of these, StarRocks further explores additional plans using&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;join commutativity&lt;/strong&gt;.&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;For joins with more than 10 tables&lt;/strong&gt;, StarRocks relies only on the&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;greedy&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;strategies, producing a total of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;11 candidate plans&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;as the basis for reordering.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;When statistics are unavailable&lt;/strong&gt;, cost-based greedy and dynamic programming approaches become unreliable. In this case, StarRocks falls back to using a single&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;left-deep&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;plan as the basis for join reordering.&lt;/li&gt; 
 &lt;/ul&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h2&gt;Distributed Join Planning&lt;/h2&gt; 
 &lt;p&gt;After covering the logical optimizations involved in join queries, we now turn to join execution in a&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;distributed environment&lt;/strong&gt;, focusing on how StarRocks optimizes distributed join planning as a distributed database.&lt;/p&gt; 
 &lt;h3&gt;4.1 MPP Parallel Execution&lt;/h3&gt; 
 &lt;p&gt;StarRocks is built on an&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;MPP (Massively Parallel Processing)&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;execution framework. The overall architecture is illustrated below. Using a simple join query as an example, the execution of&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;A JOIN B&lt;/code&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;in StarRocks typically proceeds as follows:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;Data from tables&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;A&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;B&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is read in parallel from different nodes, based on their respective data distributions.&lt;/li&gt; 
  &lt;li&gt;According to the join predicate, data from&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;A&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;B&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;is&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;reshuffled&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;so that matching rows are sent to the same set of nodes.&lt;/li&gt; 
  &lt;li&gt;The join is executed locally on each node, and the partial results are produced.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;p&gt;As shown, query execution usually involves multiple sets of machines: the nodes reading table A, the nodes reading table B, and the nodes performing the join are not necessarily the same. As a result, execution inevitably involves network transfers and data exchanges.&lt;/p&gt; 
 &lt;p&gt;These network operations introduce significant overhead. Therefore, a key goal in optimizing distributed join execution in StarRocks is to minimize network cost, while more intelligently partitioning and distributing the query plan to fully leverage the benefits of parallel execution.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.2 Distributed Join Optimization&lt;/h3&gt; 
 &lt;p&gt;We begin by introducing the distributed execution plans that StarRocks can generate. Using a simple join query as an example:&lt;/p&gt; 
 &lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; * &lt;span&gt;From&lt;/span&gt; A &lt;span&gt;Join&lt;/span&gt; B &lt;span&gt;on&lt;/span&gt; A.a = B.b&lt;/span&gt;&lt;/pre&gt;  
 &lt;div&gt; 
  &lt;span&gt;&lt;/span&gt;  
 &lt;/div&gt;  
 &lt;p&gt;In practice, StarRocks can generate five basic types of distributed join plans:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;Shuffle Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;Data from both tables A and B is shuffled based on the join key so that matching rows are sent to the same set of nodes, where the join is then executed.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Broadcast Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;The entire table B is broadcast to all nodes that hold table A, and the join is performed locally on those nodes. Compared to a shuffle join, this avoids shuffling table A, but requires broadcasting all of table B. This strategy is suitable when B is a small table.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Bucket Shuffle Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;An optimization over broadcast join. Instead of broadcasting table B to all nodes, B is shuffled according to A’s data distribution and sent only to the corresponding nodes that hold matching buckets of A. Globally, the shuffled data from B exists only once, significantly reducing network traffic compared to broadcast join. This strategy has an important constraint: the join key must be consistent with A’s distribution key.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Colocate Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;When tables A and B are created within the same colocate group, their data distributions are guaranteed to be identical. If the join key matches the distribution key, StarRocks can execute the join directly on the local nodes holding A and B, without any data shuffle.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Replicate Join&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;An experimental feature in StarRocks. If every node holding table A also contains a full copy of table B, the join can be executed locally. This approach has very strict requirements — essentially requiring the replication factor of table B to match the total number of nodes in the cluster — making it impractical in most real-world scenarios.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.3 Exploring Distributed Join Plans&lt;/h3&gt; 
 &lt;p&gt;StarRocks derives distributed join plans through&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;distribution property inference&lt;/strong&gt;. Using a shuffle join as an example:&lt;code&gt;SELECT * FROM A JOIN B ON A.a = B.b&lt;/code&gt;, the join operator propagates shuffle requirements top-down to tables A and B. If a scan node cannot satisfy the required distribution, StarRocks inserts an Enforce operator to introduce a shuffle. In the final execution plan, this shuffle is translated into an Exchange node responsible for network data transfer.&lt;/p&gt; 
 &lt;p&gt;Other distributed join strategies are derived in the same way: the join operator requests different distribution properties from its input operators, and the optimizer generates the corresponding distributed execution plans accordingly.&lt;/p&gt; 
 &lt;p&gt;&amp;nbsp;&lt;/p&gt;  
 &lt;div&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.4 Complex Distributed Joins&lt;/h3&gt; 
 &lt;p&gt;In real-world workloads, user queries are far more complex than a simple&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;code&gt;A JOIN B&lt;/code&gt;. They often involve three or more tables. For such queries, StarRocks generates a richer set of distributed execution plans, all derived from the same fundamental join strategies described earlier.&lt;/p&gt; 
 &lt;p&gt;For example:&lt;/p&gt; 
 &lt;pre&gt;&lt;span&gt;&lt;span&gt;Select&lt;/span&gt; * &lt;span&gt;From&lt;/span&gt; A &lt;span&gt;Join&lt;/span&gt; B &lt;span&gt;on&lt;/span&gt; A.a = B.b &lt;span&gt;Join&lt;/span&gt; C &lt;span&gt;on&lt;/span&gt; A.a = C.c&lt;/span&gt;&lt;/pre&gt; 
 &lt;p&gt;Using combinations of Shuffle Join and Broadcast Join, StarRocks can derive multiple distributed plans, as illustrated below.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;br&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;If Colocate Join and Bucket Shuffle Join are also considered, even more execution plans become possible:&lt;/p&gt;  
 &lt;div&gt; 
  &lt;br&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;p&gt;Despite their increased complexity, the underlying derivation logic remains the same. Distribution properties are propagated downward through the plan tree, allowing the optimizer to infer different combinations of distributed join strategies.&lt;/p&gt; 
 &lt;h3&gt;&amp;nbsp;&lt;/h3&gt; 
 &lt;h3&gt;4.5 Global Runtime Filters&lt;/h3&gt; 
 &lt;p&gt;Beyond exploring distributed execution plans, StarRocks further optimizes join performance by leveraging the execution characteristics of join operators to build&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;strong&gt;Global Runtime Filters&lt;/strong&gt;.&lt;/p&gt; 
 &lt;p&gt;The execution flow of a Hash Join in StarRocks is as follows:&lt;/p&gt; 
 &lt;ol&gt; 
  &lt;li&gt;Retrieve the complete data set from the right table.&lt;/li&gt; 
  &lt;li&gt;Build a hash table from the right table.&lt;/li&gt; 
  &lt;li&gt;Fetch data from the left table.&lt;/li&gt; 
  &lt;li&gt;Probe the hash table to evaluate join conditions.&lt;/li&gt; 
  &lt;li&gt;Produce the join results.&lt;/li&gt; 
 &lt;/ol&gt; 
 &lt;p&gt;Global Runtime Filters are applied between Step 2 and Step 3. After constructing the hash table on the right side, StarRocks derives runtime filter predicates from the observed data and pushes these filters down to the scan nodes of the left table&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;before&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;left-side data is read. This allows the left table to filter out irrelevant rows early, significantly reducing join input size.&lt;/p&gt; 
 &lt;p&gt;At present, Global Runtime Filters in StarRocks support the following filtering techniques: Min/Max filters, IN predicates, and Bloom filters. The diagram below illustrates how these filters work in practice.&lt;/p&gt;  
 &lt;div&gt; 
  &lt;br&gt; 
  &lt;div&gt;     
  &lt;/div&gt; 
 &lt;/div&gt;  
 &lt;h2&gt;Summary&lt;/h2&gt; 
 &lt;p&gt;This article has explored StarRocks’ practical experience and ongoing work in join query optimization. All of the techniques discussed are closely aligned with the core optimization principles outlined throughout the article. When optimizing SQL queries in practice, users can also apply the following guidelines together with the features provided by StarRocks to achieve better performance:&lt;/p&gt; 
 &lt;ul&gt; 
  &lt;li&gt;&lt;strong&gt;Join operators vary significantly in performance.&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;Prefer high-performance join types whenever possible and avoid expensive ones. Based on typical join output sizes, the rough performance ranking is:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;Semi Join / Anti Join &amp;gt; Inner Join &amp;gt; Outer Join &amp;gt; Full Outer Join &amp;gt; Cross Join&lt;/em&gt;.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;For hash joins, building the hash table on a smaller input is far more efficient&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;than building it on a large table.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;In multi-table joins, execute highly selective joins first&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;to substantially reduce the cost of subsequent joins.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Minimize the amount of data participating in joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;through early filtering and pruning.&lt;/li&gt; 
  &lt;li&gt;&lt;strong&gt;Reduce network overhead in distributed joins&lt;/strong&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;as much as possible to fully benefit from parallel execution.&lt;/li&gt; 
 &lt;/ul&gt; 
 &lt;h2&gt;&amp;nbsp;&lt;/h2&gt; 
 &lt;h2&gt;Case Studies&lt;/h2&gt; 
 &lt;h3&gt;Demandbase&lt;/h3&gt; 
 &lt;p&gt;By leveraging StarRocks’ On-the-Fly JOIN capabilities, Demandbase successfully replaced its existing ClickHouse clusters, optimizing performance while significantly reducing costs across multiple areas.&lt;/p&gt; 
 &lt;p&gt;Read the case study:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;a href="https://medium.com/starrocks-engineering/demandbase-ditches-denormalization-by-switching-off-clickhouse-44195d795a83"&gt;Demandbase Ditches Denormalization By Switching off ClickHouse&lt;/a&gt;&lt;/p&gt; 
 &lt;h3&gt;Naver&lt;/h3&gt; 
 &lt;p&gt;NAVER modernized its data infrastructure with StarRocks by enabling scalable, real-time analytics over multi-table joins without denormalization. The case study highlights the critical role of efficient, on-the-fly join execution in supporting production-scale analytical workloads.&lt;/p&gt; 
 &lt;p&gt;Read the case study:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;a href="https://celerdata.com/blog/how-join-changed-how-we-approach-data-infra-at-naver"&gt;How JOIN Changed How We Approach Data Infra At NAVER&lt;/a&gt;&lt;/p&gt; 
 &lt;h3&gt;Shopee&lt;/h3&gt; 
 &lt;p&gt;Data Go is a no-code query platform where Shopee business users build queries from multiple tables. Presto struggled with complex join performance and high resource usage. When Shopee switched to StarRocks for multi-table joins, they observed&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;em&gt;3×–10× performance improvements&lt;/em&gt;&lt;span&gt;&amp;nbsp;&lt;/span&gt;and a ~60% reduction in CPU usage compared with Presto on external Hive data.&lt;/p&gt; 
 &lt;p&gt;Read the case study:&lt;span&gt;&amp;nbsp;&lt;/span&gt;&lt;a href="https://www.starrocks.io/blog/how-shopee-3xed-their-query-performance-with-starrocks"&gt;How Shopee 3xed Their Query Performance With StarRocks&lt;/a&gt;&lt;/p&gt; 
&lt;/div&gt; 
&lt;div&gt; 
 &lt;span&gt;&lt;/span&gt; 
&lt;/div&gt;  
&lt;img src="https://track.hubspot.com/__ptq.gif?a=21782839&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.starrocks.io%2Fblog%2Finside-starrocks-why-joins-are-faster-than-youd-expect&amp;amp;bu=https%253A%252F%252Fwww.starrocks.io%252Fblog&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Technology</category>
      <pubDate>Wed, 21 Jan 2026 05:18:43 GMT</pubDate>
      <guid>https://www.starrocks.io/blog/inside-starrocks-why-joins-are-faster-than-youd-expect</guid>
      <dc:date>2026-01-21T05:18:43Z</dc:date>
      <dc:creator>Kate Shao</dc:creator>
    </item>
  </channel>
</rss>
