<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>DEV Community: Chen Debra</title>
    <description>The latest articles on DEV Community by Chen Debra (@chen_debra_3060b21d12b1b0).</description>
    <link>https://dev.to/chen_debra_3060b21d12b1b0</link>
    <image>
      <url>https://media2.dev.to/dynamic/image/width=90,height=90,fit=cover,gravity=auto,format=auto/https:%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png</url>
      <title>DEV Community: Chen Debra</title>
      <link>https://dev.to/chen_debra_3060b21d12b1b0</link>
    </image>
    <atom:link rel="self" type="application/rss+xml" href="https://dev.to/feed/chen_debra_3060b21d12b1b0"/>
    <language>en</language>
    <item>
      <title>⚡ Explore how Apache DolphinScheduler powers enterprise data warehouses with seamless orchestration, data processing, BI analytics, and AI applications. Discover the full data journey! 🌐
#ApacheDolphinScheduler #DataEngineering #BigData</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Tue, 22 Sep 2026 09:40:11 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/explore-how-apache-dolphinscheduler-powers-enterprise-data-warehouses-with-seamless-3ib</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/explore-how-apache-dolphinscheduler-powers-enterprise-data-warehouses-with-seamless-3ib</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/from-data-pipelines-to-intelligent-applications-building-enterprise-data-warehouses-with-apache-4om6" class="crayons-story__hidden-navigation-link"&gt;From Data Pipelines to Intelligent Applications: Building Enterprise Data Warehouses with Apache DolphinScheduler&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" alt="chen_debra_3060b21d12b1b0 profile" class="crayons-avatar__image" width="260" height="231"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Chen Debra
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Chen Debra
                
                
              
              &lt;div id="story-author-preview-content-4713976" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/chen_debra_3060b21d12b1b0" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" class="crayons-avatar__image" alt="" width="260" height="231"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Chen Debra&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/from-data-pipelines-to-intelligent-applications-building-enterprise-data-warehouses-with-apache-4om6" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Sep 22&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/from-data-pipelines-to-intelligent-applications-building-enterprise-data-warehouses-with-apache-4om6" id="article-link-4713976"&gt;
          From Data Pipelines to Intelligent Applications: Building Enterprise Data Warehouses with Apache DolphinScheduler
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apachedolphinscheduler"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apachedolphinscheduler&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datawarehouse"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datawarehouse&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/opensource"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;opensource&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/from-data-pipelines-to-intelligent-applications-building-enterprise-data-warehouses-with-apache-4om6#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            10 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>🔎 A failed task with no logs? Learn how Apache DolphinScheduler troubleshooting traces the issue from TaskInstance metadata to Master dispatch and tenant configuration.
#ApacheDolphinScheduler #DataEngineering #OpenSource</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Tue, 22 Sep 2026 09:39:43 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/a-failed-task-with-no-logs-learn-how-apache-dolphinscheduler-troubleshooting-traces-the-issue-nee</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/a-failed-task-with-no-logs-learn-how-apache-dolphinscheduler-troubleshooting-traces-the-issue-nee</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/the-task-failed-but-where-did-the-logs-go-troubleshooting-an-apache-dolphinscheduler-failed-task-2j74" class="crayons-story__hidden-navigation-link"&gt;The Task Failed, but Where Did the Logs Go? Troubleshooting an Apache DolphinScheduler Failed Task&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" alt="chen_debra_3060b21d12b1b0 profile" class="crayons-avatar__image" width="260" height="231"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Chen Debra
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Chen Debra
                
                
              
              &lt;div id="story-author-preview-content-4714623" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/chen_debra_3060b21d12b1b0" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" class="crayons-avatar__image" alt="" width="260" height="231"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Chen Debra&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/the-task-failed-but-where-did-the-logs-go-troubleshooting-an-apache-dolphinscheduler-failed-task-2j74" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Sep 22&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/the-task-failed-but-where-did-the-logs-go-troubleshooting-an-apache-dolphinscheduler-failed-task-2j74" id="article-link-4714623"&gt;
          The Task Failed, but Where Did the Logs Go? Troubleshooting an Apache DolphinScheduler Failed Task
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apachedolphinscheduler"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apachedolphinscheduler&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/opensource"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;opensource&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/bigdata"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;bigdata&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/the-task-failed-but-where-did-the-logs-go-troubleshooting-an-apache-dolphinscheduler-failed-task-2j74#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            9 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>The Task Failed, but Where Did the Logs Go? Troubleshooting an Apache DolphinScheduler Failed Task</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Tue, 22 Sep 2026 09:39:20 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/the-task-failed-but-where-did-the-logs-go-troubleshooting-an-apache-dolphinscheduler-failed-task-2j74</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/the-task-failed-but-where-did-the-logs-go-troubleshooting-an-apache-dolphinscheduler-failed-task-2j74</guid>
      <description>&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;A failure with no logs: tracing a task execution issue from metadata to the Master node&lt;/strong&gt;&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;One of the most confusing scenarios in production scheduling systems is when a task is clearly marked as &lt;strong&gt;"failed"&lt;/strong&gt; in the UI and an alert has already been generated, but critical execution information such as the task start time, execution host, execution directory, and log path are all empty.&lt;/p&gt;

&lt;p&gt;There is no Worker log, no YARN Application ID, and no Spark Driver, Executor, or MapReduce Container logs.&lt;/p&gt;

&lt;p&gt;When facing this situation, engineers often start troubleshooting from the Worker node, script paths, YARN environment, or network connectivity. However, this case revealed a different root cause:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The task had never reached the Worker node. The failure occurred while the Master was constructing the task execution context. The actual issue was neither the script nor the virtual IP, but an invalid tenant configuration in the workflow.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Key takeaway:&lt;/strong&gt;&lt;br&gt;
When a &lt;code&gt;task_instance&lt;/code&gt; is already in the FAILED state, but &lt;code&gt;host&lt;/code&gt;, &lt;code&gt;start_time&lt;/code&gt;, &lt;code&gt;log_path&lt;/code&gt;, and &lt;code&gt;execute_path&lt;/code&gt; are all &lt;code&gt;NULL&lt;/code&gt;, the first place to investigate should be the Master-side task submission and dispatch process, rather than searching for task execution logs that were never generated.&lt;/p&gt;
&lt;h1&gt;
  
  
  1. Incident Overview: Failed Task, Failed Alert, and Empty Logs
&lt;/h1&gt;

&lt;p&gt;In this case, a batch workflow was scheduled to run every day at 14:15. The scheduling UI showed that the first Shell task failed, and subsequent tasks were not executed.&lt;/p&gt;

&lt;p&gt;Meanwhile, the alert table already contained a "scheduler failed" record. However, the alert delivery logs also showed another issue:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;no bind plugin instance
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The latest failed alert can be retrieved from &lt;code&gt;t_ds_alert&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;title&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;content&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;alert_status&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;warning_type&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;log&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;alertgroup_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;create_time&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;process_instance_id&lt;/span&gt;
&lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;t_ds_alert&lt;/span&gt;
&lt;span class="k"&gt;ORDER&lt;/span&gt; &lt;span class="k"&gt;BY&lt;/span&gt; &lt;span class="n"&gt;create_time&lt;/span&gt; &lt;span class="k"&gt;DESC&lt;/span&gt;
&lt;span class="k"&gt;LIMIT&lt;/span&gt; &lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The alert content contains the workflow instance ID, task Code, task name, task type, and failure status, making it a useful entry point for reconstructing the entire troubleshooting chain.&lt;/p&gt;

&lt;p&gt;However, two issues must be distinguished:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Task execution failure&lt;/li&gt;
&lt;li&gt;Alert delivery failure&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These are independent problems and should not be treated as the same failure.&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Problem&lt;/th&gt;
&lt;th&gt;Direct Evidence&lt;/th&gt;
&lt;th&gt;Impact&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Task failure&lt;/td&gt;
&lt;td&gt;&lt;code&gt;taskState=FAILURE&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;The task itself did not execute successfully.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Alert failure&lt;/td&gt;
&lt;td&gt;&lt;code&gt;alertgroup_id=0; no bind plugin instance&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;The failure event was generated, but notifications (e.g., email/DingTalk) were not sent.&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Troubleshooting principle:&lt;/strong&gt;&lt;br&gt;
Do not treat "alert delivery failure" as the cause of task failure. First determine which stage of the task lifecycle failed, then separately troubleshoot alert routing.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h1&gt;
  
  
  2. First Evidence: Metadata Shows the Task Never Reached the Worker
&lt;/h1&gt;

&lt;p&gt;Using the &lt;code&gt;process_instance_id&lt;/code&gt; and &lt;code&gt;taskCode&lt;/code&gt; obtained from the alert, we can query &lt;code&gt;t_ds_task_instance&lt;/code&gt; and locate the exact task execution record.&lt;/p&gt;

&lt;p&gt;The result shows that the task instance was created and its state was already &lt;code&gt;6 (FAILURE)&lt;/code&gt;. However, except for &lt;code&gt;submit_time&lt;/code&gt;, all other runtime-related fields were empty.&lt;/p&gt;

&lt;p&gt;Query the task lifecycle fields:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;task_instance_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;name&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="k"&gt;state&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="k"&gt;host&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;submit_time&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;start_time&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;end_time&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;log_path&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;execute_path&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;app_link&lt;/span&gt;
&lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;t_ds_task_instance&lt;/span&gt;
&lt;span class="k"&gt;WHERE&lt;/span&gt; &lt;span class="n"&gt;process_instance_id&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="n"&gt;process_instance_id&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt;
&lt;span class="k"&gt;AND&lt;/span&gt; &lt;span class="n"&gt;task_code&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="n"&gt;task_code&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt;
&lt;span class="k"&gt;ORDER&lt;/span&gt; &lt;span class="k"&gt;BY&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="k"&gt;DESC&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Field&lt;/th&gt;
&lt;th&gt;Query Result&lt;/th&gt;
&lt;th&gt;Diagnostic Meaning&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;state&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;6&lt;/td&gt;
&lt;td&gt;The task has been marked as failed by the Master.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;submit_time&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;Has value&lt;/td&gt;
&lt;td&gt;The task instance has been created in the metadata database.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;start_time&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;The Worker never started execution.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;host&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;No Worker has been assigned yet.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;execute_path&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;The execution directory has not been created yet.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;log_path&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;No task execution log will be generated.&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;app_link&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;Not submitted to YARN.&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F4eucfqqwnxbxzc5bi9mp.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F4eucfqqwnxbxzc5bi9mp.jpg" width="799" height="519"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;At this point, we can make the first critical conclusion:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The task failed before Worker execution began.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Continuing to SSH into Worker nodes, searching for logs, running &lt;code&gt;yarn logs&lt;/code&gt;, or analyzing Spark Executor issues would not provide useful information, because the execution entity itself was never created.&lt;/p&gt;

&lt;h1&gt;
  
  
  3. A Misleading Clue: Why Does the Master Show a Different IP?
&lt;/h1&gt;

&lt;p&gt;During troubleshooting, the primary IP address of the Master node was found to be inconsistent with the Master address displayed in the scheduling frontend. After SSH-ing into the address shown in the frontend, both the hostname and shell prompt still displayed the primary IP, which initially raised concerns about incorrect routing or an address registration issue.&lt;/p&gt;

&lt;p&gt;The SSH connection tuple, hostname, and all network interface addresses were then verified:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;echo&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="nv"&gt;$SSH_CONNECTION&lt;/span&gt;&lt;span class="s2"&gt;"&lt;/span&gt;
&lt;span class="nb"&gt;hostname&lt;/span&gt; &lt;span class="nt"&gt;-f&lt;/span&gt;
ip &lt;span class="nt"&gt;-br&lt;/span&gt; addr
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The output of &lt;code&gt;ip -br addr&lt;/code&gt; showed that the primary IP was bound with a &lt;code&gt;/24&lt;/code&gt; network mask, while another address was bound with &lt;code&gt;/32&lt;/code&gt; on the same &lt;code&gt;eth0&lt;/code&gt; interface.&lt;/p&gt;

&lt;p&gt;The latter resolved to an EMR virtual hostname, indicating that it was not another machine, but an auxiliary or virtual service IP on the same node.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F1tjnqm9nvab3pue8hve6.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F1tjnqm9nvab3pue8hve6.jpg" width="799" height="519"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Testing SSH connectivity to a virtual IP on the Master node only proves that the address is reachable locally.&lt;/p&gt;

&lt;p&gt;To verify whether scheduling communication was actually working, the service port needed to be tested from a real Worker node.&lt;/p&gt;

&lt;p&gt;Run the following commands from an actual Worker node:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;nc &lt;span class="nt"&gt;-vz&lt;/span&gt; &lt;span class="nt"&gt;-w&lt;/span&gt; 3 &amp;lt;master-primary-ip&amp;gt; 5678
nc &lt;span class="nt"&gt;-vz&lt;/span&gt; &lt;span class="nt"&gt;-w&lt;/span&gt; 3 &amp;lt;master-virtual-ip&amp;gt; 5678
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Both Master addresses successfully accepted connections on port &lt;code&gt;5678&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;Meanwhile, Master heartbeat logs continued to update successfully under:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;/nodes/master/:5678
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;At this point, the possibility of an unreachable Master registration address could be ruled out.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fh8oms07vooedgwh2iopf.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fh8oms07vooedgwh2iopf.jpg" alt="4d06aba445ddef0ca0e4fe68ce316a0f" width="796" height="73"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Lesson learned:&lt;/strong&gt;&lt;br&gt;
A mismatch between the IP displayed in the UI and the host's primary IP does not necessarily indicate a configuration error. First determine whether it is a secondary IP or VIP on the same machine, then verify service connectivity from the actual communication endpoint. Avoid relying only on local self-connectivity tests.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h1&gt;
  
  
  4. Decisive Evidence: "Tenant does not exists" in Master Logs
&lt;/h1&gt;

&lt;p&gt;Since the task never reached the Worker node, the real evidence had to be found in the Master logs.&lt;/p&gt;

&lt;p&gt;By searching the Master logs with multiple stable identifiers, including the workflow instance ID, task instance ID, task Code, and task name, the complete failure chain was reconstructed.&lt;/p&gt;

&lt;p&gt;Use multiple identifiers to locate the same scheduling event:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;grep&lt;/span&gt; &lt;span class="nt"&gt;-R&lt;/span&gt; &lt;span class="nt"&gt;-nE&lt;/span&gt; &lt;span class="s1"&gt;'WorkflowInstance-&amp;lt;id&amp;gt;|TaskInstance-&amp;lt;id&amp;gt;|&amp;lt;task_code&amp;gt;|&amp;lt;task_name&amp;gt;'&lt;/span&gt; /var/log/.../master-server/
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The failure occurred within milliseconds, before the task was actually dispatched to a Worker:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;14:15:00.650 Task is ready to dispatch to worker
14:15:00.651 Tenant does not exists
14:15:00.651 Task state changes to FAILURE
14:15:00.652 Get taskExecutionContext fail
14:15:00.653 Dispatch standby task failed
14:15:00.690 Workflow state changes to FAILURE
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The log output provided the decisive evidence:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;processDefinition.tenantId=-1
processDefinition.tenantCode=null

processInstance.tenantId=-1
processInstance.tenantCode=null
processInstance.queue=null
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;When constructing the &lt;code&gt;TaskExecutionContext&lt;/code&gt;, the Master node must determine the execution tenant information. However, the current workflow did not have a valid tenant configuration.&lt;/p&gt;

&lt;p&gt;As a result, DolphinScheduler could not generate the task execution context, and the task was marked as &lt;code&gt;FAILURE&lt;/code&gt; before Worker allocation.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Root cause:&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The workflow definition and workflow instance both had &lt;code&gt;tenantId=-1&lt;/code&gt;, with an empty &lt;code&gt;tenantCode&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;The Master failed to construct the &lt;code&gt;TaskExecutionContext&lt;/code&gt;, causing the task to fail before it was dispatched to a Worker node.&lt;/p&gt;

&lt;h1&gt;
  
  
  5. Why Are There No Logs? Understanding the DolphinScheduler Task Lifecycle
&lt;/h1&gt;

&lt;p&gt;In DolphinScheduler, task logs are not generated when a &lt;code&gt;TaskInstance&lt;/code&gt; record is created in the database.&lt;/p&gt;

&lt;p&gt;Instead, logs become available only after the following steps are completed:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;The Master successfully constructs the task execution context.&lt;/li&gt;
&lt;li&gt;The task is assigned to an available Worker.&lt;/li&gt;
&lt;li&gt;The Worker creates the execution directory and starts running the task.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;Understanding this sequence is essential when troubleshooting the situation where a task fails but no logs exist.&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Stage&lt;/th&gt;
&lt;th&gt;Typical Field Changes&lt;/th&gt;
&lt;th&gt;Where to Check&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;1. Create instance&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;submit_time&lt;/code&gt; has value; &lt;code&gt;state&lt;/code&gt; = submit successful&lt;/td&gt;
&lt;td&gt;Metadata database, Master logs&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;2. Build context&lt;/td&gt;
&lt;td&gt;Populate &lt;code&gt;tenant&lt;/code&gt;, &lt;code&gt;queue&lt;/code&gt;, environment, etc.&lt;/td&gt;
&lt;td&gt;Master logs&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;3. Assign Worker&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;host&lt;/code&gt; has value&lt;/td&gt;
&lt;td&gt;Master dispatch logs, registration center&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;4. Worker startup&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;start_time&lt;/code&gt;, &lt;code&gt;execute_path&lt;/code&gt;, &lt;code&gt;log_path&lt;/code&gt; have values&lt;/td&gt;
&lt;td&gt;Worker task logs&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;5. Submit big-data job&lt;/td&gt;
&lt;td&gt;
&lt;code&gt;app_link&lt;/code&gt; / Application ID has value&lt;/td&gt;
&lt;td&gt;YARN and Container logs&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F7yt2xruqlcu6ssd9fmxx.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F7yt2xruqlcu6ssd9fmxx.jpg" width="800" height="574"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;These lifecycle fields also provide a practical troubleshooting strategy:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Field status&lt;/th&gt;
&lt;th&gt;Investigation direction&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;
&lt;code&gt;host&lt;/code&gt; is empty&lt;/td&gt;
&lt;td&gt;Check Master-side task creation and dispatch&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;
&lt;code&gt;host&lt;/code&gt; exists but &lt;code&gt;start_time&lt;/code&gt; is empty&lt;/td&gt;
&lt;td&gt;Check Worker assignment and Worker acceptance&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;
&lt;code&gt;log_path&lt;/code&gt; exists&lt;/td&gt;
&lt;td&gt;Check Worker execution logs&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Application ID exists&lt;/td&gt;
&lt;td&gt;Proceed with YARN log investigation&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;In other words, the absence of logs is itself a valuable signal.&lt;/p&gt;

&lt;h1&gt;
  
  
  6. What Role Does Tenant Play in DolphinScheduler?
&lt;/h1&gt;

&lt;p&gt;In DolphinScheduler, a Tenant is not merely a classification field displayed in the UI.&lt;/p&gt;

&lt;p&gt;It is closely related to task execution identity and resource allocation.&lt;/p&gt;

&lt;p&gt;When the Master constructs a task execution context, it needs to resolve information such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;code&gt;tenantCode&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;execution user&lt;/li&gt;
&lt;li&gt;resource queue&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;This information tells the Worker:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Which user identity should execute the task?&lt;/li&gt;
&lt;li&gt;Which resource queue should be used?&lt;/li&gt;
&lt;li&gt;Which permissions should be applied?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Without valid tenant information, the Master cannot safely create an execution environment for the task.&lt;/p&gt;

&lt;p&gt;In environments using:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Linux user switching&lt;/li&gt;
&lt;li&gt;HDFS permission control&lt;/li&gt;
&lt;li&gt;Kerberos authentication&lt;/li&gt;
&lt;li&gt;YARN queue isolation&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;tenant configuration can further affect:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Operating system user identity&lt;/li&gt;
&lt;li&gt;HDFS directory permissions&lt;/li&gt;
&lt;li&gt;Authentication tickets&lt;/li&gt;
&lt;li&gt;Resource queue assignment&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Therefore, even after repairing the tenant configuration, engineers should still verify the corresponding OS user permissions and big data platform access control.&lt;/p&gt;

&lt;h1&gt;
  
  
  7. Confirm Tenant Relationships Through Metadata
&lt;/h1&gt;

&lt;p&gt;To verify the relationship between workflows, users, and tenants, query the following metadata:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;code&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;name&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="k"&gt;version&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;user_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;u&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;user_name&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;tenant_id&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;workflow_tenant_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;u&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;tenant_id&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;user_tenant_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;t&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;matched_tenant_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
&lt;span class="n"&gt;t&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;tenant_code&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;t&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;queue_id&lt;/span&gt;
&lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;t_ds_process_definition&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;
&lt;span class="k"&gt;LEFT&lt;/span&gt; &lt;span class="k"&gt;JOIN&lt;/span&gt; &lt;span class="n"&gt;t_ds_user&lt;/span&gt; &lt;span class="n"&gt;u&lt;/span&gt; &lt;span class="k"&gt;ON&lt;/span&gt; &lt;span class="n"&gt;u&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;user_id&lt;/span&gt;
&lt;span class="k"&gt;LEFT&lt;/span&gt; &lt;span class="k"&gt;JOIN&lt;/span&gt; &lt;span class="n"&gt;t_ds_tenant&lt;/span&gt; &lt;span class="n"&gt;t&lt;/span&gt; &lt;span class="k"&gt;ON&lt;/span&gt; &lt;span class="n"&gt;t&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;tenant_id&lt;/span&gt;
&lt;span class="k"&gt;WHERE&lt;/span&gt; &lt;span class="n"&gt;pd&lt;/span&gt;&lt;span class="p"&gt;.&lt;/span&gt;&lt;span class="n"&gt;code&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="n"&gt;process_definition_code&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Confirm that the tenant exists and check historical workflow versions:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="o"&gt;*&lt;/span&gt; &lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;t_ds_tenant&lt;/span&gt; &lt;span class="k"&gt;ORDER&lt;/span&gt; &lt;span class="k"&gt;BY&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="o"&gt;*&lt;/span&gt; &lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;t_ds_user&lt;/span&gt;
&lt;span class="k"&gt;WHERE&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="n"&gt;user_id&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt;
&lt;span class="k"&gt;OR&lt;/span&gt; &lt;span class="n"&gt;user_name&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="s1"&gt;'&amp;lt;user_name&amp;gt;'&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="n"&gt;code&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;name&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="k"&gt;version&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;tenant_id&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;release_state&lt;/span&gt;
&lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;t_ds_process_definition_log&lt;/span&gt;
&lt;span class="k"&gt;WHERE&lt;/span&gt; &lt;span class="n"&gt;code&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="n"&gt;process_definition_code&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt;
&lt;span class="k"&gt;ORDER&lt;/span&gt; &lt;span class="k"&gt;BY&lt;/span&gt; &lt;span class="k"&gt;version&lt;/span&gt; &lt;span class="k"&gt;DESC&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The expected abnormal result in this case was:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;workflow_tenant_id = -1
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;and no corresponding record could be found in &lt;code&gt;t_ds_tenant&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;If a user already has a tenant but the workflow still shows &lt;code&gt;tenant_id=-1&lt;/code&gt;, it indicates that the workflow did not correctly inherit tenant information during creation, import, or version migration.&lt;/p&gt;

&lt;p&gt;Simply modifying the user configuration does not necessarily update existing workflows automatically.&lt;/p&gt;

&lt;h1&gt;
  
  
  8. Fix Procedure: Avoid Direct Updates to Metadata Tables
&lt;/h1&gt;

&lt;p&gt;The recommended approach is to repair the configuration through the DolphinScheduler UI instead of directly modifying metadata tables.&lt;/p&gt;

&lt;p&gt;Direct database updates may bypass:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;workflow version management&lt;/li&gt;
&lt;li&gt;cache synchronization&lt;/li&gt;
&lt;li&gt;permission relationships&lt;/li&gt;
&lt;li&gt;audit information&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The standard repair process is:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;1. Create or select a valid tenant&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;In &lt;strong&gt;Security Center → Tenant Management&lt;/strong&gt;, create or select a valid tenant and associate the correct YARN Queue.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;2. Bind the tenant to the workflow owner&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;In &lt;strong&gt;User Management&lt;/strong&gt;, assign the tenant to the workflow owner.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;3. Update the workflow configuration&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Edit the failed workflow, select a valid tenant, save the workflow, and generate a new workflow version.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;4. Publish the updated workflow&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Release the new workflow version and verify that:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;t_ds_process_definition.tenant_id &amp;gt; 0
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;strong&gt;5. Start a new execution&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Trigger a completely new manual run instead of recovering the old failed instance.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;6. Verify runtime permissions&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;After the task reaches the Worker node, verify:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Linux user permissions&lt;/li&gt;
&lt;li&gt;script execution permissions&lt;/li&gt;
&lt;li&gt;HDFS access&lt;/li&gt;
&lt;li&gt;Kerberos authentication&lt;/li&gt;
&lt;li&gt;YARN queue permissions&lt;/li&gt;
&lt;/ul&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Why not rerun the old instance?&lt;/strong&gt;&lt;br&gt;
A failed instance stores a snapshot of the previous workflow version, where &lt;code&gt;tenantId&lt;/code&gt; is still &lt;code&gt;-1&lt;/code&gt;. Recovering the old instance may continue using the invalid configuration. After publishing a new workflow version, create a new instance for validation.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h1&gt;
  
  
  9. How to Validate After Fixing the Issue
&lt;/h1&gt;

&lt;p&gt;After the repair is completed, validation should not stop at confirming that the workflow status turns green.&lt;/p&gt;

&lt;p&gt;A complete verification process should confirm that the task has successfully passed the original failure point.&lt;/p&gt;

&lt;p&gt;It is recommended to verify the workflow from three layers:&lt;/p&gt;

&lt;h3&gt;
  
  
  Definition layer
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;The latest workflow version has a valid &lt;code&gt;tenant_id&lt;/code&gt; greater than 0.&lt;/li&gt;
&lt;li&gt;The tenant can be correctly mapped to a record in &lt;code&gt;t_ds_tenant&lt;/code&gt;.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Instance layer
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;The new workflow instance contains valid &lt;code&gt;tenantCode&lt;/code&gt;.&lt;/li&gt;
&lt;li&gt;The resource queue information is no longer empty.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Dispatch layer
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;code&gt;TaskInstance.host&lt;/code&gt; is populated.&lt;/li&gt;
&lt;li&gt;Worker nodes receive the task successfully.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Runtime layer
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;code&gt;start_time&lt;/code&gt;, &lt;code&gt;execute_path&lt;/code&gt;, and &lt;code&gt;log_path&lt;/code&gt; are generated correctly.&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  Big data execution layer
&lt;/h3&gt;

&lt;p&gt;For tasks submitting Spark or MapReduce jobs:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;The Application ID can be retrieved.&lt;/li&gt;
&lt;li&gt;YARN logs can be queried successfully.&lt;/li&gt;
&lt;/ul&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Acceptance Field&lt;/th&gt;
&lt;th&gt;When Fault Occurs&lt;/th&gt;
&lt;th&gt;Expected After Fix&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;tenant_id&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;-1&lt;/td&gt;
&lt;td&gt;&amp;gt; 0&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;
&lt;code&gt;tenantCode&lt;/code&gt; / &lt;code&gt;queue&lt;/code&gt;
&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;Valid value&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;host&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;Worker address&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;start_time&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;Actual start time&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;
&lt;code&gt;execute_path&lt;/code&gt; / &lt;code&gt;log_path&lt;/code&gt;
&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;Actual execution directory and log path&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;code&gt;app_link&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;NULL&lt;/td&gt;
&lt;td&gt;Application ID or link appears after submission to YARN&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;h1&gt;
  
  
  Conclusion: No Logs Are Logs
&lt;/h1&gt;

&lt;p&gt;In scheduling systems, "no logs" does not mean there is no evidence.&lt;/p&gt;

&lt;p&gt;On the contrary, when &lt;code&gt;host&lt;/code&gt;, &lt;code&gt;start_time&lt;/code&gt;, &lt;code&gt;execute_path&lt;/code&gt;, &lt;code&gt;log_path&lt;/code&gt;, and &lt;code&gt;app_link&lt;/code&gt; are all empty, these fields already provide a strong indication:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The task never entered the execution layer.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Following this clue back to the Master node, the millisecond-level failure timeline eventually pointed to:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Tenant does not exists
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;A high-quality troubleshooting process is not about checking every component blindly.&lt;/p&gt;

&lt;p&gt;Instead, it requires:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Using metadata to define the investigation boundary.&lt;/li&gt;
&lt;li&gt;Using logs to reconstruct the execution timeline.&lt;/li&gt;
&lt;li&gt;Using network tests to eliminate unrelated possibilities.&lt;/li&gt;
&lt;li&gt;Separately closing the loop on task failures and alert delivery failures.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;This approach transforms troubleshooting from experience-based debugging into a systematic, explainable, and reusable diagnostic methodology.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;One-line summary:&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;When a &lt;code&gt;TaskInstance&lt;/code&gt; enters the FAILED state but all runtime fields are &lt;code&gt;NULL&lt;/code&gt;, check the Master first. In this case, the direct cause was &lt;code&gt;tenantId=-1&lt;/code&gt;, which prevented the Master from constructing the &lt;code&gt;TaskExecutionContext&lt;/code&gt;.&lt;/p&gt;

</description>
      <category>apachedolphinscheduler</category>
      <category>opensource</category>
      <category>datascience</category>
      <category>bigdata</category>
    </item>
    <item>
      <title>From Data Pipelines to Intelligent Applications: Building Enterprise Data Warehouses with Apache DolphinScheduler</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Tue, 22 Sep 2026 08:30:28 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/from-data-pipelines-to-intelligent-applications-building-enterprise-data-warehouses-with-apache-4om6</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/from-data-pipelines-to-intelligent-applications-building-enterprise-data-warehouses-with-apache-4om6</guid>
      <description>&lt;p&gt;As enterprises accelerate their digital transformation, data platforms are evolving from traditional storage and analytics systems into foundational infrastructures that support business decision-making, intelligent applications, and data asset management.&lt;/p&gt;

&lt;p&gt;Building a stable, efficient, and scalable data production system has become a key challenge in modern enterprise data warehouse (DW) initiatives.&lt;/p&gt;

&lt;p&gt;At the Apache DolphinScheduler September Meetup, community practitioner Zan Liu shared practical experiences in building enterprise-grade data warehouses. The session explored how Apache DolphinScheduler can be used as the orchestration core, working together with data integration, data processing, data applications, and operational management capabilities to build a complete data engineering ecosystem.&lt;/p&gt;

&lt;p&gt;Based on the Meetup session, this article provides an in-depth look at how Apache DolphinScheduler supports enterprise data warehouse practices across four key stages: environment setup, data processing, data applications, and operational upgrades.&lt;/p&gt;

&lt;p&gt;👉🏻 Replay the Meetup session &lt;a href="https://youtu.be/UT3_xV4Awhs?si=I4TI9uxQKt6wbF_l" rel="noopener noreferrer"&gt;here&lt;/a&gt;&lt;/p&gt;

&lt;h1&gt;
  
  
  About the Speaker
&lt;/h1&gt;

&lt;p&gt;&lt;strong&gt;Liu Zan&lt;/strong&gt;：&lt;strong&gt;Data Analytics Engineer &amp;amp; Senior Big Data Architect&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;With extensive experience in data warehousing and big data technologies, Liu Zan specializes in BI development and Dify-based low-code Agent development. He has deep expertise in building enterprise data platforms and implementing data-driven applications.&lt;/p&gt;

&lt;h1&gt;
  
  
  Environment Setup: Building a Data Foundation for Enterprise Data Warehouses
&lt;/h1&gt;

&lt;p&gt;The first step in building an enterprise data warehouse is selecting a technology architecture that can support long-term business growth and establishing a stable, reliable data infrastructure.&lt;/p&gt;

&lt;p&gt;During the early stages of big data development, the Hadoop ecosystem became the foundation for many enterprise data platforms due to its distributed storage and computing capabilities. However, as real-time data requirements continue to increase, traditional architectures have gradually revealed limitations in certain scenarios.&lt;/p&gt;

&lt;p&gt;On one hand, traditional Hadoop-based architectures are primarily designed for batch processing workloads. Data jobs often experience minute-level latency, making them less suitable for real-time analytics and fast decision-making scenarios.&lt;/p&gt;

&lt;p&gt;On the other hand, traditional architectures tightly couple computing and storage resources. When enterprises need to scale a specific type of resource, such as computing capacity or storage capacity, the other resource is often affected as well, resulting in lower resource utilization.&lt;/p&gt;

&lt;p&gt;In addition, the complexity of the big data ecosystem introduces higher costs in terms of component management, version compatibility, system maintenance, and technical learning.&lt;/p&gt;

&lt;p&gt;Therefore, modern enterprise data warehouse practices are increasingly adopting lightweight and high-performance OLAP architectures.&lt;/p&gt;

&lt;h2&gt;
  
  
  1. Technology Selection: Using OLAP Databases to Support High-Performance Analytics
&lt;/h2&gt;

&lt;p&gt;During the early phase of data warehouse development, the team adopted a traditional big data warehouse architecture. However, as data volume increased and business analytics requirements became more complex, several limitations gradually emerged in real-world usage.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F76xlvmyz9sdjwde92bay.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F76xlvmyz9sdjwde92bay.jpg" width="800" height="736"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;First, traditional data warehouse architectures are mainly designed for offline batch processing scenarios. Job execution usually involves relatively high latency, with many tasks taking several minutes to complete, making it difficult to meet the low-latency requirements of certain business analytics scenarios.&lt;/p&gt;

&lt;p&gt;Second, traditional architectures suffer from tightly coupled storage and compute resources. Since computing nodes are responsible for both data storage and processing, scaling either compute or storage independently becomes difficult. This can lead to inefficient resource utilization and increase the complexity of platform expansion.&lt;/p&gt;

&lt;p&gt;Furthermore, traditional big data ecosystems usually rely on multiple components working together. The large number of components, combined with version compatibility requirements, increases the complexity of platform deployment, learning, and long-term operations.&lt;/p&gt;

&lt;p&gt;To address these challenges in performance, scalability, and operations, the team adopted a distributed OLAP database architecture in its next-generation data warehouse implementation, improving analytics capabilities and overall platform efficiency.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F3yqt2wi707ock19fjomv.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F3yqt2wi707ock19fjomv.jpg" width="800" height="711"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;In the new architecture, the orchestration layer adopts the open-source project Apache DolphinScheduler for workflow orchestration, task scheduling, dependency management, retry and alerting, resource management, and visual operations.&lt;/p&gt;

&lt;p&gt;The data integration layer adopts another open-source project, Apache SeaTunnel, which provides data source connectivity, batch and streaming data synchronization, data transformation, and cleansing capabilities.&lt;/p&gt;

&lt;p&gt;The new architecture delivers several improvements.&lt;/p&gt;

&lt;p&gt;First, it provides significantly improved query performance. Based on distributed OLAP architecture, it enables sub-second responses for large-scale data queries. In wide-table aggregation scenarios, performance can improve by 5 to 10 times compared with traditional solutions, better supporting fast business analytics.&lt;/p&gt;

&lt;p&gt;Second, the new architecture provides stronger scalability. Compared with traditional Hadoop ecosystems that rely on multiple components such as HBase and Spark, the new solution reduces dependencies on numerous third-party components, resulting in a simpler architecture and lower maintenance complexity.&lt;/p&gt;

&lt;p&gt;Meanwhile, based on distributed cluster technology, the architecture provides stronger concurrent processing capabilities, reduces single-node I/O pressure, and improves overall platform stability.&lt;/p&gt;

&lt;p&gt;Through this technology selection, the team successfully upgraded from a traditional data warehouse architecture to a modern OLAP-based architecture, establishing a more reliable data foundation for subsequent data synchronization, processing, and business applications.&lt;/p&gt;

&lt;h2&gt;
  
  
  2. Environment Setup: Reducing the Complexity of Big Data Platform Deployment
&lt;/h2&gt;

&lt;p&gt;After completing the architecture selection, the next step was to build the infrastructure required for running data processing workflows.&lt;/p&gt;

&lt;p&gt;During the environment setup process, the team used a cluster management tool to deploy the big data environment. The tool provides fast deployment capabilities, compatibility with open-source ecosystems, and simplified operations and maintenance.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fu16hzz7jd47zjo0uf59d.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fu16hzz7jd47zjo0uf59d.jpg" width="799" height="649"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Through centralized management, users can quickly initialize large-scale big data clusters while covering multiple areas including data integration, storage, computing engines, task scheduling, and permission management.&lt;/p&gt;

&lt;p&gt;From an operations perspective, the tool enables unified monitoring and management of clusters, nodes, and services, improving daily maintenance efficiency.&lt;/p&gt;

&lt;p&gt;After completing the infrastructure setup, the team further deployed Apache DolphinScheduler to provide unified scheduling capabilities for subsequent data workflows.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. Deployment Process: Setting Up Apache DolphinScheduler
&lt;/h2&gt;

&lt;p&gt;Within the overall data warehouse architecture, Apache DolphinScheduler is responsible for data workflow orchestration and scheduling management.&lt;/p&gt;

&lt;p&gt;By deploying DolphinScheduler, the team was able to integrate subsequent data synchronization and data processing tasks into a unified workflow management system, providing visual orchestration and automated execution capabilities.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fziqx9jjqkg4i7zrp3evl.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fziqx9jjqkg4i7zrp3evl.jpg" width="800" height="387"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fo7rtnhi05a90o7d3px16.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fo7rtnhi05a90o7d3px16.jpg" width="798" height="334"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h1&gt;
  
  
  Data Processing: Connecting the Data Production Pipeline with Apache DolphinScheduler
&lt;/h1&gt;

&lt;p&gt;After completing the infrastructure setup, data processing becomes the core stage of enterprise data warehouse development. In this practice, the team focused on data synchronization, data transformation, and task scheduling, using Apache DolphinScheduler to orchestrate and manage data workflows.&lt;/p&gt;

&lt;p&gt;Within the overall data processing pipeline, DolphinScheduler is responsible for workflow management and task scheduling, while Apache SeaTunnel handles data synchronization. Together, the two projects enable an automated process from data ingestion to data processing.&lt;/p&gt;

&lt;p&gt;This practice mainly covers Apache DolphinScheduler deployment, SeaTunnel integration, mirror table synchronization, model-layer processing, exchange table applications, and exploration of real-time data synchronization solutions.&lt;/p&gt;

&lt;h2&gt;
  
  
  1. Apache DolphinScheduler Application: Enabling Unified Data Task Scheduling
&lt;/h2&gt;

&lt;p&gt;During the data processing phase, the team adopted Apache DolphinScheduler to centrally manage data workflows.&lt;/p&gt;

&lt;p&gt;Compared with traditional approaches that rely on manually maintained scripts to manage task execution dependencies, DolphinScheduler provides visual workflow orchestration capabilities.&lt;/p&gt;

&lt;p&gt;Users can define workflows through a drag-and-drop interface and organize different data processing tasks into DAG workflows based on their dependencies, significantly reducing workflow management complexity.&lt;/p&gt;

&lt;p&gt;At the same time, DolphinScheduler supports multiple task types and provides cross-language extensibility, enabling it to adapt to diverse data processing scenarios. In enterprise data warehouse projects, different types of data tasks can be integrated into a unified scheduling system, simplifying workflow management and maintenance.&lt;/p&gt;

&lt;p&gt;In terms of reliability, DolphinScheduler adopts a decentralized architecture, improving scheduling system stability and ensuring continuous execution of data workflows.&lt;/p&gt;

&lt;p&gt;By introducing DolphinScheduler, the team connected data synchronization and data processing steps into a unified workflow, enabling centralized scheduling throughout the data production pipeline.&lt;/p&gt;

&lt;h2&gt;
  
  
  2. Apache SeaTunnel Integration: Building Data Synchronization Pipelines
&lt;/h2&gt;

&lt;p&gt;For data ingestion and synchronization, the team introduced Apache SeaTunnel as the data integration component responsible for transferring data between different data sources.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fa3z6ezp9w1o7hbleoi5v.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fa3z6ezp9w1o7hbleoi5v.jpg" width="800" height="668"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Based on its Connector architecture, SeaTunnel supports multiple data synchronization scenarios, including offline synchronization, real-time synchronization, full synchronization, and incremental synchronization, reducing the complexity of managing data integration tasks.&lt;/p&gt;

&lt;p&gt;At the execution engine layer, SeaTunnel uses the Zeta engine by default while also supporting Flink and Spark as Connector execution engines, providing greater flexibility for different business scenarios.&lt;/p&gt;

&lt;p&gt;In addition, SeaTunnel improves data transfer efficiency through parallel reading and writing mechanisms, delivering stable, high-throughput, and low-latency data synchronization capabilities.&lt;/p&gt;

&lt;p&gt;Within the overall data processing architecture, DolphinScheduler and SeaTunnel work together: DolphinScheduler handles workflow orchestration and execution scheduling, while SeaTunnel performs data synchronization operations. Together, they provide the foundation for efficient data movement across the data warehouse pipeline.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. Mirror Table Synchronization: Improving Data Synchronization Flexibility
&lt;/h2&gt;

&lt;p&gt;During the data synchronization process, the team adopted a mirror table synchronization approach to complete data migration.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Schema synchronization, flexible configuration, and simple operations&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;This solution provides a configuration experience similar to DataX while offering more comprehensive data synchronization capabilities.&lt;/p&gt;

&lt;p&gt;In practical applications, the team used the tool to synchronize table schemas and then leveraged DolphinScheduler to centrally schedule synchronization tasks, enabling automated data migration workflows.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Data synchronization (DolphinScheduler)&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;By integrating synchronization tasks into the scheduling system, data migration workflows can be automatically executed according to predefined processes. Combined with task status management, this approach improves the controllability and reliability of data processing operations.&lt;/p&gt;

&lt;h2&gt;
  
  
  4. Model Layer Processing: Simplifying Data Transformation Workflows
&lt;/h2&gt;

&lt;p&gt;After completing data synchronization, the team further processed data at the model layer.&lt;/p&gt;

&lt;p&gt;During the model-layer data cleansing stage, Shell scripts were used for business logic processing. This scripting approach improved development efficiency while reducing future maintenance costs.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F7dx6eekadtyxiewjm74p.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F7dx6eekadtyxiewjm74p.jpg" width="610" height="535"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;In the actual workflow, model-layer processing tasks were also managed through DolphinScheduler, forming a complete data processing pipeline together with upstream and downstream synchronization tasks.&lt;/p&gt;

&lt;h2&gt;
  
  
  5. Exchange Table Application: Enabling Seamless Data Switching
&lt;/h2&gt;

&lt;p&gt;To improve data reliability and stability during updates, the team adopted an exchange table mechanism for data switching.&lt;/p&gt;

&lt;p&gt;Through this approach, table data can be switched seamlessly without affecting business users, minimizing the impact of data update operations.&lt;/p&gt;

&lt;p&gt;Combined with DolphinScheduler’s workflow orchestration capabilities, data processing and table switching operations can be executed automatically according to predefined sequences, further improving the stability of the data production pipeline.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fmuhbmbdxoax8hx7fggxh.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fmuhbmbdxoax8hx7fggxh.jpg" width="789" height="512"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  6. Real-Time Solution Exploration: Investigating CDC-Based Real-Time Synchronization
&lt;/h2&gt;

&lt;p&gt;Beyond offline data processing workflows, the team also explored solutions for real-time data synchronization scenarios.&lt;/p&gt;

&lt;p&gt;This practice investigated a CDC-based real-time synchronization architecture without relying on Flink. The solution features a simpler architecture, strong performance, and easier operations and maintenance.&lt;/p&gt;

&lt;p&gt;Through exploring real-time synchronization solutions, the team further enhanced its data processing capabilities and prepared the platform to support a wider range of future data application scenarios.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fd6ex8mlv6idgo5b1p2z2.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fd6ex8mlv6idgo5b1p2z2.jpg" width="649" height="509"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h1&gt;
  
  
  Data Applications: From Data Analytics to Intelligent Data Services
&lt;/h1&gt;

&lt;p&gt;After data ingestion, synchronization, and transformation are completed, data needs to be further leveraged to support business analytics and application scenarios.&lt;/p&gt;

&lt;p&gt;In this data warehouse implementation, the team focused on three key areas: BI applications, AI agent applications, and data governance, expanding the platform’s data service capabilities.&lt;/p&gt;

&lt;h2&gt;
  
  
  1. BI Applications: Supporting Data Analytics and Business Visualization
&lt;/h2&gt;

&lt;p&gt;At the data application layer, the team connected databases with BI tools through database interfaces to enable analytics and visualization.&lt;/p&gt;

&lt;p&gt;Based on built-in JDBC drivers, the database can be connected directly to BI platforms, providing strong concurrency capabilities and fast query response times. Meanwhile, compatibility with the MySQL protocol enables integration with existing data analytics tools.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Reporting application (FineReport)&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;For practical applications, the team used FineReport for reporting scenarios and DataEase for dashboard visualization.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Dashboard application (DataEase)&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Through the previous data processing workflow, data synchronization and transformation tasks are centrally scheduled by Apache DolphinScheduler, ensuring that data updates are completed according to predefined processes and providing a stable foundation for BI analytics and visualization.&lt;/p&gt;

&lt;h2&gt;
  
  
  2. AI Agent Applications: Exploring Data Querying Scenarios
&lt;/h2&gt;

&lt;p&gt;In traditional data analytics scenarios, users typically rely on reports or query tools to access data. To further expand data application methods, the team developed a data query AI agent based on Dify.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F8upnylr0mbc42rxz11td.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F8upnylr0mbc42rxz11td.jpg" width="800" height="399"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;By building a data query AI agent, users can access data through a more convenient interaction method, providing new possibilities for data applications.&lt;/p&gt;

&lt;p&gt;This practice demonstrates that a data platform needs to provide not only reliable data processing capabilities but also strong data support for upper-layer applications. The data consumed by intelligent applications still relies on continuous data generation through underlying data warehouse synchronization and processing workflows.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. Data Governance: Enhancing Metadata Management Capabilities
&lt;/h2&gt;

&lt;p&gt;As enterprise data assets continue to grow, data governance has become an essential component of modern data warehouse development.&lt;/p&gt;

&lt;p&gt;In this practice, the team used OpenMetadata to support metadata management for Doris databases.&lt;/p&gt;

&lt;p&gt;Through metadata management, data asset information stored in databases can be centrally managed, providing better support for data usage, maintenance, and future expansion.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F6mofm2xzwb6s2qxfv8ma.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F6mofm2xzwb6s2qxfv8ma.jpg" width="800" height="415"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h1&gt;
  
  
  Operations and Maintenance Upgrades: Ensuring Stable Data Warehouse Operations
&lt;/h1&gt;

&lt;p&gt;After completing data warehouse construction and application implementation, long-term platform stability still requires a comprehensive operations and maintenance system.&lt;/p&gt;

&lt;p&gt;In this practice, the team enhanced platform operations capabilities in three key areas: log monitoring, upgrade management, and failure alerting.&lt;/p&gt;

&lt;h2&gt;
  
  
  1. Log Monitoring: Improving System Observability
&lt;/h2&gt;

&lt;p&gt;During daily operations, the team uses log monitoring to gain real-time visibility into the status of different platform components.&lt;/p&gt;

&lt;p&gt;By monitoring component logs in real time and combining metric collection with customized alert mechanisms, the team can quickly identify and address potential issues.&lt;/p&gt;

&lt;p&gt;In practice, the team uses Apache DolphinScheduler to view task execution logs while integrating Prometheus and Grafana for daily monitoring.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fp94zrpu382fleg3bb7ou.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fp94zrpu382fleg3bb7ou.jpg" width="800" height="442"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Among these tools, DolphinScheduler provides visibility into task execution logs, while Prometheus and Grafana monitor platform metrics. Together, they improve overall system observability and operational efficiency.&lt;/p&gt;

&lt;h2&gt;
  
  
  2. Upgrade Management: Simplifying Doris Cluster Operations
&lt;/h2&gt;

&lt;p&gt;For database cluster maintenance, the team adopted DorisManager for Doris cluster management.&lt;/p&gt;

&lt;p&gt;Through this all-in-one Doris cluster management tool, cluster upgrades and maintenance operations can be simplified, improving overall platform management efficiency.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. Failure Alerting: Establishing Task Failure Notification Mechanisms
&lt;/h2&gt;

&lt;p&gt;To detect data task failures in a timely manner, the team configured alert instances in Apache DolphinScheduler and integrated them with Feishu bots for failure notifications.&lt;/p&gt;

&lt;p&gt;When a task fails during execution, the system can automatically send notifications through the alert mechanism, helping relevant teams quickly identify and resolve issues.&lt;/p&gt;

&lt;h1&gt;
  
  
  Conclusion: Apache DolphinScheduler Enables End-to-End Enterprise Data Warehouse Development
&lt;/h1&gt;

&lt;p&gt;At the Apache DolphinScheduler September Meetup, the community shared a practical enterprise data warehouse implementation covering the complete lifecycle from environment setup and data processing to data applications and operational management.&lt;/p&gt;

&lt;p&gt;Within this architecture, Apache DolphinScheduler runs throughout the entire data production workflow, connecting data synchronization, data transformation, and operational management through workflow orchestration and task scheduling capabilities.&lt;/p&gt;

&lt;p&gt;Apache SeaTunnel handles data synchronization, OLAP databases provide analytical capabilities, while BI tools, AI agents, and metadata management systems further extend data application scenarios.&lt;/p&gt;

&lt;p&gt;Through this implementation, the team built a comprehensive data platform covering both data processing and data applications, providing reliable support for stable enterprise data warehouse operations.&lt;/p&gt;

</description>
      <category>apachedolphinscheduler</category>
      <category>datawarehouse</category>
      <category>datascience</category>
      <category>opensource</category>
    </item>
    <item>
      <title>🔍 Ever wondered how Apache DolphinScheduler’s Master turns workflow commands into reliable task execution? Dive into the source code and trace its startup, scheduling, RPC, and failover flow. ⚙️
#ApacheDolphinScheduler #OpenSource #DataEngineering</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 18 Sep 2026 08:08:07 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/ever-wondered-how-apache-dolphinschedulers-master-turns-workflow-commands-into-reliable-task-3fjl</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/ever-wondered-how-apache-dolphinschedulers-master-turns-workflow-commands-into-reliable-task-3fjl</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/inside-dolphinschedulers-master-startup-process-a-source-code-walkthrough-5hll" class="crayons-story__hidden-navigation-link"&gt;Inside DolphinScheduler’s Master Startup Process: A Source Code Walkthrough&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" alt="chen_debra_3060b21d12b1b0 profile" class="crayons-avatar__image" width="260" height="231"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Chen Debra
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Chen Debra
                
                
              
              &lt;div id="story-author-preview-content-4682962" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/chen_debra_3060b21d12b1b0" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" class="crayons-avatar__image" alt="" width="260" height="231"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Chen Debra&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/inside-dolphinschedulers-master-startup-process-a-source-code-walkthrough-5hll" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Sep 18&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/inside-dolphinschedulers-master-startup-process-a-source-code-walkthrough-5hll" id="article-link-4682962"&gt;
          Inside DolphinScheduler’s Master Startup Process: A Source Code Walkthrough
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apachedolphinscheduler"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apachedolphinscheduler&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/opensource"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;opensource&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/github"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;github&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/inside-dolphinschedulers-master-startup-process-a-source-code-walkthrough-5hll#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            7 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Inside DolphinScheduler’s Master Startup Process: A Source Code Walkthrough</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 18 Sep 2026 08:06:11 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/inside-dolphinschedulers-master-startup-process-a-source-code-walkthrough-5hll</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/inside-dolphinschedulers-master-startup-process-a-source-code-walkthrough-5hll</guid>
      <description>&lt;p&gt;Apache DolphinScheduler is a distributed, highly extensible, and visual workflow scheduler designed for creating, scheduling, and monitoring enterprise-scale big data workflows. As a core scheduling component, the Master node receives task commands, schedules DAG workflows, dispatches tasks to Workers, and handles fault tolerance and high availability across the cluster.&lt;/p&gt;

&lt;p&gt;This article walks through the Master startup process in DolphinScheduler 3.2.0 from the source-code level. By tracing the startup sequence and examining the key modules involved, you can quickly build an understanding of how the Master service initializes and how its scheduling engine works.&lt;/p&gt;

&lt;h2&gt;
  
  
  1. How a Manually Triggered Workflow Starts
&lt;/h2&gt;

&lt;p&gt;When you click &lt;strong&gt;“Run”&lt;/strong&gt; in the Web UI to manually trigger a workflow, the workflow does not start executing immediately. Instead, DolphinScheduler first writes a Command to the database.&lt;/p&gt;

&lt;p&gt;The entry point is &lt;code&gt;ExecutorController&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="nd"&gt;@PostMapping&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;value&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="s"&gt;"start-process-instance"&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="nc"&gt;Result&lt;/span&gt; &lt;span class="nf"&gt;startProcessInstance&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nd"&gt;@RequestBody&lt;/span&gt; &lt;span class="nc"&gt;StartProcessInstanceCommand&lt;/span&gt; &lt;span class="n"&gt;command&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="c1"&gt;// Parameter validation and permission checks...&lt;/span&gt;
    &lt;span class="n"&gt;executorService&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;execProcessInstance&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;command&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="k"&gt;return&lt;/span&gt; &lt;span class="nc"&gt;Result&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;success&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The subsequent call chain is:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;ExecutorServiceImpl#execProcessInstance(...)
→ createCommand(...)
→ CommandServiceImpl#createCommand(Command)
→ CommandMapper#insert(Command)
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;At this stage, DolphinScheduler only inserts a record into the &lt;code&gt;t_ds_command&lt;/code&gt; table. After startup, the Master continuously polls this table and consumes pending commands, after which it actually restores and executes the corresponding workflow instances.&lt;/p&gt;

&lt;h2&gt;
  
  
  2. MasterServer Startup Entry Point and Overall Flow
&lt;/h2&gt;

&lt;p&gt;The Master startup entry point is the &lt;code&gt;org.apache.dolphinscheduler.server.master.MasterServer&lt;/code&gt; class. Its &lt;code&gt;@PostConstruct&lt;/code&gt;-annotated &lt;code&gt;run()&lt;/code&gt; method initializes and starts the major components in sequence:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="nd"&gt;@PostConstruct&lt;/span&gt;
&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;run&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="kd"&gt;throws&lt;/span&gt; &lt;span class="nc"&gt;SchedulerException&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="c1"&gt;// 1. Start the RPC services (Server + Client)&lt;/span&gt;
    &lt;span class="n"&gt;masterRPCServer&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
    &lt;span class="n"&gt;masterRpcClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// 2. Load Task plugins&lt;/span&gt;
    &lt;span class="n"&gt;taskPluginManager&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;loadPlugin&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// 3. Start the registry client (register the Master and monitor cluster changes)&lt;/span&gt;
    &lt;span class="n"&gt;masterRegistryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
    &lt;span class="n"&gt;masterRegistryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;setRegistryStoppable&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;this&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// 4. Start the core scheduling engine&lt;/span&gt;
    &lt;span class="n"&gt;masterSchedulerBootstrap&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// 5. Start the asynchronous event processing service&lt;/span&gt;
    &lt;span class="n"&gt;eventExecuteService&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// 6. Start the failover thread&lt;/span&gt;
    &lt;span class="n"&gt;failoverExecuteThread&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// 7. Start the Quartz scheduler&lt;/span&gt;
    &lt;span class="n"&gt;schedulerApi&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="err"&gt;​&lt;/span&gt;
    &lt;span class="c1"&gt;// Add a JVM shutdown hook&lt;/span&gt;
    &lt;span class="nc"&gt;Runtime&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getRuntime&lt;/span&gt;&lt;span class="o"&gt;().&lt;/span&gt;&lt;span class="na"&gt;addShutdownHook&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;Thread&lt;/span&gt;&lt;span class="o"&gt;(()&lt;/span&gt; &lt;span class="o"&gt;-&amp;gt;&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="o"&gt;(!&lt;/span&gt;&lt;span class="nc"&gt;ServerLifeCycleManager&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;isStopped&lt;/span&gt;&lt;span class="o"&gt;())&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
            &lt;span class="n"&gt;close&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"MasterServer shutdownHook"&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="o"&gt;}));&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Let's walk through each step to understand the implementation details and the design behind them.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. Starting the Master RPC Services
&lt;/h2&gt;

&lt;p&gt;Inside DolphinScheduler, the Master uses Netty to implement RPC communication. The RPC layer consists of an &lt;strong&gt;RPC Server&lt;/strong&gt;, which receives heartbeats and status updates from Workers, and an &lt;strong&gt;RPC Client&lt;/strong&gt;, which sends task commands to Workers.&lt;/p&gt;

&lt;h3&gt;
  
  
  3.1 Starting the RPC Server
&lt;/h3&gt;

&lt;p&gt;The &lt;code&gt;MasterRPCServer.start()&lt;/code&gt; method initializes the RPC server as follows:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;start&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;log&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;info&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Starting Master RPC Server..."&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="nc"&gt;NettyServerConfig&lt;/span&gt; &lt;span class="n"&gt;serverConfig&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;masterConfig&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getMasterRpcServerConfig&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
    &lt;span class="n"&gt;serverConfig&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;setListenPort&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;masterConfig&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getListenPort&lt;/span&gt;&lt;span class="o"&gt;());&lt;/span&gt;
    &lt;span class="k"&gt;this&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;nettyRemotingServer&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;NettyRemotingServer&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;serverConfig&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="c1"&gt;// Register all MasterRpcProcessors to handle incoming messages&lt;/span&gt;
    &lt;span class="k"&gt;for&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;MasterRpcProcessor&lt;/span&gt; &lt;span class="n"&gt;processor&lt;/span&gt; &lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="n"&gt;masterRpcProcessors&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="k"&gt;this&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;nettyRemotingServer&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;registerProcessor&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;processor&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="k"&gt;this&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;nettyRemotingServer&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
    &lt;span class="n"&gt;log&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;info&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Started Master RPC Server..."&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;When &lt;code&gt;NettyRemotingServer&lt;/code&gt; is constructed, DolphinScheduler selects either Epoll or NIO based on the runtime environment and initializes the &lt;code&gt;bossGroup&lt;/code&gt; and &lt;code&gt;workerGroup&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="nf"&gt;NettyRemotingServer&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;NettyServerConfig&lt;/span&gt; &lt;span class="n"&gt;config&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;Epoll&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;isAvailable&lt;/span&gt;&lt;span class="o"&gt;())&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;bossGroup&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;EpollEventLoopGroup&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;bossFactory&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="n"&gt;workGroup&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;EpollEventLoopGroup&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;config&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getWorkerThread&lt;/span&gt;&lt;span class="o"&gt;(),&lt;/span&gt; &lt;span class="n"&gt;workerFactory&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt; &lt;span class="k"&gt;else&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;bossGroup&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;NioEventLoopGroup&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;bossFactory&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="n"&gt;workGroup&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;NioEventLoopGroup&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;config&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getWorkerThread&lt;/span&gt;&lt;span class="o"&gt;(),&lt;/span&gt; &lt;span class="n"&gt;workerFactory&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="c1"&gt;// Initialize ServerBootstrap and register handlers&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;During &lt;code&gt;start()&lt;/code&gt;, &lt;code&gt;ServerBootstrap.bind(...)&lt;/code&gt; binds the server to the configured port. The default RPC port is &lt;strong&gt;5678&lt;/strong&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="n"&gt;serverBootstrap&lt;/span&gt;
    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;group&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;bossGroup&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;workGroup&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;channel&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;NettyUtils&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getServerSocketChannelClass&lt;/span&gt;&lt;span class="o"&gt;())&lt;/span&gt;
    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;childHandler&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;ChannelInitializer&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;SocketChannel&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="kd"&gt;protected&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;initChannel&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;SocketChannel&lt;/span&gt; &lt;span class="n"&gt;ch&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
            &lt;span class="n"&gt;initNettyChannel&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;ch&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="o"&gt;});&lt;/span&gt;
&lt;span class="nc"&gt;ChannelFuture&lt;/span&gt; &lt;span class="n"&gt;future&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;serverBootstrap&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;bind&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;serverConfig&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getListenPort&lt;/span&gt;&lt;span class="o"&gt;()).&lt;/span&gt;&lt;span class="na"&gt;sync&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Once the server successfully binds to the port, the Master can receive RPC requests from Workers and other components.&lt;/p&gt;

&lt;h3&gt;
  
  
  3.2 Starting the RPC Client
&lt;/h3&gt;

&lt;p&gt;&lt;code&gt;MasterRpcClient.start()&lt;/code&gt; initializes the &lt;code&gt;NettyRemotingClient&lt;/code&gt;, but it does not proactively establish connections to Workers:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;start&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;client&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;NettyRemotingClient&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;masterConfig&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getMasterRpcClientConfig&lt;/span&gt;&lt;span class="o"&gt;());&lt;/span&gt;
    &lt;span class="n"&gt;log&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;info&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Success initialized MasterRPCClient..."&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The &lt;code&gt;NettyRemotingClient&lt;/code&gt; constructor also selects Epoll or NIO depending on the environment. It initializes the &lt;code&gt;workerGroup&lt;/code&gt;, &lt;code&gt;callbackExecutor&lt;/code&gt;, and &lt;code&gt;responseFutureExecutor&lt;/code&gt;, and starts the response-future scanning mechanism:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="n"&gt;bootstrap&lt;/span&gt;
    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;group&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;workerGroup&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;channel&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;NettyUtils&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getSocketChannelClass&lt;/span&gt;&lt;span class="o"&gt;())&lt;/span&gt;
    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;handler&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;ChannelInitializer&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;SocketChannel&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;initChannel&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;SocketChannel&lt;/span&gt; &lt;span class="n"&gt;ch&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
            &lt;span class="n"&gt;ch&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;pipeline&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt;
              &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;addLast&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;IdleStateHandler&lt;/span&gt;&lt;span class="o"&gt;(...))&lt;/span&gt;
              &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;addLast&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;NettyDecoder&lt;/span&gt;&lt;span class="o"&gt;(),&lt;/span&gt; &lt;span class="n"&gt;clientHandler&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;encoder&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="o"&gt;});&lt;/span&gt;
&lt;span class="n"&gt;responseFutureExecutor&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;scheduleWithFixedDelay&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nl"&gt;ResponseFuture:&lt;/span&gt;&lt;span class="o"&gt;:&lt;/span&gt;&lt;span class="n"&gt;scanFutureTable&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="mi"&gt;0&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="nc"&gt;TimeUnit&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;SECONDS&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;In other words, the Master initializes the client-side RPC infrastructure during startup. Actual connections to Workers are established when the Master needs to dispatch tasks.&lt;/p&gt;

&lt;h2&gt;
  
  
  4. Plugin Loading
&lt;/h2&gt;

&lt;p&gt;DolphinScheduler uses Java SPI to dynamically load Task plugins, enabling integration with multiple execution engines such as Hive, Spark, and Flink.&lt;/p&gt;

&lt;p&gt;The implementation of &lt;code&gt;TaskPluginManager.loadPlugin()&lt;/code&gt; is:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;loadPlugin&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="nc"&gt;PrioritySPIFactory&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;TaskChannelFactory&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="n"&gt;factory&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;PrioritySPIFactory&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&amp;gt;(&lt;/span&gt;&lt;span class="nc"&gt;TaskChannelFactory&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;class&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="k"&gt;for&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;Map&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;Entry&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;String&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="nc"&gt;TaskChannelFactory&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="n"&gt;entry&lt;/span&gt; &lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="n"&gt;factory&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getSPIMap&lt;/span&gt;&lt;span class="o"&gt;().&lt;/span&gt;&lt;span class="na"&gt;entrySet&lt;/span&gt;&lt;span class="o"&gt;())&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="nc"&gt;String&lt;/span&gt; &lt;span class="n"&gt;name&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;entry&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getKey&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
        &lt;span class="nc"&gt;TaskChannelFactory&lt;/span&gt; &lt;span class="n"&gt;plugin&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;entry&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getValue&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
        &lt;span class="n"&gt;taskChannelFactoryMap&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;put&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;name&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;plugin&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="n"&gt;taskChannelMap&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;put&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;name&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;plugin&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;create&lt;/span&gt;&lt;span class="o"&gt;());&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Internally, &lt;code&gt;PrioritySPIFactory&lt;/code&gt; uses &lt;code&gt;ServiceLoader.load(spiClass)&lt;/code&gt; to scan implementations registered under &lt;code&gt;META-INF/services&lt;/code&gt; and handles conflicts between implementations with the same name:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="k"&gt;for&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="no"&gt;T&lt;/span&gt; &lt;span class="n"&gt;impl&lt;/span&gt; &lt;span class="o"&gt;:&lt;/span&gt; &lt;span class="nc"&gt;ServiceLoader&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;load&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;spiClass&lt;/span&gt;&lt;span class="o"&gt;))&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="nc"&gt;String&lt;/span&gt; &lt;span class="n"&gt;key&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;impl&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getIdentify&lt;/span&gt;&lt;span class="o"&gt;().&lt;/span&gt;&lt;span class="na"&gt;getName&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;map&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;containsKey&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;key&lt;/span&gt;&lt;span class="o"&gt;))&lt;/span&gt; &lt;span class="n"&gt;resolveConflict&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;impl&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="k"&gt;else&lt;/span&gt; &lt;span class="n"&gt;map&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;put&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;key&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;impl&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This SPI-based architecture allows Task-related capabilities to be extended without tightly coupling the core scheduling engine to every specific execution technology.&lt;/p&gt;

&lt;h2&gt;
  
  
  5. Registry Client Initialization and Heartbeat Maintenance
&lt;/h2&gt;

&lt;p&gt;The Master communicates with ZooKeeper, or another supported registry, through &lt;code&gt;masterRegistryClient&lt;/code&gt;. This component handles several key responsibilities:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;
&lt;strong&gt;Register the current Master node:&lt;/strong&gt; Create an ephemeral node under &lt;code&gt;/dolphinscheduler/master&lt;/code&gt; and store heartbeat information.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Maintain heartbeats:&lt;/strong&gt; Periodically update node information to prevent the cluster from treating the Master as unavailable.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Monitor cluster changes:&lt;/strong&gt; Subscribe to &lt;code&gt;/dolphinscheduler/servers&lt;/code&gt; so the Master can dynamically detect Master and Worker nodes joining or leaving the cluster.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;The startup logic is:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;start&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="k"&gt;this&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;masterHeartBeatTask&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;MasterHeartBeatTask&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;masterConfig&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;registryClient&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="n"&gt;registry&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt; &lt;span class="c1"&gt;// Register the Master and start the heartbeat&lt;/span&gt;
    &lt;span class="n"&gt;registryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;addConnectionStateListener&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;MasterConnectionStateListener&lt;/span&gt;&lt;span class="o"&gt;(...));&lt;/span&gt;
    &lt;span class="n"&gt;registryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;subscribe&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;RegistryNodeType&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;ALL_SERVERS&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getRegistryPath&lt;/span&gt;&lt;span class="o"&gt;(),&lt;/span&gt; &lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;MasterRegistryDataListener&lt;/span&gt;&lt;span class="o"&gt;());&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The core registration logic is:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;registry&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;registryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;remove&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;masterPath&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="n"&gt;registryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;persistEphemeral&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;masterPath&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="nc"&gt;JSONUtils&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;toJsonString&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;heartbeat&lt;/span&gt;&lt;span class="o"&gt;));&lt;/span&gt;
    &lt;span class="k"&gt;while&lt;/span&gt; &lt;span class="o"&gt;(!&lt;/span&gt;&lt;span class="n"&gt;registryClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;checkNodeExists&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;host&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="no"&gt;MASTER&lt;/span&gt;&lt;span class="o"&gt;))&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="nc"&gt;ThreadUtils&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;sleep&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;3000&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="n"&gt;masterHeartBeatTask&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The &lt;code&gt;MasterRegistryDataListener&lt;/code&gt; handles registry events through &lt;code&gt;handleMasterEvent()&lt;/code&gt; and &lt;code&gt;handleWorkerEvent()&lt;/code&gt;, triggering failover processing or resource cleanup when cluster membership changes.&lt;/p&gt;

&lt;h2&gt;
  
  
  6. Starting the Core Scheduling Engine
&lt;/h2&gt;

&lt;p&gt;The core scheduling engine is started by &lt;code&gt;MasterSchedulerBootstrap&lt;/code&gt;, which brings together three major components:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Command recovery:&lt;/strong&gt; Queries all pending Commands, creates the corresponding &lt;code&gt;WorkflowExecuteRunnable&lt;/code&gt; instances, and adds them to the execution cache and event queue.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Event loop:&lt;/strong&gt; &lt;code&gt;WorkflowEventLooper&lt;/code&gt; continuously consumes events from &lt;code&gt;workflowEventQueue&lt;/code&gt; and invokes the appropriate handler based on the event type.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Task executor:&lt;/strong&gt; &lt;code&gt;MasterTaskExecutorBootstrap&lt;/code&gt; starts the thread pools and queues responsible for consuming pending tasks and dispatching them to Workers through RPC.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The startup sequence is:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;synchronized&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;start&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="kd"&gt;super&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;              &lt;span class="c1"&gt;// part1: recover and submit Commands&lt;/span&gt;
    &lt;span class="n"&gt;workflowEventLooper&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt; &lt;span class="c1"&gt;// part2: start the event loop&lt;/span&gt;
    &lt;span class="n"&gt;masterTaskExecutorBootstrap&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt; &lt;span class="c1"&gt;// part3: dispatch tasks&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  6.1 Recovering Commands
&lt;/h3&gt;

&lt;p&gt;The recovery process can be illustrated as follows:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="nc"&gt;List&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;Command&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="n"&gt;commands&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;findCommands&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="n"&gt;commands&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;parallelStream&lt;/span&gt;&lt;span class="o"&gt;().&lt;/span&gt;&lt;span class="na"&gt;forEach&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;cmd&lt;/span&gt; &lt;span class="o"&gt;-&amp;gt;&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="nc"&gt;Optional&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;WorkflowExecuteRunnable&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="n"&gt;opt&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;factory&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;create&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;cmd&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;opt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;isPresent&lt;/span&gt;&lt;span class="o"&gt;())&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;cacheManager&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;cache&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;runnable&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="n"&gt;workflowEventQueue&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;addEvent&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="k"&gt;new&lt;/span&gt; &lt;span class="nc"&gt;WorkflowEvent&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="no"&gt;START_WORKFLOW&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="o"&gt;));&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt;
&lt;span class="o"&gt;});&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The Master retrieves pending Commands, creates the corresponding workflow execution objects, caches them, and then pushes &lt;code&gt;START_WORKFLOW&lt;/code&gt; events into the workflow event queue.&lt;/p&gt;

&lt;p&gt;This is the key transition from a command stored in the database to an actual workflow execution process.&lt;/p&gt;

&lt;h3&gt;
  
  
  6.2 The Event Loop
&lt;/h3&gt;

&lt;p&gt;&lt;code&gt;WorkflowEventLooper&lt;/code&gt; implements &lt;code&gt;Runnable&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;run&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="k"&gt;while&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="no"&gt;RUNNING&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
        &lt;span class="nc"&gt;WorkflowEvent&lt;/span&gt; &lt;span class="n"&gt;event&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;queue&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;poolEvent&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
        &lt;span class="k"&gt;try&lt;/span&gt; &lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;MDCAutoClosableContext&lt;/span&gt; &lt;span class="n"&gt;ctx&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;setWorkflowIdMDC&lt;/span&gt;&lt;span class="o"&gt;(...))&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
            &lt;span class="n"&gt;handlerMap&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;get&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;event&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;getType&lt;/span&gt;&lt;span class="o"&gt;()).&lt;/span&gt;&lt;span class="na"&gt;handle&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;event&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
        &lt;span class="o"&gt;}&lt;/span&gt;
    &lt;span class="o"&gt;}&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;When a &lt;code&gt;START_WORKFLOW&lt;/code&gt; event is received, it is routed to &lt;code&gt;WorkflowStartHandler&lt;/code&gt;, which invokes &lt;code&gt;WorkflowExecuteRunnable.call()&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="nc"&gt;WorkflowStartStatus&lt;/span&gt; &lt;span class="nf"&gt;startWorkflow&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;initTaskQueue&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;     &lt;span class="c1"&gt;// Initialize the DAG task queue&lt;/span&gt;
    &lt;span class="n"&gt;submitPostNode&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="kc"&gt;null&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt; &lt;span class="c1"&gt;// Submit the first node&lt;/span&gt;
    &lt;span class="k"&gt;return&lt;/span&gt; &lt;span class="no"&gt;SUCCESS&lt;/span&gt;&lt;span class="o"&gt;;&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;At this point, the workflow execution engine begins processing the DAG and submitting its executable tasks.&lt;/p&gt;

&lt;h3&gt;
  
  
  6.3 Task Dispatch
&lt;/h3&gt;

&lt;p&gt;&lt;code&gt;MasterTaskExecutorBootstrap&lt;/code&gt; starts three processing loops:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="n"&gt;globalTaskDispatchLooper&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="n"&gt;masterDelayTaskLooper&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="n"&gt;asyncMasterTaskDelayLooper&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Among them, &lt;code&gt;globalTaskDispatchLooper&lt;/code&gt; retrieves &lt;code&gt;DefaultTaskExecuteRunnable&lt;/code&gt; instances from &lt;code&gt;globalTaskDispatchWaitingQueue&lt;/code&gt; and invokes &lt;code&gt;taskDispatcher.dispatch()&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="nc"&gt;Message&lt;/span&gt; &lt;span class="n"&gt;msg&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="n"&gt;taskDispatchRequest&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;convert2Command&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="n"&gt;masterRpcClient&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;sendSyncCommand&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;host&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;msg&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="n"&gt;timeout&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The task dispatch request is converted into an RPC message and sent synchronously to the target Worker.&lt;/p&gt;

&lt;p&gt;This is the point where the Master moves from workflow-level orchestration to actual task execution on a Worker node.&lt;/p&gt;

&lt;h2&gt;
  
  
  7. Event Processing Service
&lt;/h2&gt;

&lt;p&gt;Event processing is divided into two main categories:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Workflow events:&lt;/strong&gt; Events such as task status changes and workflow blocking.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Stream events:&lt;/strong&gt; Custom events associated with streaming tasks.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;code&gt;EventExecuteService.start()&lt;/code&gt; starts the corresponding thread pools and continuously consumes events from the &lt;code&gt;stateEvents&lt;/code&gt; and &lt;code&gt;taskEvents&lt;/code&gt; queues:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="n"&gt;workflowEventHandler&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt; &lt;span class="c1"&gt;// Submit to workflowExecuteThreadPool&lt;/span&gt;
&lt;span class="n"&gt;streamTaskEventHandler&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt; &lt;span class="c1"&gt;// Submit to streamTaskExecuteThreadPool&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;By processing events asynchronously through dedicated thread pools, the Master can separate event production from event handling and avoid coupling the execution of different types of events too tightly.&lt;/p&gt;

&lt;h2&gt;
  
  
  8. Failover Processing
&lt;/h2&gt;

&lt;p&gt;The &lt;code&gt;failoverExecuteThread&lt;/code&gt; periodically checks the health of Master and Worker nodes. When a node becomes unavailable, DolphinScheduler handles two types of failover scenarios: &lt;strong&gt;Master Failover&lt;/strong&gt; and &lt;strong&gt;Worker Failover&lt;/strong&gt;.&lt;/p&gt;

&lt;h3&gt;
  
  
  8.1 Master Failover
&lt;/h3&gt;

&lt;p&gt;When a Master node goes down, the registry removes its node. Other Master nodes detect the change and trigger failover processing:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="n"&gt;failoverService&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;failoverServerWhenDown&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;serverHost&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="no"&gt;MASTER&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;code&gt;doFailoverMaster&lt;/code&gt; then:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Queries the ProcessInstances that require failover.&lt;/li&gt;
&lt;li&gt;Calls &lt;code&gt;processService.processNeedFailoverProcessInstances(processInstance)&lt;/code&gt; for each instance.&lt;/li&gt;
&lt;li&gt;Writes the required execution commands back to &lt;code&gt;t_ds_command&lt;/code&gt;.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;This allows another available Master to take over the affected workflow instances and continue the scheduling process.&lt;/p&gt;

&lt;h3&gt;
  
  
  8.2 Worker Failover
&lt;/h3&gt;

&lt;p&gt;When a Worker node goes down:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="n"&gt;failoverService&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;failoverServerWhenDown&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="n"&gt;workerHost&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="no"&gt;WORKER&lt;/span&gt;&lt;span class="o"&gt;);&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;code&gt;failoverWorker&lt;/code&gt; queries the &lt;code&gt;TaskInstance&lt;/code&gt; records for tasks that were running on the failed Worker. For each unfinished task, it:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Sets &lt;code&gt;state = NEED_FAULT_TOLERANCE&lt;/code&gt;.&lt;/li&gt;
&lt;li&gt;Updates the database through &lt;code&gt;taskInstanceDao.upsert&lt;/code&gt;.&lt;/li&gt;
&lt;li&gt;Submits a &lt;code&gt;TaskStateEvent&lt;/code&gt; to &lt;code&gt;workflowExecuteThreadPool&lt;/code&gt;, allowing the task to be dispatched again to another available Worker.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Through this mechanism, task execution can recover from Worker failures without requiring the workflow itself to be restarted from scratch.&lt;/p&gt;

&lt;h2&gt;
  
  
  9. Starting the Quartz Scheduler
&lt;/h2&gt;

&lt;p&gt;In addition to manually triggered workflows, the Master uses Quartz to support scheduled execution.&lt;/p&gt;

&lt;p&gt;&lt;code&gt;SchedulerApi&lt;/code&gt; injects an &lt;code&gt;org.quartz.Scheduler&lt;/code&gt; instance:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="nd"&gt;@Override&lt;/span&gt;
&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kt"&gt;void&lt;/span&gt; &lt;span class="nf"&gt;start&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt; &lt;span class="kd"&gt;throws&lt;/span&gt; &lt;span class="nc"&gt;SchedulerException&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;scheduler&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;start&lt;/span&gt;&lt;span class="o"&gt;();&lt;/span&gt;
&lt;span class="o"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The Quartz Scheduler is used to trigger scheduled jobs periodically, including tasks such as workflow dependency handling and scheduled workflow execution.&lt;/p&gt;

&lt;p&gt;This complements the command-driven execution path used for manually triggered workflows.&lt;/p&gt;

&lt;h2&gt;
  
  
  10. Summary and Key Takeaways
&lt;/h2&gt;

&lt;p&gt;This article has walked through the complete startup and execution flow of the Master node in DolphinScheduler 3.2.0, from manually writing a Command to the database, to &lt;code&gt;MasterServer&lt;/code&gt; initialization, RPC service setup, plugin loading, registry integration, core scheduling engine startup, event processing, failover handling, and Quartz-based scheduled execution.&lt;/p&gt;

&lt;p&gt;At the architectural level, the Master demonstrates a modular and decoupled design:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Network communication:&lt;/strong&gt; RPC communication is implemented with Netty.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Plugin extensibility:&lt;/strong&gt; Java SPI is used to dynamically load TaskChannel implementations.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;High availability:&lt;/strong&gt; Registry-based heartbeats and the Failover service provide fault-tolerance capabilities.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Scheduling engine:&lt;/strong&gt; Parallel processing and an event-driven architecture work together to drive workflow execution.&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Scheduled execution:&lt;/strong&gt; Quartz provides additional scheduling capabilities for time-based tasks.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;At the source-code level, the core logic is built around database operations, queues, event processing, and RPC messaging. Developers can continue tracing the call chain to explore the implementation details that are not covered here, and combine the source code with the DolphinScheduler community documentation for a deeper understanding of how the Master works.&lt;/p&gt;

&lt;p&gt;Once you follow the startup path from &lt;code&gt;MasterServer.run()&lt;/code&gt; through command recovery, event handling, task dispatch, and failover, the Master is no longer a black box: it becomes a clearly connected execution pipeline that turns workflow commands into reliable task execution across the cluster.&lt;/p&gt;

</description>
      <category>apachedolphinscheduler</category>
      <category>opensource</category>
      <category>github</category>
      <category>datascience</category>
    </item>
    <item>
      <title>🚀 Apache DolphinScheduler 3.4.3 is here! Enhanced security, faster queries, stronger recovery, and more. Explore the latest release! #ApacheDolphinScheduler #DataOrchestration #OpenSource</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 18 Sep 2026 07:48:30 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-is-here-enhanced-security-faster-queries-stronger-recovery-and-3m7c</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-is-here-enhanced-security-faster-queries-stronger-recovery-and-3m7c</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-released-stronger-security-stability-and-scheduling-reliability-2da6" class="crayons-story__hidden-navigation-link"&gt;Apache DolphinScheduler 3.4.3 Released: Stronger Security, Stability, and Scheduling Reliability&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" alt="chen_debra_3060b21d12b1b0 profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Chen Debra
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Chen Debra
                
                
              
              &lt;div id="story-author-preview-content-4682776" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/chen_debra_3060b21d12b1b0" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Chen Debra&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-released-stronger-security-stability-and-scheduling-reliability-2da6" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Sep 18&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-released-stronger-security-stability-and-scheduling-reliability-2da6" id="article-link-4682776"&gt;
          Apache DolphinScheduler 3.4.3 Released: Stronger Security, Stability, and Scheduling Reliability
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apachedolphinecheduler"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apachedolphinecheduler&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/java"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;java&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/github"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;github&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/bigdata"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;bigdata&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-released-stronger-security-stability-and-scheduling-reliability-2da6#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            5 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Apache DolphinScheduler 3.4.3 Released: Stronger Security, Stability, and Scheduling Reliability</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 18 Sep 2026 07:48:09 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-released-stronger-security-stability-and-scheduling-reliability-2da6</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinscheduler-343-released-stronger-security-stability-and-scheduling-reliability-2da6</guid>
      <description>&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F8hx7to1qggn1jp8xr2b0.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F8hx7to1qggn1jp8xr2b0.jpg" width="800" height="586"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Apache DolphinScheduler 3.4.3 has officially been released. This release delivers continued improvements across scheduling, permissions, security, performance, and stability, with a focus on real-world production scenarios such as task and workflow instance queries, API access control, sensitive information protection, and task failure recovery. In addition, new missed-scheduling recovery strategies are planned for Apache DolphinScheduler 3.5.0.&lt;/p&gt;

&lt;p&gt;As an open-source data orchestration platform designed for complex data workflow scenarios, Apache DolphinScheduler continues to strengthen its scheduling and orchestration capabilities through community collaboration. Version 3.4.3 further improves reliability in areas such as scheduling recovery, enterprise-grade access control, and large-scale task execution, providing a stronger foundation for stable operation in production environments.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://github.com/apache/dolphinscheduler/releases/tag/3.4.3" rel="noopener noreferrer"&gt;View the Apache DolphinScheduler 3.4.3 Release&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Stronger Access Control for Improved Platform Security
&lt;/h2&gt;

&lt;p&gt;One of the key areas of improvement in 3.4.3 is access control.&lt;/p&gt;

&lt;p&gt;Enterprise data platforms typically involve multiple users, projects, data sources, and workflows. Clear and well-defined permission boundaries are therefore essential to maintaining a secure operating environment.&lt;/p&gt;

&lt;p&gt;Version 3.4.3 further improves access control across multiple APIs, covering scenarios such as user list access, data source authorization lists, cluster queries, workflow modifications, and sub-workflow references. Related permissions have also been further standardized and refined.&lt;/p&gt;

&lt;p&gt;The release also further restricts access to audit logs for non-administrator users, ensuring that users can only access audit information within the scope of their permissions. In addition, task-level data source access now includes corresponding permission checks, providing stronger controls over data access.&lt;/p&gt;

&lt;p&gt;Together, these improvements establish clearer permission boundaries across users, projects, workflows, and data sources, providing more comprehensive security for enterprise environments with multiple users and collaborative workflows.&lt;/p&gt;

&lt;h2&gt;
  
  
  Enhanced Protection for Sensitive Information
&lt;/h2&gt;

&lt;p&gt;Beyond access control, Apache DolphinScheduler 3.4.3 also introduces targeted improvements to sensitive information protection.&lt;/p&gt;

&lt;p&gt;The release removes plaintext passwords from Worker logs and prevents raw passwords from being exposed through parameter validation error messages. These changes reduce the risk of sensitive information leakage in both log output and error messages.&lt;/p&gt;

&lt;p&gt;The release also fixes a Sonar Token exposure issue and removes some unused code, further improving the project's overall security and engineering quality.&lt;/p&gt;

&lt;h2&gt;
  
  
  More Efficient Instance Queries at Scale
&lt;/h2&gt;

&lt;p&gt;As the number of workflows and task instances continues to grow, querying task and workflow instances has become a frequent operation in production environments.&lt;/p&gt;

&lt;p&gt;To address this scenario, version 3.4.3 introduces several query optimizations. New indexes, including &lt;code&gt;idx_project_submit_time&lt;/code&gt; and &lt;code&gt;idx_project_start_time&lt;/code&gt;, have been added for task instance and workflow instance queries. The Mapper query logic has also been optimized to exclude unnecessary large text fields from list queries.&lt;/p&gt;

&lt;p&gt;These improvements reduce data retrieval during queries and help lower database load, improving query efficiency and the overall user experience in environments with large numbers of task instances.&lt;/p&gt;

&lt;h2&gt;
  
  
  Improved Task Failure Recovery and Retry Handling
&lt;/h2&gt;

&lt;p&gt;The ability to recover from task failures directly affects the continuity of data processing. Version 3.4.3 includes multiple fixes covering task failures, retries, and workflow recovery.&lt;/p&gt;

&lt;p&gt;The release adds application termination handling during task failover, preventing related applications from continuing to run after a task failure. It also adjusts retry time calculation so that the next execution is scheduled based on &lt;code&gt;endTime + retryInterval&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;For scenarios where failed tasks are recreated, 3.4.3 fixes an issue where runtime states were not properly reset. During workflow recovery, the release also ensures that forced-success states are correctly preserved.&lt;/p&gt;

&lt;p&gt;In addition, this version fixes issues where Worker Group parameters were not applied as specified when rerunning workflows, as well as abnormal TaskGroup slot handling when tasks were paused or terminated. These fixes further strengthen task recovery in exceptional execution scenarios.&lt;/p&gt;

&lt;h2&gt;
  
  
  More Robust Workflow Dependencies and Execution Logic
&lt;/h2&gt;

&lt;p&gt;At the workflow execution layer, version 3.4.3 includes multiple fixes related to DAGs and dependent tasks.&lt;/p&gt;

&lt;p&gt;The release fixes an issue where &lt;code&gt;DAG.addEdge&lt;/code&gt; could accept an edge that creates a cycle in path-convergence scenarios, further strengthening DAG validation and ensuring workflow correctness.&lt;/p&gt;

&lt;p&gt;For dependent tasks, 3.4.3 fixes an issue with parsing &lt;code&gt;ALL&lt;/code&gt;-type dependencies. It also resolves a problem where a task instance could fail while the workflow instance remained in a running state when using the Continue strategy, enabling workflow status to more accurately reflect actual execution states under complex dependency conditions.&lt;/p&gt;

&lt;p&gt;In addition, the release addresses issues related to SQL task data source field persistence, DataX tasks reading Job Definitions from associated resource files, and S3 resource lists containing more than 1,000 records.&lt;/p&gt;

&lt;h2&gt;
  
  
  Continued Improvements to Kubernetes, Alerts, and Other Capabilities
&lt;/h2&gt;

&lt;p&gt;For cloud-native deployments, version 3.4.3 fixes an issue in the Helm Chart that could result in duplicate &lt;code&gt;app.kubernetes.io/name&lt;/code&gt; labels in ConfigMaps and updates the MySQL Helm Chart version.&lt;/p&gt;

&lt;p&gt;Alerting capabilities have also been improved. These changes include removing redundant plugin definition table checks during AlertServer startup and adjusting the behavior of Alert Script test notifications. The release also fixes an issue that could cause Kubernetes Alert HTTP tests to fail.&lt;/p&gt;

&lt;p&gt;In addition, the release removes some deprecated APIs and unused code, while further improving documentation related to parameter precedence, upgrades, data sources, and configuration. These changes contribute to better maintainability and a smoother user experience.&lt;/p&gt;

&lt;h2&gt;
  
  
  Missed-Scheduling Recovery Strategies Planned for 3.5.0
&lt;/h2&gt;

&lt;p&gt;Beyond the improvements included in this release, the community continues to advance Apache DolphinScheduler's scheduling capabilities.&lt;/p&gt;

&lt;p&gt;The missed-scheduling recovery strategy proposed in DSIP-107 has completed the relevant implementation and has been merged into the development branch. It is planned for inclusion in version 3.5.0.&lt;/p&gt;

&lt;p&gt;The feature supports three strategies for handling missed schedules:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;code&gt;SKIP_MISSED&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;&lt;code&gt;FIRE_ONCE_NOW&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;&lt;code&gt;FIRE_ALL_MISSED&lt;/code&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These strategies provide different ways to handle schedules that were missed, giving users greater control over how scheduled workflows should be recovered after interruptions or other exceptional circumstances.&lt;/p&gt;

&lt;h2&gt;
  
  
  Continuously Strengthening Enterprise-Grade Data Orchestration
&lt;/h2&gt;

&lt;p&gt;From missed-scheduling recovery and access control to query performance and task failure recovery, Apache DolphinScheduler 3.4.3 focuses closely on the challenges encountered in real-world production environments.&lt;/p&gt;

&lt;p&gt;Rather than simply adding new features, this release continues to improve the core qualities required for long-running data workflows, including &lt;strong&gt;scheduling reliability, security, query performance, and task stability&lt;/strong&gt;. The upcoming missed-scheduling recovery strategies will further strengthen scheduling behavior in exceptional scenarios. More granular access controls establish stronger security boundaries for multi-user enterprise environments, while improvements to task recovery, dependency handling, and query performance further enhance platform reliability in complex production environments.&lt;/p&gt;

&lt;p&gt;For the complete list of updates, see the &lt;a href="https://github.com/apache/dolphinscheduler/releases/tag/3.4.3" rel="noopener noreferrer"&gt;Release Notes&lt;/a&gt;.&lt;/p&gt;

&lt;h2&gt;
  
  
  Thanks to Our Contributors
&lt;/h2&gt;

&lt;p&gt;The Apache DolphinScheduler 3.4.3 release would not have been possible without the continued participation and contributions of the community. We would like to thank the following 20 contributors for contributing code, submitting issues, participating in testing and reviews, and helping move the community forward:&lt;/p&gt;

&lt;p&gt;det101, wcmolin, liang-wenjie, njnu-seafish, qiuyanjun888, destinyoooo, HomminLee, ruanwenjun, kittimzhe, SbloodyS, eye-gu, vlaborie, nkuprins, nanxiuzi, SEPURI-SAI-KRISHNA, hiSandog, hellodml, zhang-arvin, nikhiln64, yan9651688&lt;/p&gt;

&lt;p&gt;Thank you to every contributor for your time, expertise, and continued support. The ongoing collaboration of the global Apache DolphinScheduler community enables the project to continuously improve its scheduling, orchestration, security, and stability capabilities while serving an ever-growing range of data engineering use cases.&lt;/p&gt;

&lt;p&gt;We invite you to try Apache DolphinScheduler 3.4.3, explore the latest improvements, and join the community to help advance the open-source data orchestration ecosystem.&lt;/p&gt;

</description>
      <category>apachedolphinecheduler</category>
      <category>java</category>
      <category>github</category>
      <category>bigdata</category>
    </item>
    <item>
      <title>Modern Data Stack was built for humans 🚀 Harness Engineering makes agent‑driven data work production‑safe.

#DataEngineering #AgenticAI #HarnessEngineering</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 11 Sep 2026 09:25:39 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/modern-data-stack-was-built-for-humans-harness-engineering-makes-agent-driven-data-work-2fpj</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/modern-data-stack-was-built-for-humans-harness-engineering-makes-agent-driven-data-work-2fpj</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/rebuilding-data-engineering-with-harness-engineering-a-new-paradigm-for-the-agent-era-enm" class="crayons-story__hidden-navigation-link"&gt;Rebuilding Data Engineering with Harness Engineering: A New Paradigm for the Agent Era&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" alt="chen_debra_3060b21d12b1b0 profile" class="crayons-avatar__image" width="260" height="231"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Chen Debra
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Chen Debra
                
                
              
              &lt;div id="story-author-preview-content-4630651" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/chen_debra_3060b21d12b1b0" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" class="crayons-avatar__image" alt="" width="260" height="231"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Chen Debra&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/rebuilding-data-engineering-with-harness-engineering-a-new-paradigm-for-the-agent-era-enm" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Sep 11&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/rebuilding-data-engineering-with-harness-engineering-a-new-paradigm-for-the-agent-era-enm" id="article-link-4630651"&gt;
          Rebuilding Data Engineering with Harness Engineering: A New Paradigm for the Agent Era
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/dataengineering"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;dataengineering&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/harnessengineering"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;harnessengineering&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/agents"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;agents&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/ai"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;ai&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/rebuilding-data-engineering-with-harness-engineering-a-new-paradigm-for-the-agent-era-enm#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            21 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Rebuilding Data Engineering with Harness Engineering: A New Paradigm for the Agent Era</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 11 Sep 2026 09:23:45 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/rebuilding-data-engineering-with-harness-engineering-a-new-paradigm-for-the-agent-era-enm</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/rebuilding-data-engineering-with-harness-engineering-a-new-paradigm-for-the-agent-era-enm</guid>
      <description>&lt;p&gt;As data platforms evolve from serving primarily human users to serving AI agents, the biggest change may not be the tools themselves, but the way data engineering is organized and delivered.&lt;/p&gt;

&lt;p&gt;Over the past decade, data engineering has gone through a major wave of specialization. Large, monolithic data platforms have gradually evolved into what is now commonly known as the Modern Data Stack, a composable ecosystem of databases, compute engines, data integration and transformation tools, governance platforms, orchestration systems, and BI solutions. This specialization has dramatically improved engineering efficiency and shifted the industry from building massive, tightly coupled systems toward assembling flexible, modular capabilities.&lt;/p&gt;

&lt;p&gt;But as Agentic AI enters the data engineering workflow, a fundamental limitation is becoming increasingly visible: &lt;strong&gt;the modern data stack was designed for people, not agents.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The next generation of data platforms will need to solve a different problem. It is no longer enough to help people operate increasingly sophisticated tools. The challenge is to enable agents to execute engineering work within the right business context, technical boundaries, security controls, and governance framework.&lt;/p&gt;

&lt;p&gt;This is where &lt;strong&gt;Harness Engineering&lt;/strong&gt; comes in.&lt;/p&gt;

&lt;p&gt;The core idea is simple: AI can generate SQL, code, pipelines, and workflows at unprecedented speed. But generating engineering artifacts is not the same as delivering reliable engineering outcomes. A production-ready system needs to make those outputs &lt;strong&gt;trusted, verifiable, controlled, recoverable, and accountable&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;In other words, the real opportunity in the Agentic AI era is not simply to make AI generate more data engineering. It is to build the engineering system that allows AI-generated work to safely reach production.&lt;/p&gt;

&lt;h2&gt;
  
  
  1. Data Platforms Are Entering the Agentic Era
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Two Different Paths, One Destination
&lt;/h3&gt;

&lt;p&gt;The evolution of major data platforms points toward the same fundamental shift.&lt;/p&gt;

&lt;p&gt;Snowflake is moving from &lt;strong&gt;Data Warehouse → Data Cloud → AI Work Interface → Enterprise Agent Platform&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Databricks is evolving from &lt;strong&gt;Data Lake → Lakehouse → Data + AI Engineering → Agent-ready Execution Platform&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Although their approaches differ, both are converging around four core capabilities:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Context. Capability. Governance. Execution.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The data platform of the future will not simply store and process enterprise data. It will increasingly serve as the infrastructure through which agents understand enterprise context and take action.&lt;/p&gt;

&lt;h3&gt;
  
  
  Snowflake: The Data Platform Becomes an AI Entry Point
&lt;/h3&gt;

&lt;p&gt;Snowflake's evolution is not simply about adding AI features to a data warehouse. It is about reorganizing data, semantics, governance, applications, and agents around AI.&lt;/p&gt;

&lt;p&gt;Its trajectory can be broadly viewed in three stages:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Cloud Data Warehouse&lt;/strong&gt;&lt;br&gt;
Storage, compute, sharing, and governance&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;AI + Data Platform&lt;/strong&gt;&lt;br&gt;
AI, data, semantics, and governance&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Agentic Enterprise Infrastructure&lt;/strong&gt;&lt;br&gt;
Coco, CoWork, Skills, and Agents&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FkTmjthEbuk5Yyclf6uBk_lC0y3Mq-oGXOEx0731iM0DznMhEi2cYYHzq9A_oEenyUMwnhXqDsBIX3AAMbezp_JJaoT9qk-ez4wTUIskGwEbYz-K3OqGP99M3XOoBXCBIEsiU3SPW6VvS7bleRBOkcBTNZKsfgXeiLQza8a1uE-dE9lSQA0yWi3dPQUFzOdPl%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FkTmjthEbuk5Yyclf6uBk_lC0y3Mq-oGXOEx0731iM0DznMhEi2cYYHzq9A_oEenyUMwnhXqDsBIX3AAMbezp_JJaoT9qk-ez4wTUIskGwEbYz-K3OqGP99M3XOoBXCBIEsiU3SPW6VvS7bleRBOkcBTNZKsfgXeiLQza8a1uE-dE9lSQA0yWi3dPQUFzOdPl%3Fpurpose%3Dfullsize" alt="Image" width="1975" height="1609"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;This shift has three important implications.&lt;/p&gt;

&lt;p&gt;First, the value of a data platform is expanding from &lt;strong&gt;managing data&lt;/strong&gt; to &lt;strong&gt;enabling agents to act on data&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Second, the enterprise AI interface is moving beyond SQL and BI toward natural language and agent-driven workflows.&lt;/p&gt;

&lt;p&gt;Third, data is increasingly being understood as &lt;strong&gt;AI context&lt;/strong&gt;, rather than simply something to be stored and queried.&lt;/p&gt;

&lt;p&gt;The question is no longer just whether an enterprise can access its data. It is whether its data can provide the context an agent needs to make the right decision and take the right action.&lt;/p&gt;

&lt;h3&gt;
  
  
  Databricks: Turning Data and AI Engineering into an Agent Runtime
&lt;/h3&gt;

&lt;p&gt;Databricks is taking a similar direction from a different starting point.&lt;/p&gt;

&lt;p&gt;The goal is not simply to add AI capabilities to the Lakehouse. Instead, data, models, notebooks, pipelines, governance, and applications are becoming part of an &lt;strong&gt;agent-ready engineering environment&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FU3r1GKIbGK38Ja0FEdBJ87Or0Y1f7so4Gv6Ujl0lS2dvMwOfujhrltjIIsgrso9F3WP7aPg4U5YV7BTh5RLR3kLoD3Ay-qFuzNoE-Sa__O6sG3Xf_58H4mK9cWdpAvm9hrWatrfT6QYOaUtl1QopXuNuPCXndPbC72eSSD2zZzM02E3ntPriEUMBYfGUQcrw%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FU3r1GKIbGK38Ja0FEdBJ87Or0Y1f7so4Gv6Ujl0lS2dvMwOfujhrltjIIsgrso9F3WP7aPg4U5YV7BTh5RLR3kLoD3Ay-qFuzNoE-Sa__O6sG3Xf_58H4mK9cWdpAvm9hrWatrfT6QYOaUtl1QopXuNuPCXndPbC72eSSD2zZzM02E3ntPriEUMBYfGUQcrw%3Fpurpose%3Dfullsize" alt="Image" width="838" height="559"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;For a data platform to be truly agent-ready, five capabilities matter:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Data must be discoverable.&lt;/li&gt;
&lt;li&gt;Schemas must be understandable.&lt;/li&gt;
&lt;li&gt;Metrics must have clear business definitions.&lt;/li&gt;
&lt;li&gt;Workflows must be executable.&lt;/li&gt;
&lt;li&gt;Actions must be auditable.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Being "agent-ready" is therefore much more than allowing a model to access data. It means making &lt;strong&gt;data, semantics, compute, governance, and execution&lt;/strong&gt; work together for agents.&lt;/p&gt;

&lt;h3&gt;
  
  
  The User Is Changing: From Humans to Agents
&lt;/h3&gt;

&lt;p&gt;The more fundamental change is happening on the user side.&lt;/p&gt;

&lt;p&gt;Traditional data platforms were built primarily for:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Data Engineers&lt;/li&gt;
&lt;li&gt;Data Analysts&lt;/li&gt;
&lt;li&gt;BI Users&lt;/li&gt;
&lt;li&gt;Platform Engineers&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The next generation will increasingly serve:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Coding Agents&lt;/li&gt;
&lt;li&gt;Data Agents&lt;/li&gt;
&lt;li&gt;Business Agents&lt;/li&gt;
&lt;li&gt;Operations Agents&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2Fl_JS1c0RnpNLiZ2rWMSDe1r_dlnAZKV-n1fBZcKLDxq6ps7OFooeaX9X_ecc5Gp4kCRGUgyz05iNDHTusq7cEhNxcn2hGwcOQ_-9FJG19EDU8wZUpTTBQ-AWofP4H-ppBg_ZPzNQhS9eANUkV1UNAkRq1a4dQZYNptLnxcPZoZHmszgHFxx2m8ke3iOD2jsB%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2Fl_JS1c0RnpNLiZ2rWMSDe1r_dlnAZKV-n1fBZcKLDxq6ps7OFooeaX9X_ecc5Gp4kCRGUgyz05iNDHTusq7cEhNxcn2hGwcOQ_-9FJG19EDU8wZUpTTBQ-AWofP4H-ppBg_ZPzNQhS9eANUkV1UNAkRq1a4dQZYNptLnxcPZoZHmszgHFxx2m8ke3iOD2jsB%3Fpurpose%3Dfullsize" alt="Image" width="3000" height="1687"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;These users have fundamentally different needs.&lt;/p&gt;

&lt;p&gt;Human users need &lt;strong&gt;UIs, documentation, and guided workflows&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Agents need &lt;strong&gt;APIs, Skills, Context, Policies, and structured Feedback&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;That difference has architectural consequences. A platform designed around human interaction cannot simply expose more APIs and expect to become agent-native. The underlying engineering model needs to change.&lt;/p&gt;

&lt;h3&gt;
  
  
  The Core Question Is Changing
&lt;/h3&gt;

&lt;p&gt;The contrast between the past decade and the next one is becoming increasingly clear.&lt;/p&gt;

&lt;p&gt;Over the past ten years, we built data platforms that helped people operate tools: write SQL, configure pipelines, build DAGs, inspect logs, and troubleshoot failed jobs.&lt;/p&gt;

&lt;p&gt;Over the next decade, the goal will be to build data engineering platforms where &lt;strong&gt;people define the outcome and agents orchestrate the work&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;The workflow shifts from:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Write SQL → Build Pipeline → Configure DAG → Monitor → Fix&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;to:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Understand Intent → Plan → Invoke Capabilities → Execute → Validate → Learn&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The central question for data platforms is therefore changing from:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;How do we make tools easier for people to operate?&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;to:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;How do we enable agents to execute engineering work within the right context and security boundaries?&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FylysFMA2-_20KQyVuKbGFfk1aH7sBS0Wr_P29pjzbypQ71Aa2HZZLReUyyGRFtCv4QWcw9CMbBaYKS5BKVAo_R-iiwJhiY9MWZV3ZE-02nS_t0hovGqPUp4GY_6WVCg9GTdcTJJW9SjgAW77NtPnrCnnyBLTpaqWoyJBSGRMi8rCcLMSj4We9RA5If4gcSJT%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FylysFMA2-_20KQyVuKbGFfk1aH7sBS0Wr_P29pjzbypQ71Aa2HZZLReUyyGRFtCv4QWcw9CMbBaYKS5BKVAo_R-iiwJhiY9MWZV3ZE-02nS_t0hovGqPUp4GY_6WVCg9GTdcTJJW9SjgAW77NtPnrCnnyBLTpaqWoyJBSGRMi8rCcLMSj4We9RA5If4gcSJT%3Fpurpose%3Dfullsize" alt="Image" width="2752" height="1536"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Looking back, this shift follows a familiar pattern.&lt;/p&gt;

&lt;p&gt;The &lt;strong&gt;Database Era&lt;/strong&gt; focused on storage, queries, and transactions.&lt;/p&gt;

&lt;p&gt;The &lt;strong&gt;Big Data Platform Era&lt;/strong&gt; focused on scale, distributed computing, and large-scale data processing.&lt;/p&gt;

&lt;p&gt;The &lt;strong&gt;Modern Data Stack Era&lt;/strong&gt; introduced cloud infrastructure, modular architectures, and standardized tools.&lt;/p&gt;

&lt;p&gt;Now, with agents becoming a new class of data platform user, we are entering the &lt;strong&gt;Agentic Data Stack Era&lt;/strong&gt;, where Context, Skills, Control, and Harness become first-class engineering concerns.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FOFNNHEqr29dEpsTr3NYXy1qUdl0nDFGD8k_Tkz5b-NN9SZGt1I18hkt2nC1Co5nJB7uxijy-84lWlTOGk7vxuLOHsEsuwmwoTpPGuH3lu4jpOFJxxiHxyHuSMfOcUDgQWAOwQSblO8PtWo5q1pxckBsO6Z2gMAaYKlDuQ8Y2oe-yaNefVFszCRNcs81XZWJD%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FOFNNHEqr29dEpsTr3NYXy1qUdl0nDFGD8k_Tkz5b-NN9SZGt1I18hkt2nC1Co5nJB7uxijy-84lWlTOGk7vxuLOHsEsuwmwoTpPGuH3lu4jpOFJxxiHxyHuSMfOcUDgQWAOwQSblO8PtWo5q1pxckBsO6Z2gMAaYKlDuQ8Y2oe-yaNefVFszCRNcs81XZWJD%3Fpurpose%3Dfullsize" alt="Image" width="1024" height="1024"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  2. The Limits of the Modern Data Stack
&lt;/h2&gt;

&lt;h3&gt;
  
  
  First, Recognize What It Got Right
&lt;/h3&gt;

&lt;p&gt;Before talking about what comes next, it is important to recognize how successful the Modern Data Stack has been.&lt;/p&gt;

&lt;p&gt;Over the past decade, it addressed many of the biggest challenges in enterprise data infrastructure.&lt;/p&gt;

&lt;p&gt;The old model relied on heavy projects, specialized hardware, extensive customization, long delivery cycles, and tightly coupled systems.&lt;/p&gt;

&lt;p&gt;The Modern Data Stack introduced modular engineering, cloud resources, standardized tools, and composability.&lt;/p&gt;

&lt;p&gt;It transformed data engineering from building one large system into &lt;strong&gt;assembling a set of specialized capabilities&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FGBDyAA-Obh9r2V9svuGKW00j25xvVjBKFRD2eIvxPHPZUQsXWU--7aIU12np76A7o3ghUxEfOViM3We-TcABnMXargLBirKQIW5o58wUorYJ1kJO3KKyKHVw6BkgaqLWacFz1eI79CkKg_z_VT8ttv_WoXPT9Ioagjuana_HLvT6R_soCZrUewm38pvNRezF%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FGBDyAA-Obh9r2V9svuGKW00j25xvVjBKFRD2eIvxPHPZUQsXWU--7aIU12np76A7o3ghUxEfOViM3We-TcABnMXargLBirKQIW5o58wUorYJ1kJO3KKyKHVw6BkgaqLWacFz1eI79CkKg_z_VT8ttv_WoXPT9Ioagjuana_HLvT6R_soCZrUewm38pvNRezF%3Fpurpose%3Dfullsize" alt="Image" width="1200" height="675"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Instead of procurement, deployment, customization, and tightly coupled integration, teams could combine &lt;strong&gt;Cloud, Open Source, SaaS, and Standard APIs&lt;/strong&gt; like building blocks.&lt;/p&gt;

&lt;p&gt;That brought something the traditional data platform could not offer at the same scale: freedom to choose, evolve, replace, and recombine individual components.&lt;/p&gt;

&lt;p&gt;Its most important contribution was arguably &lt;strong&gt;tool standardization&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Data engineering was decomposed into a professional toolchain:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Source&lt;/strong&gt;&lt;br&gt;
Business systems, SaaS, APIs&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Ingestion&lt;/strong&gt;&lt;br&gt;
Synchronization, CDC, files&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Storage&lt;/strong&gt;&lt;br&gt;
Warehouses, lakehouses, compute&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Transformation&lt;/strong&gt;&lt;br&gt;
SQL, models, testing&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Orchestration&lt;/strong&gt;&lt;br&gt;
DAGs, scheduling, retries&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Governance&lt;/strong&gt;&lt;br&gt;
Catalog, lineage, permissions&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;BI&lt;/strong&gt;&lt;br&gt;
Dashboards and self-service analytics&lt;/p&gt;

&lt;p&gt;Each layer became more specialized, and that specialization dramatically improved engineering productivity.&lt;/p&gt;

&lt;h3&gt;
  
  
  But There Is a Fundamental Limitation
&lt;/h3&gt;

&lt;p&gt;The problem is also hidden in that success:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The Modern Data Stack was designed for humans, not agents.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FsEtulkfDorVf9z-5m3zV5mHtkhm78NaoKLgoPMQxEZQRjmtHnxLfO4zwWHoU0Fon9bXGprpw-itNrU54QJ7w2YCL7dUlWpq3sYpMZp8ey12j39XROksG8GB7__oeF1ir7DXfZZ0icphsZD1mJ90KFew636B-P9kU-2EW1dP9n8LV9WKRs-gqm1xcu9AByCMN%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FsEtulkfDorVf9z-5m3zV5mHtkhm78NaoKLgoPMQxEZQRjmtHnxLfO4zwWHoU0Fon9bXGprpw-itNrU54QJ7w2YCL7dUlWpq3sYpMZp8ey12j39XROksG8GB7__oeF1ir7DXfZZ0icphsZD1mJ90KFew636B-P9kU-2EW1dP9n8LV9WKRs-gqm1xcu9AByCMN%3Fpurpose%3Dfullsize" alt="Image" width="1200" height="800"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;When humans use data tools, they fill in missing context almost automatically.&lt;/p&gt;

&lt;p&gt;They read documentation. They understand business terminology. They resolve ambiguity. They recognize risk. They know when something looks suspicious. And, critically, they understand who is accountable for the outcome.&lt;/p&gt;

&lt;p&gt;Agents do not automatically have that context.&lt;/p&gt;

&lt;p&gt;When an agent encounters SQL, documentation, DAGs, logs, business rules, and data catalogs, it cannot simply assume that the missing information is obvious.&lt;/p&gt;

&lt;p&gt;This reveals an uncomfortable truth about today's supposedly automated data platforms:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Much of their "automation" still depends on humans silently filling in the gaps.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  AI Is Amplifying the Problem
&lt;/h3&gt;

&lt;p&gt;Generative AI is changing the economics of software generation.&lt;/p&gt;

&lt;p&gt;The cost of producing SQL, code, DAGs, and configuration is rapidly approaching zero.&lt;/p&gt;

&lt;p&gt;As a result, the scarce part of data engineering is shifting.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;SQL, code, DAGs, and configuration are becoming commodities.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;What remains scarce is:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Context. Verification. Governance. Controlled Execution. Accountability.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The old scarce skills were:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Write SQL → Build Pipelines → Configure DAGs&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The emerging scarce skills are:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Provide Context → Verify Results → Govern Actions → Control Execution&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FpwzFBRKbE4PCkVLPPy1RcCdiIhPpkyK5XLM2P0lvuzmaIo0bGruK3yulnMO_Cp6XcRPlcT5NCrh8askqfxqRe_gYuWdLBruxhsFCqYNC_kU609Ss3JIKvY1q2-bwkyICUAdgbKTkIs-J5oWTAAIKwCZ7j5jj-WugQ7JkwWnmDSEVbLlmRlfbqsp4ucjzgn-E%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FpwzFBRKbE4PCkVLPPy1RcCdiIhPpkyK5XLM2P0lvuzmaIo0bGruK3yulnMO_Cp6XcRPlcT5NCrh8askqfxqRe_gYuWdLBruxhsFCqYNC_kU609Ss3JIKvY1q2-bwkyICUAdgbKTkIs-J5oWTAAIKwCZ7j5jj-WugQ7JkwWnmDSEVbLlmRlfbqsp4ucjzgn-E%3Fpurpose%3Dfullsize" alt="Image" width="1024" height="1024"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;The real challenge in the AI era is therefore not generating data engineering artifacts.&lt;/p&gt;

&lt;p&gt;It is &lt;strong&gt;safely delivering those artifacts into production&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;This is why the conversation is shifting.&lt;/p&gt;

&lt;p&gt;The question is no longer simply:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;Can AI generate code?&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;It is increasingly:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;How does AI-generated work become production-ready?&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;For engineering teams, the real concern is not that AI can write SQL. It is what happens after the SQL has been written.&lt;/p&gt;

&lt;p&gt;Who verifies it?&lt;/p&gt;

&lt;p&gt;Who approves it?&lt;/p&gt;

&lt;p&gt;Who owns the outcome?&lt;/p&gt;

&lt;p&gt;Who is responsible if it is wrong?&lt;/p&gt;

&lt;p&gt;Three problems become particularly important.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Complexity.&lt;/strong&gt; There are already enough tools. Will agents create even more hidden dependencies, one-off scripts, and temporary workflows?&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Accountability.&lt;/strong&gt; If an agent generates SQL, ETL, or a DAG, who confirms that it is correct? Who approves it? Who owns the result?&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Production risk.&lt;/strong&gt; The most dangerous scenario is not necessarily that an agent writes incorrect SQL. It is that the incorrect SQL &lt;strong&gt;runs successfully&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;And beneath these concerns are six additional engineering requirements:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Validation, Ownership, Lineage, Security, Rollback, and Audit.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;AI can dramatically reduce the cost of generation. But without engineering controls, it can also dramatically increase operational complexity.&lt;/p&gt;

&lt;p&gt;Data engineering has no meaningful concept of "close enough."&lt;/p&gt;

&lt;p&gt;In many AI applications, an inaccurate answer may simply result in a poor user experience.&lt;/p&gt;

&lt;p&gt;In data engineering, an incorrect result can flow directly into financial reports, business operations, customer decisions, and automated systems.&lt;/p&gt;

&lt;p&gt;A missed CDC event, incorrect metric definition, incomplete dataset, or untraceable transformation can propagate through an entire chain:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Wrong Data → Wrong Decision → Wrong Action&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The most dangerous failure is therefore not an agent that fails to execute.&lt;/p&gt;

&lt;p&gt;It is an agent that produces the wrong result and &lt;strong&gt;successfully executes it in production&lt;/strong&gt;.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. Harness Engineering: Turning Generation into Delivery
&lt;/h2&gt;

&lt;p&gt;The answer is an engineering layer between agents and the underlying tools: &lt;strong&gt;the Harness&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;A Harness provides the controls, context, verification, and recovery mechanisms required to turn AI-generated work into production engineering.&lt;/p&gt;

&lt;h3&gt;
  
  
  Three Types of Evidence for Production Delivery
&lt;/h3&gt;

&lt;p&gt;A production-grade Harness should provide three types of evidence.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Outcome Evidence&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;First, prove that the system actually improves delivery rather than simply adding another AI interface.&lt;/p&gt;

&lt;p&gt;Does it reduce delivery time rather than simply moving work from execution to review?&lt;/p&gt;

&lt;p&gt;Are errors discovered earlier?&lt;/p&gt;

&lt;p&gt;Are rework, context switching, and waiting reduced?&lt;/p&gt;

&lt;p&gt;Does the team actually complete engineering work faster?&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Process Evidence&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Next, every step should be explainable, traceable, and recoverable.&lt;/p&gt;

&lt;p&gt;Can the boundaries between input, generation, approval, and execution be traced?&lt;/p&gt;

&lt;p&gt;When something goes wrong, can the team determine whether the problem originated in Context, Skill, Runtime, or Policy?&lt;/p&gt;

&lt;p&gt;Can the system retry, roll back, or hand control back to a human without forcing the team to start over?&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Governance Evidence&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Finally, high-risk actions must be explicitly constrained rather than implicitly delegated to the model.&lt;/p&gt;

&lt;p&gt;Which actions can run automatically?&lt;/p&gt;

&lt;p&gt;Which require approval?&lt;/p&gt;

&lt;p&gt;Are those rules defined by Policy?&lt;/p&gt;

&lt;p&gt;Does human intervention happen only at meaningful decision points?&lt;/p&gt;

&lt;p&gt;Can the audit trail answer:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Who did what, when, why, and what happened as a result?&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FEcUFHnM8RoAlJcIXYpXENHsLV_Am5FYTk1D1DHk5olXJJ4_BQve_fF6ZYbIpRk7zHFv4yPYW7JNPbQ7b21LxGoEcBZW5Gflo0KaCrwHdvWfZHguZHbGfZpi1FC_LStH6wG8sgNj-T-LcN6oFAZ9eYobGK0i-eKIrhzo-yuI7CaLKwSauyqLKKe0FnX0Mv9Kl%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FEcUFHnM8RoAlJcIXYpXENHsLV_Am5FYTk1D1DHk5olXJJ4_BQve_fF6ZYbIpRk7zHFv4yPYW7JNPbQ7b21LxGoEcBZW5Gflo0KaCrwHdvWfZHguZHbGfZpi1FC_LStH6wG8sgNj-T-LcN6oFAZ9eYobGK0i-eKIrhzo-yuI7CaLKwSauyqLKKe0FnX0Mv9Kl%3Fpurpose%3Dfullsize" alt="Image" width="1536" height="1024"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Without these three types of evidence, an agent simply generates content faster.&lt;/p&gt;

&lt;p&gt;With them, the agent begins to &lt;strong&gt;deliver engineering outcomes&lt;/strong&gt;.&lt;/p&gt;

&lt;h3&gt;
  
  
  Seven Sign-off Gates for Production
&lt;/h3&gt;

&lt;p&gt;Before an agent-driven workflow reaches production, organizations should be able to verify seven critical sign-off gates.&lt;/p&gt;

&lt;p&gt;These are not product features. They are production-readiness checkpoints.&lt;/p&gt;

&lt;p&gt;Missing even one of them can turn a promising demo into an operational risk.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;1. Intent can be verified&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Goals, boundaries, and acceptance criteria must be captured as structured inputs rather than relying on informal instructions.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;2. Context is complete&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Business, data, permission, and execution context must be available. The model should not be expected to guess what the enterprise means.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;3. The plan can be reviewed&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;SQL, DAGs, and task steps should be readable by humans, verifiable by systems, and reviewable from a risk perspective.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;4. Execution is controlled&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Permissions, target environments, execution paths, and Skill selection must operate within defined Policies.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;5. Results can be validated&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Results should pass data-quality checks, business-definition checks, reconciliation, or lineage validation rather than being accepted simply because execution succeeded.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;6. Failures can be recovered&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The system must know whether to retry, roll back, or escalate to a human. Failures cannot simply remain unresolved.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;7. Actions are auditable&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Plans, approvals, executions, results, and version changes should be recorded across the full lifecycle.&lt;/p&gt;

&lt;p&gt;These seven gates are not designed to make agents smarter.&lt;/p&gt;

&lt;p&gt;They are designed to give organizations a clear answer to a more important question:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;When is it safe to delegate?&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Three Foundations: Correctness, Capability, and Context
&lt;/h3&gt;

&lt;p&gt;A Harness is only as effective as the three foundations underneath it.&lt;/p&gt;

&lt;h4&gt;
  
  
  SQL That Runs Is Not Necessarily Business-Correct
&lt;/h4&gt;

&lt;p&gt;In data engineering, successful execution is only the lowest bar.&lt;/p&gt;

&lt;p&gt;There are at least four levels of correctness:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Syntactic correctness → Execution correctness → Data correctness → Business correctness&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;A query can be technically correct and still produce the wrong business result.&lt;/p&gt;

&lt;p&gt;Consider revenue.&lt;/p&gt;

&lt;p&gt;Should "Revenue" mean:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Order Amount?&lt;/li&gt;
&lt;li&gt;Paid Amount?&lt;/li&gt;
&lt;li&gt;Recognized Revenue?&lt;/li&gt;
&lt;li&gt;Net Revenue?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The query engine can determine whether the SQL is valid.&lt;/p&gt;

&lt;p&gt;Only enterprise context can determine whether the calculation is &lt;strong&gt;business-correct&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Common failure modes include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Choosing the wrong data source&lt;/li&gt;
&lt;li&gt;Using the wrong metric definition&lt;/li&gt;
&lt;li&gt;Applying the wrong time window&lt;/li&gt;
&lt;li&gt;Creating duplicates through joins&lt;/li&gt;
&lt;li&gt;Ignoring business rules&lt;/li&gt;
&lt;/ul&gt;

&lt;h4&gt;
  
  
  Without Capability, Agents Generate More Temporary Scripts
&lt;/h4&gt;

&lt;p&gt;There is another risk.&lt;/p&gt;

&lt;p&gt;If agents can only interact with raw SQL, Python, Shell, or low-level APIs, they may simply generate more one-off scripts.&lt;/p&gt;

&lt;p&gt;Temporary scripts are typically:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;One-time implementations&lt;/li&gt;
&lt;li&gt;Difficult to standardize&lt;/li&gt;
&lt;li&gt;Difficult to audit&lt;/li&gt;
&lt;li&gt;Difficult to roll back&lt;/li&gt;
&lt;li&gt;Poor at providing structured feedback&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Engineering Capabilities are different.&lt;/p&gt;

&lt;p&gt;A Capability is a reusable, structured Skill with defined inputs, outputs, policies, validation, and recovery behavior.&lt;/p&gt;

&lt;p&gt;The difference is fundamental:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Without Capability, AI simply upgrades "humans writing temporary scripts" into "AI generating temporary scripts."&lt;/strong&gt;&lt;/p&gt;

&lt;h4&gt;
  
  
  Without Context, Agents Guess What the Enterprise Means
&lt;/h4&gt;

&lt;p&gt;Enterprise data is not simply a collection of tables and columns.&lt;/p&gt;

&lt;p&gt;It is a context system containing business semantics, technical relationships, historical rules, and organizational ownership.&lt;/p&gt;

&lt;p&gt;Consider a seemingly simple request:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;Calculate revenue from high-value customers over the last 30 days.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;An agent needs to know:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;How are high-value customers defined?&lt;/li&gt;
&lt;li&gt;What revenue definition should be used?&lt;/li&gt;
&lt;li&gt;What exactly counts as the last 30 days?&lt;/li&gt;
&lt;li&gt;Which data source is authoritative?&lt;/li&gt;
&lt;li&gt;How should refunds be handled?&lt;/li&gt;
&lt;li&gt;Who owns and approves the result?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These questions correspond to four types of context:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Business Context&lt;/strong&gt;&lt;br&gt;
&lt;strong&gt;Data Context&lt;/strong&gt;&lt;br&gt;
&lt;strong&gt;Execution Context&lt;/strong&gt;&lt;br&gt;
&lt;strong&gt;Organizational Context&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Without Context, an agent is not understanding the enterprise.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;It is guessing.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Software Itself Needs to Be Rebuilt for Agents
&lt;/h3&gt;

&lt;p&gt;AI is also forcing software companies to answer a broader question:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;What does software look like when agents, rather than humans, are its primary users?&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FB8dM0SK1R_fPlBm-5QV1tXUV5OyRGXrTzvVP0UxDvTZcxz_SBlskwY5T73DgaPNyzCREeLC0IES3I18zA8db6W4DTkNkJgtrk1X6KbGdHq2IVVlNqK_zhfAympbqN0SZa55iz0sNj8PiyTN6TeGFzojuNSlDpmBp80k6DwVD12KqEch0KRq5Lk_mY69CUJvO%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FB8dM0SK1R_fPlBm-5QV1tXUV5OyRGXrTzvVP0UxDvTZcxz_SBlskwY5T73DgaPNyzCREeLC0IES3I18zA8db6W4DTkNkJgtrk1X6KbGdHq2IVVlNqK_zhfAympbqN0SZa55iz0sNj8PiyTN6TeGFzojuNSlDpmBp80k6DwVD12KqEch0KRq5Lk_mY69CUJvO%3Fpurpose%3Dfullsize" alt="Image" width="1024" height="1024"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Traditional software is &lt;strong&gt;UI-first&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;A person opens an interface, finds a feature, fills in configuration, clicks Run, and handles exceptions.&lt;/p&gt;

&lt;p&gt;Agent-native software needs to become &lt;strong&gt;Skill-first&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;That means:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Skills are discoverable&lt;/li&gt;
&lt;li&gt;Context can be injected&lt;/li&gt;
&lt;li&gt;Policies can constrain actions&lt;/li&gt;
&lt;li&gt;Feedback can be consumed programmatically&lt;/li&gt;
&lt;li&gt;Results can be verified&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Established vendors have significant legacy systems to work with. New companies have more freedom to rethink their architectures.&lt;/p&gt;

&lt;p&gt;In this environment, the speed at which organizations recognize and respond to the shift may become a major competitive advantage.&lt;/p&gt;

&lt;p&gt;Every piece of software that matters to data engineering will need to reconsider how an agent interacts with it.&lt;/p&gt;

&lt;p&gt;The conclusion is straightforward:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Agents do not primarily lack intelligence. They lack the engineering system required to turn intelligence into reliable outcomes.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Today's models can already work with:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;SQL, ETL, DAGs, Logs, Fixes, and Plans.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;But they cannot independently take responsibility for:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Context, Permissions, Validation, Impact, Rollback, and Accountability.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Between &lt;strong&gt;Generation Capability&lt;/strong&gt; and &lt;strong&gt;Production Capability&lt;/strong&gt; sits an entire Engineering System.&lt;/p&gt;

&lt;p&gt;The limiting factor for agents is therefore no longer intelligence alone.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;It is engineering.&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  4. The Five-Layer Agentic Data Stack
&lt;/h2&gt;

&lt;p&gt;A next-generation data platform can be organized into five layers:&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/media%2F17890945545003%2F17890958110893.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/media%2F17890945545003%2F17890958110893.jpg" width="800" height="400"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;The architectural principle is critical:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Agents should not directly call underlying tools.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Instead, they should access those capabilities through the Harness, while Context from L3 and Controls from L4 define what the agent is allowed to do.&lt;/p&gt;

&lt;h3&gt;
  
  
  Breaking Down the Five Layers
&lt;/h3&gt;

&lt;h4&gt;
  
  
  L1: Deterministic Execution
&lt;/h4&gt;

&lt;p&gt;Intelligence should not be responsible for determinism.&lt;/p&gt;

&lt;p&gt;The Runtime is.&lt;/p&gt;

&lt;p&gt;It executes SQL, synchronizes data, runs batch and CDC workloads, schedules jobs, and produces logs and status information.&lt;/p&gt;

&lt;p&gt;Typical components include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Databases and warehouses&lt;/li&gt;
&lt;li&gt;Lakehouses and Iceberg&lt;/li&gt;
&lt;li&gt;Apache Spark&lt;/li&gt;
&lt;li&gt;Apache Flink&lt;/li&gt;
&lt;li&gt;Apache SeaTunnel&lt;/li&gt;
&lt;li&gt;Apache DolphinScheduler&lt;/li&gt;
&lt;li&gt;SQL engines&lt;/li&gt;
&lt;li&gt;Data quality engines&lt;/li&gt;
&lt;/ul&gt;

&lt;h4&gt;
  
  
  L2: The Data Engineering Harness
&lt;/h4&gt;

&lt;p&gt;After an agent understands the goal, generates a plan, and initiates a request, that request should pass through a controlled engineering layer.&lt;/p&gt;

&lt;p&gt;A typical Harness includes:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Skill Definition → Permission Check → Context Injection → Validation → Observability → Rollback&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Only then should the request become a controlled Skill that can interact with databases, operating systems, development platforms, and cloud services.&lt;/p&gt;

&lt;p&gt;Without a Harness, the pattern is:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Direct Scripts → Direct APIs → Difficult Verification → Difficult Auditing&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;With a Harness, it becomes:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Standardized Skills → Clear Boundaries → Structured Feedback → Human Takeover&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The value of the Harness is therefore to transform an agent's ability to generate content into an enterprise's ability to &lt;strong&gt;reliably deliver engineering work&lt;/strong&gt;.&lt;/p&gt;

&lt;h4&gt;
  
  
  L3: Semantic &amp;amp; Knowledge Layer
&lt;/h4&gt;

&lt;p&gt;The Semantic Layer answers questions such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Which table is trusted?&lt;/li&gt;
&lt;li&gt;What does this metric actually mean?&lt;/li&gt;
&lt;li&gt;Is this field sensitive?&lt;/li&gt;
&lt;li&gt;Where did this data come from?&lt;/li&gt;
&lt;li&gt;Who will be affected downstream?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;It combines:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Business Rules, Metadata, Lineage, Metrics, Glossary, Ontology, Data Contracts, and Execution Memory.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;In the BI era, the Semantic Layer primarily helped people understand data.&lt;/p&gt;

&lt;p&gt;In the Agentic era, it becomes a &lt;strong&gt;context layer that agents must consult before taking action&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Without a Semantic Layer, an agent may understand column names.&lt;/p&gt;

&lt;p&gt;It will not necessarily understand the enterprise behind them.&lt;/p&gt;

&lt;h4&gt;
  
  
  L4: Agentic Orchestration Control Plane
&lt;/h4&gt;

&lt;p&gt;The orchestration layer must evolve beyond simply running DAGs.&lt;/p&gt;

&lt;p&gt;Traditional orchestration looks like:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Human defines DAG → Scheduler executes Tasks&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Agentic orchestration looks like:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Human defines Goal → Agent plans → Control Plane enforces boundaries → Human intervenes when necessary&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The control plane introduces checkpoints such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Goal Description&lt;/li&gt;
&lt;li&gt;Skill Checking&lt;/li&gt;
&lt;li&gt;Policy Checking&lt;/li&gt;
&lt;li&gt;Human Gate&lt;/li&gt;
&lt;li&gt;Audit&lt;/li&gt;
&lt;li&gt;Multi-task Configuration&lt;/li&gt;
&lt;/ul&gt;

&lt;h4&gt;
  
  
  L5: Business Intent
&lt;/h4&gt;

&lt;p&gt;The biggest shift happens at the top of the stack.&lt;/p&gt;

&lt;p&gt;The traditional approach asks people to describe &lt;strong&gt;steps&lt;/strong&gt;:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;Synchronize table A to table B.&lt;br&gt;
Write this SQL.&lt;br&gt;
Configure the DAG.&lt;br&gt;
Run it every day at 8 AM.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;The Agentic approach asks people to define &lt;strong&gt;outcomes&lt;/strong&gt;:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;Generate a daily dataset containing revenue from high-value customers, use the approved revenue definition, do not overwrite production tables, and require human approval if the result changes by more than 5%.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;Business intent should therefore cover eight dimensions:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Goal, Data Scope, Time Window, Quality Requirements, Cost Constraints, Risk Boundaries, Approval Conditions, and Acceptance Criteria.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The future of data engineering is not about people describing every step.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;It is about people defining goals, boundaries, and acceptance criteria.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2Fe-5JY5ydC7KbtKowiis8heyLFvh8njl3H_IgujduZXWdqCrOF-6bWp90EVte7GduY-IJ-Rzvv3cCccIiDJVtOQvDYaKa0rgiS7926I8DOeONaZfbyRucYxanC2J0OOOATzlbca4XsG0qqH3ebYxwC8Uv1LusjW7bEglEhsjIjxy_L9uMolRnjqBLoDvT-zgG%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2Fe-5JY5ydC7KbtKowiis8heyLFvh8njl3H_IgujduZXWdqCrOF-6bWp90EVte7GduY-IJ-Rzvv3cCccIiDJVtOQvDYaKa0rgiS7926I8DOeONaZfbyRucYxanC2J0OOOATzlbca4XsG0qqH3ebYxwC8Uv1LusjW7bEglEhsjIjxy_L9uMolRnjqBLoDvT-zgG%3Fpurpose%3Dfullsize" alt="Image" width="1200" height="1500"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FUSQnNw1HUJb78_-i6psO8YR2vu_rz-kDuBvjI-3-u2KvYlvmxjMgztLfxkmh1ymQSEZJFF2x59OzFwiwYHy5VDP8YITQ2P9trY6aPWtYQAkGdXJD8tRDxipXeXi1W0m6ADBNurE7Uc9SUaBTsna2m07NLt9J8x_UjzAx-Se9ifDIPI0QLKJNtkrhZSxzSeBH%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FUSQnNw1HUJb78_-i6psO8YR2vu_rz-kDuBvjI-3-u2KvYlvmxjMgztLfxkmh1ymQSEZJFF2x59OzFwiwYHy5VDP8YITQ2P9trY6aPWtYQAkGdXJD8tRDxipXeXi1W0m6ADBNurE7Uc9SUaBTsna2m07NLt9J8x_UjzAx-Se9ifDIPI0QLKJNtkrhZSxzSeBH%3Fpurpose%3Dfullsize" alt="Image" width="1564" height="882"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Why Re-Layer the Stack?
&lt;/h3&gt;

&lt;p&gt;The next-generation data platform is not simply the old platform with more plugins.&lt;/p&gt;

&lt;p&gt;It requires a new division of responsibilities.&lt;/p&gt;

&lt;p&gt;The traditional stack is organized around:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Storage → Compute → Orchestration → Governance → BI&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The Agentic Data Stack is organized around:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Intent → Control → Semantic → Harness → Runtime&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Together, these layers answer four fundamental questions:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Where does the agent get its business goals?&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;How does it understand enterprise capabilities and meaning?&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Which capabilities can it invoke?&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Who controls execution, validation, and rollback?&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The Agentic era therefore requires data platforms to rethink the relationship between &lt;strong&gt;intent, context, capability, control, and execution&lt;/strong&gt;.&lt;/p&gt;

&lt;h3&gt;
  
  
  The Minimum Viable Loop
&lt;/h3&gt;

&lt;p&gt;The five-layer architecture comes together through a simple Harness loop:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Intent → Context → Plan → Skill → Execute → Validate → Review → Feedback&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Intent&lt;/strong&gt; defines the business goal, constraints, and acceptance criteria.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Context&lt;/strong&gt; supplies business, data, execution, and organizational information.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Plan&lt;/strong&gt; breaks the goal into synchronization, transformation, quality, orchestration, and other tasks.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Skill&lt;/strong&gt; invokes standardized engineering capabilities.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Execute&lt;/strong&gt; runs the work deterministically through the Runtime.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Validate&lt;/strong&gt; checks data results, quality, and business rules.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Review&lt;/strong&gt; brings humans into critical decisions.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Feedback&lt;/strong&gt; sends logs, metrics, exceptions, and approval outcomes back to the agent.&lt;/p&gt;

&lt;p&gt;This creates a continuous loop rather than a one-shot generation process.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Without the loop, an agent generates content.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;With the loop, an agent starts delivering engineering work.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Three Design Principles
&lt;/h3&gt;

&lt;h4&gt;
  
  
  A Skill Is Not a Prompt. It Is a Controlled Execution Unit.
&lt;/h4&gt;

&lt;p&gt;A Prompt is an instruction.&lt;/p&gt;

&lt;p&gt;A Skill is an engineering capability composed of:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Input, Context, Policy, Execution, Validation, Rollback, and Output.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;A Tool API exposes a low-level operation and defines parameters, leaving failure handling to the caller.&lt;/p&gt;

&lt;p&gt;An Engineering Skill encapsulates an end-to-end engineering intent, defines its Context and Policy, and includes validation and recovery mechanisms.&lt;/p&gt;

&lt;p&gt;The distinction is fundamental:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;A Prompt determines how an agent responds. A Skill determines whether an agent can execute safely.&lt;/strong&gt;&lt;/p&gt;

&lt;h4&gt;
  
  
  CLI for Agents, GUI for Humans
&lt;/h4&gt;

&lt;p&gt;The interface model also needs to change.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;CLI/API&lt;/strong&gt; should serve execution and feedback:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Structured inputs and outputs&lt;/li&gt;
&lt;li&gt;Easy programmatic invocation&lt;/li&gt;
&lt;li&gt;Testability&lt;/li&gt;
&lt;li&gt;Version control&lt;/li&gt;
&lt;li&gt;Skills&lt;/li&gt;
&lt;li&gt;MCP&lt;/li&gt;
&lt;li&gt;SDKs&lt;/li&gt;
&lt;li&gt;Declarative configuration&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;GUI&lt;/strong&gt; should serve understanding, review, and governance:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Inspect agent plans and generated artifacts&lt;/li&gt;
&lt;li&gt;Review SQL, DAGs, logs, and results&lt;/li&gt;
&lt;li&gt;Monitor permissions and risk&lt;/li&gt;
&lt;li&gt;Take control when uncertainty or exceptions arise&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;A complete workflow looks like this:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Human defines the goal through the GUI → Agent invokes Skills through CLI/API → Runtime executes → GUI presents DAGs, SQL, logs, and risk information → Human approves or takes over&lt;/strong&gt;&lt;/p&gt;

&lt;h4&gt;
  
  
  Human-in-the-Loop Should Protect Risk Boundaries, Not Approve Everything
&lt;/h4&gt;

&lt;p&gt;Human-in-the-loop does not mean humans should approve every agent action.&lt;/p&gt;

&lt;p&gt;Every action should first pass through Policy-based automated screening and then be handled according to its risk level.&lt;/p&gt;

&lt;p&gt;A practical model is:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Low risk → Automatic execution&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Medium risk → Execute and notify&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;High risk → Human approval&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Critical risk → Block or require dual approval&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;This leads to three principles:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Intervene based on risk, not every step.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Approve critical decisions, not mechanical actions.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Humans define the Policy; agents operate within it.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;For example:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Low risk:&lt;/strong&gt; reading metadata, querying development environments, generating documentation.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Medium risk:&lt;/strong&gt; creating development tasks, running low-cost validation.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;High risk:&lt;/strong&gt; writing to production, changing schemas, modifying critical metrics.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Critical risk:&lt;/strong&gt; deleting core tables, bulk overwrites, or high-risk operations involving sensitive data.&lt;/p&gt;

&lt;p&gt;The goal is not to turn humans into approval bottlenecks.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Humans should design the risk boundaries within which agents operate.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Clear Ownership and Feedback-Driven Recovery
&lt;/h3&gt;

&lt;p&gt;Production ownership also needs to be explicit.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Business Sign-off:&lt;/strong&gt; Business owners define goals, constraints, and acceptance criteria.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Platform Governance:&lt;/strong&gt; Platform owners define Policies, permissions, approvals, rollback mechanisms, and environment boundaries.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Automated Execution:&lt;/strong&gt; Agents and the Harness supply Context, generate Plans, invoke Skills, execute through the Runtime, and return logs, validation results, and exceptions.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Audit:&lt;/strong&gt; Reviewers and audit systems step in for high-risk, uncertain, or acceptance-conflicting decisions and maintain records of critical decisions.&lt;/p&gt;

&lt;p&gt;Human intervention should generally be reserved for three situations:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;A high-risk action could modify critical data or business state.&lt;/li&gt;
&lt;li&gt;An exception cannot be safely recovered and the retry or rollback boundary is unclear.&lt;/li&gt;
&lt;li&gt;The result conflicts with the acceptance criteria and requires a final business decision.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;The product organizations ultimately need is not a demo that can write SQL.&lt;/p&gt;

&lt;p&gt;It is a system where &lt;strong&gt;goals, permissions, execution, and outcomes can be managed as one accountable loop&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;A truly Agentic system must also be able to learn from execution feedback.&lt;/p&gt;

&lt;p&gt;A practical Feedback Loop is:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Plan → Execute → Observe → Diagnose → Repair → Validate → Continue / Rollback / Escalate&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;This loop is supported by &lt;strong&gt;Execution Memory&lt;/strong&gt;, which continuously captures context, execution history, and operational experience.&lt;/p&gt;

&lt;p&gt;Feedback can come from:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Execution state&lt;/li&gt;
&lt;li&gt;Logs&lt;/li&gt;
&lt;li&gt;Metrics&lt;/li&gt;
&lt;li&gt;Data quality&lt;/li&gt;
&lt;li&gt;Lineage impact&lt;/li&gt;
&lt;li&gt;Human feedback&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The agent can then:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Automatically repair&lt;/strong&gt; configuration, parameters, SQL, or workflow issues when the root cause is clear.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Adjust the plan&lt;/strong&gt; by changing resources, sequencing, or execution strategies.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Stop or escalate&lt;/strong&gt; when the system cannot safely resolve the problem.&lt;/p&gt;

&lt;p&gt;The key to Agentic systems is therefore not simply autonomous execution.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;It is knowing what happened after execution.&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  5. From Concept to Practice
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Case Study: A Complete Data Engineering Loop
&lt;/h3&gt;

&lt;p&gt;A meaningful Agentic data engineering demo should prove more than the ability to generate SQL.&lt;/p&gt;

&lt;p&gt;The real test is whether an agent can complete an end-to-end engineering workflow.&lt;/p&gt;

&lt;p&gt;Starting from a business goal, the agent should be able to:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Discover data → Create an integration task → Execute synchronization → Generate SQL transformations → Build a workflow DAG → Execute the workflow → Read logs and diagnose issues → Repair and retry → Present the result for human review&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;In this model:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Agent Planning&lt;/strong&gt; handles planning.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Harness Control&lt;/strong&gt; manages permissions, policies, validation, and execution boundaries.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Runtime Execution&lt;/strong&gt; performs deterministic engineering work.&lt;/p&gt;

&lt;p&gt;The value of the demo is therefore not proving that a model can generate SQL.&lt;/p&gt;

&lt;p&gt;It is proving that an agent can use a Harness to &lt;strong&gt;orchestrate multiple deterministic engineering systems into a complete delivery workflow&lt;/strong&gt;.&lt;/p&gt;

&lt;h2&gt;
  
  
  Two Practical Harness Implementations
&lt;/h2&gt;

&lt;p&gt;The theory becomes meaningful when it is reflected in real engineering systems.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Apache SeaTunnel&lt;/strong&gt; and &lt;strong&gt;Apache DolphinScheduler&lt;/strong&gt; provide two complementary examples of how Harness principles can be applied in open source data engineering.&lt;/p&gt;

&lt;p&gt;SeaTunnel represents the &lt;strong&gt;data integration capability&lt;/strong&gt; side: how a foundational data integration engine can evolve into a Skill that agents can discover, invoke, validate, and recover.&lt;/p&gt;

&lt;p&gt;DolphinScheduler represents the &lt;strong&gt;engineering execution and orchestration&lt;/strong&gt; side: how agent-generated work can become a real, executable, observable, and reviewable engineering asset.&lt;/p&gt;

&lt;p&gt;One manages how data moves.&lt;/p&gt;

&lt;p&gt;The other manages how engineering workflows run reliably.&lt;/p&gt;

&lt;p&gt;Together, they illustrate how the &lt;strong&gt;L1 Runtime and L2 Harness&lt;/strong&gt; layers can work together in real-world data engineering.&lt;/p&gt;

&lt;h3&gt;
  
  
  Apache SeaTunnel CLI: Turning Data Integration into a Skill
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FJWRHA2w3_dGMNuOwIYLXOoMqzmzYsoPeLJ6dS50XWaOi62E2AeMkirhmXeIzjbaRTDyQAeHz5BcQwmJDGBFZedwxtA_tKRxyAtzWO7kE9WPFJJ7q0JxmFLwi4ALCXdCzLwUOvm4K3u9o6IMD16skVXlo2L-iLeoUVLkFP4OQj6MRr-HDEj3-RD0-CrqIXRXZ%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FJWRHA2w3_dGMNuOwIYLXOoMqzmzYsoPeLJ6dS50XWaOi62E2AeMkirhmXeIzjbaRTDyQAeHz5BcQwmJDGBFZedwxtA_tKRxyAtzWO7kE9WPFJJ7q0JxmFLwi4ALCXdCzLwUOvm4K3u9o6IMD16skVXlo2L-iLeoUVLkFP4OQj6MRr-HDEj3-RD0-CrqIXRXZ%3Fpurpose%3Dfullsize" alt="Image" width="1080" height="1811"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FQAFnSPr3qb0Uu61hb-9f6t7OsMqU30LzPD26wKO-Z7L0KZr2ELOK1A50-O-Bh2WTOmfUcXZwL2XcfTkygyI4-ihWmNwpt75sTJX-cAP8Un6NyIpMiABLBCVJrkk_-W1EcIdw8ocggyfRvFUNWJSN6ywmVa0yYwatSgeScgtMeF-yINHLpKNFVYhWf2SoW11K%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2FQAFnSPr3qb0Uu61hb-9f6t7OsMqU30LzPD26wKO-Z7L0KZr2ELOK1A50-O-Bh2WTOmfUcXZwL2XcfTkygyI4-ihWmNwpt75sTJX-cAP8Un6NyIpMiABLBCVJrkk_-W1EcIdw8ocggyfRvFUNWJSN6ywmVa0yYwatSgeScgtMeF-yINHLpKNFVYhWf2SoW11K%3Fpurpose%3Dfullsize" alt="Image" width="1672" height="941"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;The &lt;strong&gt;Apache SeaTunnel CLI&lt;/strong&gt; provides capabilities such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Data source discovery, including Source, Schema, Table, and Field&lt;/li&gt;
&lt;li&gt;Automatic SeaTunnel Job generation&lt;/li&gt;
&lt;li&gt;Batch synchronization and CDC execution&lt;/li&gt;
&lt;li&gt;Structured execution feedback, including status, logs, row counts, and errors&lt;/li&gt;
&lt;li&gt;Error-driven repair and retry&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;An agent's intent can be translated into a set of SeaTunnel Skills:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;DiscoverSource()&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;InspectSchema()&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;CreateBatchSync()&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;CreateCDC()&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;ValidateMapping()&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;RunSyncJob()&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The Data Integration Runtime then handles the execution loop:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Execute → Logs → Repair → Retry&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;This is an important architectural shift.&lt;/p&gt;

&lt;p&gt;Instead of asking an agent to generate another temporary data integration script, SeaTunnel exposes reusable, structured capabilities that can be incorporated into an agent-driven engineering workflow.&lt;/p&gt;

&lt;p&gt;The future direction for SeaTunnel CLI includes:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Context-aware Mapping&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Expanding from Batch and CDC operations into a broader &lt;strong&gt;Data Flow Skill&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Self-healing Data Integration&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;AI-ready Data Pipelines&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The long-term destination of SeaTunnel CLI is therefore not simply a better command-line interface.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;It is a Data Integration Skill that agents can reliably discover and use.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Apache DolphinScheduler: Turning Generated Work into Engineering Order
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2F3FI4_ppKZjNe2WryiUBcMsUwMEwCgtRSGlaewCiy8hSSxY4BlBN7057bYvp2Vi4aLk-vRZMW35-jbxPXSjdIjrRUFakh--v9GKIGopxqalQw0sAWnsxvDneDtof9WW7TAEry0p8t2zFPxBxLHr6Ha8kEVCX6QdkaWu4pBemCcCMGoJ14Dz251WEeIP259WHR%3Fpurpose%3Dfullsize" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fimages.openai.com%2Fstatic-rsc-4%2F3FI4_ppKZjNe2WryiUBcMsUwMEwCgtRSGlaewCiy8hSSxY4BlBN7057bYvp2Vi4aLk-vRZMW35-jbxPXSjdIjrRUFakh--v9GKIGopxqalQw0sAWnsxvDneDtof9WW7TAEry0p8t2zFPxBxLHr6Ha8kEVCX6QdkaWu4pBemCcCMGoJ14Dz251WEeIP259WHR%3Fpurpose%3Dfullsize" alt="Image" width="1200" height="1203"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Apache DolphinScheduler provides another critical part of the Harness architecture.&lt;/p&gt;

&lt;p&gt;Its capabilities include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Generating Workflow DAGs&lt;/li&gt;
&lt;li&gt;Establishing task dependencies automatically&lt;/li&gt;
&lt;li&gt;Creating real engineering assets such as Definitions, Instances, and Versions&lt;/li&gt;
&lt;li&gt;Executing and monitoring workflows through status, timing, retries, and logs&lt;/li&gt;
&lt;li&gt;Repairing failed workflows and rerunning nodes&lt;/li&gt;
&lt;li&gt;Providing a GUI for human review&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The shift can be summarized as:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Traditional Workflow&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Human defines every Task → Human configures dependencies → Scheduler executes DAG&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Agentic Workflow&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Human defines business goal → Agent generates execution plan → Policy checks boundaries → DolphinScheduler executes Workflow → Agent repairs issues / Human reviews&lt;/p&gt;

&lt;p&gt;This is where orchestration becomes more than task scheduling.&lt;/p&gt;

&lt;p&gt;DolphinScheduler provides the engineering structure required to turn generated plans into persistent, observable, and manageable workflow assets.&lt;/p&gt;

&lt;p&gt;Its future direction can include:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Policy-aware Orchestration&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Human Review Gates&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Self-healing Workflows&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Multi-Agent Coordination&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Data Engineers Are Not Disappearing. Their Role Is Expanding.
&lt;/h3&gt;

&lt;p&gt;The rise of agents does not eliminate data engineering.&lt;/p&gt;

&lt;p&gt;It changes where data engineers create value.&lt;/p&gt;

&lt;p&gt;The role is evolving through five stages:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;SQL Writer&lt;/strong&gt;&lt;br&gt;
Focused on writing SQL to solve individual problems&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Pipeline Builder&lt;/strong&gt;&lt;br&gt;
Configuring data flows and pipelines&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Workflow Operator&lt;/strong&gt;&lt;br&gt;
Running and monitoring data workflows&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Platform Engineer&lt;/strong&gt;&lt;br&gt;
Building platforms and infrastructure&lt;/p&gt;

&lt;p&gt;→ &lt;strong&gt;Agent Capability Designer&lt;/strong&gt;&lt;br&gt;
Designing agent capabilities and the engineering systems behind them&lt;/p&gt;

&lt;p&gt;Traditional data engineering work includes writing SQL, Python, and Spark code, configuring connectors and ETL jobs, building DAGs, managing schedules, troubleshooting failures, and documenting data.&lt;/p&gt;

&lt;p&gt;The next generation of data engineers will increasingly focus on five types of design:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Context Designer&lt;/strong&gt;&lt;br&gt;
Design the business and data context that allows agents to understand the enterprise.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Skill Designer&lt;/strong&gt;&lt;br&gt;
Create reusable data capabilities that agents can reliably invoke and combine.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Policy Designer&lt;/strong&gt;&lt;br&gt;
Define rules and constraints that make agent actions controlled and trustworthy.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Evaluation Designer&lt;/strong&gt;&lt;br&gt;
Design evaluation criteria and mechanisms that make agent performance measurable and sustainable.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Agent Engineering Commander&lt;/strong&gt;&lt;br&gt;
Coordinate the broader engineering system and delivery process to amplify the capabilities of data teams.&lt;/p&gt;

&lt;p&gt;Six core capabilities will remain essential:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Data modeling, business abstraction, architecture design, data governance, risk judgment, and accountability.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;These are not becoming less important.&lt;/p&gt;

&lt;p&gt;They are becoming the foundation for designing Context, Skills, and Policies that agents can actually use.&lt;/p&gt;

&lt;p&gt;The most valuable data engineers of the future will therefore not necessarily be the people who build the most pipelines.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;They will be the people who know how to organize data engineering capabilities so that agents can use them safely and effectively.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Start with High-Frequency, Low-Risk Work
&lt;/h3&gt;

&lt;p&gt;Organizations should not begin Harness adoption by attempting to automate an entire data pipeline.&lt;/p&gt;

&lt;p&gt;The better approach is to start with &lt;strong&gt;high-frequency, low-risk tasks&lt;/strong&gt; where boundaries are clear and outcomes can be verified.&lt;/p&gt;

&lt;p&gt;Prove that the system can deliver reliably within a well-defined scope, then gradually expand its autonomy.&lt;/p&gt;

&lt;p&gt;Four categories are particularly suitable for the first phase:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Data discovery, metadata enrichment, and schema understanding&lt;/strong&gt;&lt;br&gt;
These are generally read-heavy tasks with limited write risk.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;SQL drafting, rule validation, and DAG assembly&lt;/strong&gt;&lt;br&gt;
These follow a "generate first, review second" model.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Integration task creation, parameter orchestration, and environment checks&lt;/strong&gt;&lt;br&gt;
These are repetitive engineering tasks that can be standardized and templated.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Log diagnosis, repair recommendations, and retry orchestration&lt;/strong&gt;&lt;br&gt;
These form operational loops that can be replayed and verified.&lt;/p&gt;

&lt;p&gt;By contrast, organizations should avoid fully autonomous execution for high-risk tasks such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Deleting, overwriting, or bulk-modifying production data&lt;/li&gt;
&lt;li&gt;Changing critical metric definitions or restructuring cross-domain master data&lt;/li&gt;
&lt;li&gt;Schema changes or high-cost writes without approval and rollback mechanisms&lt;/li&gt;
&lt;li&gt;Complex cross-team workflows where ownership is unclear or results cannot be automatically verified&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;A practical Harness adoption path can be divided into three stages.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Collaborative Assistance&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Agents discover metadata, generate drafts, and provide recommendations while humans review the results.&lt;/p&gt;

&lt;p&gt;The goal is to teach the system how to operate under verification.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Controlled Execution&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Harness-managed agents execute non-critical, reversible, and verifiable tasks.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Governed Autonomy&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Only after Policies, auditing, validation, and rollback mechanisms are mature should organizations expand the agent's autonomous authority.&lt;/p&gt;

&lt;p&gt;The principle is simple:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Start small. Prove reliability. Expand the boundary of autonomy.&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Conclusion: The Future Is Trusted Agentic Data Engineering
&lt;/h2&gt;

&lt;p&gt;The future of data engineering is not about removing humans from the loop.&lt;/p&gt;

&lt;p&gt;It is about changing what humans do.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Humans define the goal.&lt;br&gt;
Agents execute the work.&lt;br&gt;
Harness Engineering governs the delivery.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The model can be summarized as:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Business Intent + Agent Intelligence + Enterprise Context + Engineering Skills + Policy &amp;amp; Control + Human Review = Trusted Agentic Data Engineering&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The fundamental shift is not simply about better prompts, larger context windows, or more capable models.&lt;/p&gt;

&lt;p&gt;It is about building an engineering system around those models that enables agents to operate continuously, safely, and accountably.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The agent generates.&lt;br&gt;
The Runtime executes.&lt;br&gt;
The Harness makes the outcome trustworthy.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;That is the deeper transformation taking place in data engineering.&lt;/p&gt;

&lt;p&gt;The next generation of data platforms will not be defined simply by how many tools they integrate or how much code their AI can generate. They will be defined by how effectively they connect &lt;strong&gt;business intent, enterprise context, engineering capabilities, execution controls, verification, and human accountability&lt;/strong&gt; into one continuous delivery loop.&lt;/p&gt;

&lt;p&gt;And that may be the real architecture of data engineering in the Agentic AI era.&lt;/p&gt;

</description>
      <category>dataengineering</category>
      <category>harnessengineering</category>
      <category>agents</category>
      <category>ai</category>
    </item>
    <item>
      <title>🚀 Apache DolphinScheduler’s August updates bring stronger security, smarter missed-fire handling, performance gains, and critical bug fixes. Explore the changes! #DolphinScheduler #Apache #DataOps</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 11 Sep 2026 09:13:12 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinschedulers-august-updates-bring-stronger-security-smarter-missed-fire-handling-140h</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/apache-dolphinschedulers-august-updates-bring-stronger-security-smarter-missed-fire-handling-140h</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/whats-new-in-apache-dolphinscheduler-this-august-stronger-security-smarter-scheduling-and-2fle" class="crayons-story__hidden-navigation-link"&gt;What’s New in Apache DolphinScheduler This August: Stronger Security, Smarter Scheduling, and Better Stability&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" alt="chen_debra_3060b21d12b1b0 profile" class="crayons-avatar__image" width="260" height="231"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/chen_debra_3060b21d12b1b0" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Chen Debra
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Chen Debra
                
                
              
              &lt;div id="story-author-preview-content-4630561" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/chen_debra_3060b21d12b1b0" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F1533306%2Fc0ea3a94-ba17-47c8-9304-4571fb1adaf9.png" class="crayons-avatar__image" alt="" width="260" height="231"&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Chen Debra&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/whats-new-in-apache-dolphinscheduler-this-august-stronger-security-smarter-scheduling-and-2fle" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Sep 11&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/whats-new-in-apache-dolphinscheduler-this-august-stronger-security-smarter-scheduling-and-2fle" id="article-link-4630561"&gt;
          What’s New in Apache DolphinScheduler This August: Stronger Security, Smarter Scheduling, and Better Stability
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apachedolphinscheduler"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apachedolphinscheduler&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/dataengineering"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;dataengineering&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/github"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;github&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/chen_debra_3060b21d12b1b0/whats-new-in-apache-dolphinscheduler-this-august-stronger-security-smarter-scheduling-and-2fle#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            10 min read
          &lt;/small&gt;
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>What’s New in Apache DolphinScheduler This August: Stronger Security, Smarter Scheduling, and Better Stability</title>
      <dc:creator>Chen Debra</dc:creator>
      <pubDate>Fri, 11 Sep 2026 09:12:54 +0000</pubDate>
      <link>https://dev.to/chen_debra_3060b21d12b1b0/whats-new-in-apache-dolphinscheduler-this-august-stronger-security-smarter-scheduling-and-2fle</link>
      <guid>https://dev.to/chen_debra_3060b21d12b1b0/whats-new-in-apache-dolphinscheduler-this-august-stronger-security-smarter-scheduling-and-2fle</guid>
      <description>&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F1mfpxaxfxbxe0k48jdv2.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F1mfpxaxfxbxe0k48jdv2.jpg" width="800" height="647"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;The Apache DolphinScheduler August Monthly Report is here! Over the past month, the community continued to move the project forward with new improvements across functionality, performance, stability, and ecosystem development. Let’s take a closer look at the updates that stood out in August. And as always, a huge thank-you to everyone who contributed code, shared feedback, and supported the community. Every contribution helps DolphinScheduler continue to evolve!&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h2&gt;
  
  
  📊 August at a Glance
&lt;/h2&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Metric&lt;/th&gt;
&lt;th&gt;Value&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;🚀 PRs Merged&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;23&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;👥 Contributors&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;10&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;➕ Lines Added&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;+3,702&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;➖ Lines Deleted&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;-1,327&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;🔀 Net Lines Changed&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;+2,375&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;📁 Modules Touched&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;8&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;📝 Documentation Files Touched&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;5&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;🧪 Test Files Touched&lt;/td&gt;
&lt;td&gt;&lt;strong&gt;28&lt;/strong&gt;&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;h2&gt;
  
  
  🏆 Top Contributors
&lt;/h2&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Rank&lt;/th&gt;
&lt;th&gt;GitHub Username&lt;/th&gt;
&lt;th&gt;Primary Contribution Area&lt;/th&gt;
&lt;th&gt;PRs&lt;/th&gt;
&lt;th&gt;+Lines&lt;/th&gt;
&lt;th&gt;-Lines&lt;/th&gt;
&lt;th&gt;Overall Score&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;🥇&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;Tests&lt;/td&gt;
&lt;td&gt;10&lt;/td&gt;
&lt;td&gt;1602&lt;/td&gt;
&lt;td&gt;1001&lt;/td&gt;
&lt;td&gt;88.59&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;🥈&lt;/td&gt;
&lt;td&gt;@njnu-seafish&lt;/td&gt;
&lt;td&gt;Performance&lt;/td&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;td&gt;818&lt;/td&gt;
&lt;td&gt;288&lt;/td&gt;
&lt;td&gt;29.09&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;🥉&lt;/td&gt;
&lt;td&gt;@SEPURI-SAI-KRISHNA&lt;/td&gt;
&lt;td&gt;Debugging &amp;amp; Fixes&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;229&lt;/td&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;td&gt;18.77&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;4.&lt;/td&gt;
&lt;td&gt;@kittimzhe&lt;/td&gt;
&lt;td&gt;Documentation&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;14.02&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;5.&lt;/td&gt;
&lt;td&gt;@nikhiln64&lt;/td&gt;
&lt;td&gt;Debugging &amp;amp; Fixes&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;344&lt;/td&gt;
&lt;td&gt;8&lt;/td&gt;
&lt;td&gt;10.16&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;6.&lt;/td&gt;
&lt;td&gt;@hellodml&lt;/td&gt;
&lt;td&gt;Debugging &amp;amp; Fixes&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;136&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;9.45&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;7.&lt;/td&gt;
&lt;td&gt;@zhang-arvin&lt;/td&gt;
&lt;td&gt;Debugging &amp;amp; Fixes&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;10&lt;/td&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;td&gt;9.04&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;8.&lt;/td&gt;
&lt;td&gt;@liang-wenjie&lt;/td&gt;
&lt;td&gt;Tests&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;542&lt;/td&gt;
&lt;td&gt;7&lt;/td&gt;
&lt;td&gt;7.82&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;9.&lt;/td&gt;
&lt;td&gt;@hiSandog&lt;/td&gt;
&lt;td&gt;Tests&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;17&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;6.06&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;10.&lt;/td&gt;
&lt;td&gt;@SbloodyS&lt;/td&gt;
&lt;td&gt;Architecture &amp;amp; Engineering&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;0&lt;/td&gt;
&lt;td&gt;11&lt;/td&gt;
&lt;td&gt;6.01&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;h2&gt;
  
  
  🔄 Code Changes
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Breakdown by Category: Features / Performance / Bug Fixes / Architecture
&lt;/h3&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Category&lt;/th&gt;
&lt;th&gt;PRs&lt;/th&gt;
&lt;th&gt;Share&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Features&lt;/td&gt;
&lt;td&gt;6&lt;/td&gt;
&lt;td&gt;26.1%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Performance Improvements&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;8.7%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Bug Fixes&lt;/td&gt;
&lt;td&gt;9&lt;/td&gt;
&lt;td&gt;39.1%&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Architecture Improvements&lt;/td&gt;
&lt;td&gt;6&lt;/td&gt;
&lt;td&gt;26.1%&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;&lt;strong&gt;Bug fixes accounted for the largest share of this month’s changes, at 39.1% (9 PRs).&lt;/strong&gt; Combined with 6 feature PRs and 2 performance improvements, the overall development rhythm was clear: &lt;strong&gt;stability first, with a steady stream of new capabilities.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  📦 Key Modules
&lt;/h3&gt;

&lt;blockquote&gt;
&lt;p&gt;Ranked by the number of PR touches.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Rank&lt;/th&gt;
&lt;th&gt;Module&lt;/th&gt;
&lt;th&gt;PR Touches&lt;/th&gt;
&lt;th&gt;Lines Added&lt;/th&gt;
&lt;th&gt;Lines Deleted&lt;/th&gt;
&lt;th&gt;Net Change&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;1.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-api&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;82&lt;/td&gt;
&lt;td&gt;+2,132&lt;/td&gt;
&lt;td&gt;-1,016&lt;/td&gt;
&lt;td&gt;+1,116&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;2.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-dao&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;25&lt;/td&gt;
&lt;td&gt;+307&lt;/td&gt;
&lt;td&gt;-201&lt;/td&gt;
&lt;td&gt;+106&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;3.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;docs&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;14&lt;/td&gt;
&lt;td&gt;+68&lt;/td&gt;
&lt;td&gt;-48&lt;/td&gt;
&lt;td&gt;+20&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;4.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-ui&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;10&lt;/td&gt;
&lt;td&gt;+74&lt;/td&gt;
&lt;td&gt;-27&lt;/td&gt;
&lt;td&gt;+47&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;5.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-master&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;9&lt;/td&gt;
&lt;td&gt;+374&lt;/td&gt;
&lt;td&gt;-12&lt;/td&gt;
&lt;td&gt;+362&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;6.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-common&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;6&lt;/td&gt;
&lt;td&gt;+64&lt;/td&gt;
&lt;td&gt;-6&lt;/td&gt;
&lt;td&gt;+58&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;7.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-scheduler-plugin&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;6&lt;/td&gt;
&lt;td&gt;+229&lt;/td&gt;
&lt;td&gt;-4&lt;/td&gt;
&lt;td&gt;+225&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;8.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;dolphinscheduler-task-plugin&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;+305&lt;/td&gt;
&lt;td&gt;-2&lt;/td&gt;
&lt;td&gt;+303&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;9.&lt;/td&gt;
&lt;td&gt;&lt;code&gt;misc&lt;/code&gt;&lt;/td&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;td&gt;+149&lt;/td&gt;
&lt;td&gt;-11&lt;/td&gt;
&lt;td&gt;+138&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;&lt;strong&gt;What the module breakdown tells us:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;🏛️ &lt;strong&gt;API (15 touches)&lt;/strong&gt; was the clear focus of this month’s changes, with 15/23 PRs involving the API module. Most of the work centered on &lt;strong&gt;strengthening authorization and removing obsolete APIs&lt;/strong&gt;.&lt;/li&gt;
&lt;li&gt;🏗️ &lt;strong&gt;DAO (7 touches)&lt;/strong&gt; mainly supported the API changes, including query optimization by excluding large text fields and updates to authorization query logic.&lt;/li&gt;
&lt;li&gt;🎨 &lt;strong&gt;UI (5 touches)&lt;/strong&gt; saw several touchpoints, but most were relatively small changes (+49 net lines), serving primarily as supporting updates.&lt;/li&gt;
&lt;li&gt;🔧 &lt;strong&gt;Master (4 touches)&lt;/strong&gt; focused on scheduling stability, including task retries and failure recovery.&lt;/li&gt;
&lt;/ul&gt;

&lt;h2&gt;
  
  
  🎯 8 Changes Users Will Notice Most
&lt;/h2&gt;

&lt;blockquote&gt;
&lt;p&gt;The following updates are ranked by their potential user impact and focus on the changes that matter most in real-world deployments.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h3&gt;
  
  
  1. 🔐 Stronger Permissions and Security — 7 PRs
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18561" rel="noopener noreferrer"&gt;#18561&lt;/a&gt; — [Fix-18559][API] Align workflow mutations with project write permissions (#18561)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @ruanwenjun&lt;/li&gt;
&lt;li&gt;Change size: +426 / -128 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 7&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; Tenant isolation is now stricter, sensitive operations are better protected, and the overall security and compliance posture is stronger. Seven PRs this month focused on strengthening the permission model, covering project write-permission checks, cross-project authorization for sub-workflows, datasource and cluster authorization, Actuator endpoint authentication, user-list data masking, and authorization API optimization. These changes are particularly important for enterprise and multi-tenant deployments.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Key scenarios to verify:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;✅ When creating, updating, or deleting a workflow, does the system strictly verify project-level &lt;strong&gt;write&lt;/strong&gt; permissions?&lt;/li&gt;
&lt;li&gt;✅ When referencing a sub-workflow, can the system verify that the user also has permission to access the &lt;strong&gt;referenced workflow&lt;/strong&gt;, preventing unauthorized access across projects?&lt;/li&gt;
&lt;li&gt;✅ When a task definition references a datasource, does the system verify that the current user has access to that datasource?&lt;/li&gt;
&lt;li&gt;✅ Do cluster query APIs enforce permission checks consistently?&lt;/li&gt;
&lt;li&gt;✅ Are user-list responses properly masked, with permissions appropriately restricted?&lt;/li&gt;
&lt;li&gt;✅ Do sensitive Actuator endpoints require authentication before they can be accessed?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Full list of related PRs:&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;PR #&lt;/th&gt;
&lt;th&gt;Title&lt;/th&gt;
&lt;th&gt;Author&lt;/th&gt;
&lt;th&gt;Diff&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18561" rel="noopener noreferrer"&gt;#18561&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18559][API] Align workflow mutations with project write permissions (#18561)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+426/-128&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18597" rel="noopener noreferrer"&gt;#18597&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18596][API] Enforce permission checks for sub-workflow references (#18597)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+450/-0&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18566" rel="noopener noreferrer"&gt;#18566&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18565][API] Validate datasource access for task definitions (#18566)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+342/-17&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18564" rel="noopener noreferrer"&gt;#18564&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18563][API] Refine datasource authorization list APIs (#18564)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+102/-62&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18583" rel="noopener noreferrer"&gt;#18583&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18582][Authentication] Align actuator endpoint matching (#18583)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+149/-11&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18560" rel="noopener noreferrer"&gt;#18560&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18558][API] Harden user list access and responses (#18560)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+65/-28&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18590" rel="noopener noreferrer"&gt;#18590&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18589][API] Align cluster query permissions (#18590)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+59/-20&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;



&lt;h3&gt;
  
  
  2. 🛡️ Stability and Bug Fixes — 4 PRs
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18573" rel="noopener noreferrer"&gt;#18573&lt;/a&gt; — [Fix-18570][Master] Detect wrapped CommandDuplicateHandleException in bootstrapError (#18570) (#18573)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @hellodml&lt;/li&gt;
&lt;li&gt;Change size: +136 / -1 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 4&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; These fixes address abnormal workflow and task states, directly improving reliability in production. There were 9 bug-fix PRs in total this month, making bug fixing the largest category. The other bug-fix PRs are covered in dedicated sections for permissions, K8s, DataX, and documentation. The four core fixes below focus on issues outside those categories.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Key scenarios to verify:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;✅ &lt;strong&gt;Failed task retries:&lt;/strong&gt; When recreating a failed task instance, is the runtime state correctly reset to prevent stale state from causing retry failures?&lt;/li&gt;
&lt;li&gt;✅ &lt;strong&gt;Retry timing:&lt;/strong&gt; Is the retry scheduled based on &lt;code&gt;endTime + retryInterval&lt;/code&gt; rather than &lt;code&gt;startTime&lt;/code&gt;?&lt;/li&gt;
&lt;li&gt;✅ &lt;strong&gt;Master startup failures:&lt;/strong&gt; Can &lt;code&gt;bootstrapError&lt;/code&gt; correctly identify a &lt;code&gt;CommandDuplicateHandleException&lt;/code&gt; after it has been wrapped?&lt;/li&gt;
&lt;li&gt;✅ &lt;strong&gt;Log messages:&lt;/strong&gt; Has the typo in &lt;code&gt;DataSourceServiceImpl&lt;/code&gt; log messages been corrected to avoid confusion during troubleshooting?&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Full list of related PRs:&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;PR #&lt;/th&gt;
&lt;th&gt;Title&lt;/th&gt;
&lt;th&gt;Author&lt;/th&gt;
&lt;th&gt;Diff&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18573" rel="noopener noreferrer"&gt;#18573&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18570][Master] Detect wrapped CommandDuplicateHandleException in bootstrapError (#18570) (#18573)&lt;/td&gt;
&lt;td&gt;@hellodml&lt;/td&gt;
&lt;td&gt;+136/-1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18541" rel="noopener noreferrer"&gt;#18541&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18540][Master] Reset the runtime state when recreating a failed task instance (#18541)&lt;/td&gt;
&lt;td&gt;@SEPURI-SAI-KRISHNA&lt;/td&gt;
&lt;td&gt;+132/-0&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18539" rel="noopener noreferrer"&gt;#18539&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Fix-18538][Master] Schedule task retry at endTime + retryInterval (#18539)&lt;/td&gt;
&lt;td&gt;@SEPURI-SAI-KRISHNA&lt;/td&gt;
&lt;td&gt;+97/-3&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18581" rel="noopener noreferrer"&gt;#18581&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18580][api] Fix typo in DataSourceServiceImpl log message (#18581)&lt;/td&gt;
&lt;td&gt;@kittimzhe&lt;/td&gt;
&lt;td&gt;+1/-1&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;



&lt;h3&gt;
  
  
  3. ⏰ Scheduling Policies / Missed-Fire Handling — 1 PR
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18464" rel="noopener noreferrer"&gt;#18464&lt;/a&gt; — [DSIP-18454][Scheduler] Add schedule missed fire policy (#18464)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @liang-wenjie&lt;/li&gt;
&lt;li&gt;Change size: +542 / -7 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 1&lt;/li&gt;
&lt;/ul&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Policy&lt;/th&gt;
&lt;th&gt;Behavior&lt;/th&gt;
&lt;th&gt;Best For&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Skip&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Skip missed triggers and wait for the next cron trigger&lt;/td&gt;
&lt;td&gt;When missed runs should simply be skipped to avoid putting additional load on the system&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;FireOnceNow&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Trigger only once immediately and discard the remaining missed runs&lt;/td&gt;
&lt;td&gt;When only the most recent missed run needs to be executed&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;FireAll&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Execute all missed triggers sequentially according to their original scheduled times&lt;/td&gt;
&lt;td&gt;Financial, reconciliation, and other scenarios where every scheduled run must be executed&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; This is &lt;strong&gt;the biggest new feature of the month&lt;/strong&gt;. If the Master is down or the scheduling thread is blocked and cron triggers are missed, you can now configure one of three policies to determine how DolphinScheduler handles those missed triggers.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Key scenarios to verify:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;✅ When creating or editing a schedule, does the UI display the &lt;strong&gt;Missed Fire Policy&lt;/strong&gt; dropdown?&lt;/li&gt;
&lt;li&gt;✅ Simulate a two-hour Master outage by stopping the Master process and starting it again. Do the different policies behave as expected?&lt;/li&gt;
&lt;li&gt;✅ Does the database upgrade script (&lt;code&gt;3.5.0_schema&lt;/code&gt;) execute correctly, and does the &lt;code&gt;t_ds_schedule&lt;/code&gt; table contain the newly added field?&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  4. ⚡ Performance Improvements — 2 PRs
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18444" rel="noopener noreferrer"&gt;#18444&lt;/a&gt; — [Improvement-18443][API&amp;amp;DAO] Optimize WorkflowInstanceMapper to exclude large text fields from list queries (#18444)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @njnu-seafish&lt;/li&gt;
&lt;li&gt;Change size: +790 / -268 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 2&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; Workflow-instance and task-instance list queries no longer fetch large &lt;code&gt;TEXT&lt;/code&gt; fields such as &lt;code&gt;global_params&lt;/code&gt; and &lt;code&gt;process_instance_json&lt;/code&gt; unnecessarily. This can significantly improve &lt;strong&gt;list-page response times, database I/O, and memory usage&lt;/strong&gt;, with particularly noticeable benefits in large-scale deployments.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Key scenarios to verify:&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;✅ Compare workflow-instance list-page response times before and after the optimization, especially with 1,000+ instances.&lt;/li&gt;
&lt;li&gt;✅ Open an individual workflow instance and verify that large fields such as global parameters are still displayed correctly. The detail API continues to retrieve them.&lt;/li&gt;
&lt;li&gt;✅ Verify that list-page export, search, and other functions continue to work as expected.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Full list of related PRs:&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;PR #&lt;/th&gt;
&lt;th&gt;Title&lt;/th&gt;
&lt;th&gt;Author&lt;/th&gt;
&lt;th&gt;Diff&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18444" rel="noopener noreferrer"&gt;#18444&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18443][API&amp;amp;DAO] Optimize WorkflowInstanceMapper to exclude large text fields from list queries (#18444)&lt;/td&gt;
&lt;td&gt;@njnu-seafish&lt;/td&gt;
&lt;td&gt;+790/-268&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18442" rel="noopener noreferrer"&gt;#18442&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18441][API&amp;amp;DAO] Optimize TaskInstanceMapper to exclude large text fields from list queries (#18442)&lt;/td&gt;
&lt;td&gt;@njnu-seafish&lt;/td&gt;
&lt;td&gt;+17/-5&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;



&lt;h3&gt;
  
  
  5. 🔌 Task Types and New Data Sources — 1 PR
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18434" rel="noopener noreferrer"&gt;#18434&lt;/a&gt; — [Fix-18389][DataX] Read job definition from attached resource file when custom json is empty (#18434)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @nikhiln64&lt;/li&gt;
&lt;li&gt;Change size: +344 / -8 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 1&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; Previously, DataX users had to paste JSON content into the custom JSON field in the UI and could not reuse files from the Resource Center. The new behavior allows DolphinScheduler to &lt;strong&gt;automatically read the job definition from a resource file when the custom JSON field is empty&lt;/strong&gt;, bringing the experience in line with how SQL tasks load SQL from resource files.&lt;/p&gt;

&lt;h3&gt;
  
  
  6. 🧹 API Cleanup and Engineering Improvements — 5 PRs
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18569" rel="noopener noreferrer"&gt;#18569&lt;/a&gt; — [Improvement-18568][API] Remove obsolete task update-with-upstream API (#18569)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @ruanwenjun&lt;/li&gt;
&lt;li&gt;Change size: +2 / -507 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 5&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; This month, the community removed three groups of obsolete APIs covering cluster query-by-code, task update-with-upstream, and dynamic sub-workflow functionality, while also updating &lt;code&gt;incompatible.md&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;For &lt;strong&gt;users upgrading from an earlier version&lt;/strong&gt;, this is an important area to review. If your organization has a custom frontend or automation scripts that rely on any of these legacy APIs, make sure to update them before upgrading.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Full list of related PRs:&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;PR #&lt;/th&gt;
&lt;th&gt;Title&lt;/th&gt;
&lt;th&gt;Author&lt;/th&gt;
&lt;th&gt;Diff&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18569" rel="noopener noreferrer"&gt;#18569&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18568][API] Remove obsolete task update-with-upstream API (#18569)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+2/-507&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18557" rel="noopener noreferrer"&gt;#18557&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Improvement-18556][API] Remove obsolete dynamic sub-workflow API (#18557)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+2/-138&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18584" rel="noopener noreferrer"&gt;#18584&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Chore][API] Remove obsolete cluster query-by-code API (#18584)&lt;/td&gt;
&lt;td&gt;@ruanwenjun&lt;/td&gt;
&lt;td&gt;+5/-90&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18408" rel="noopener noreferrer"&gt;#18408&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Chore][Common] Handle parentless paths in FileUtils (#18408)&lt;/td&gt;
&lt;td&gt;@hiSandog&lt;/td&gt;
&lt;td&gt;+17/-1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18599" rel="noopener noreferrer"&gt;#18599&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Chore] Remove unused code- #18599 (#18599)&lt;/td&gt;
&lt;td&gt;@SbloodyS&lt;/td&gt;
&lt;td&gt;+0/-11&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;



&lt;h3&gt;
  
  
  7. ☸️ K8s and Cloud-Native Deployment — 1 PR
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18574" rel="noopener noreferrer"&gt;#18574&lt;/a&gt; — [Fix-17883] Fix K8s Alert HTTP test sending failed by using IP for non-StatefulSet pods (#17883) (#18574)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @zhang-arvin&lt;/li&gt;
&lt;li&gt;Change size: +10 / -3 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 1&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; For non-StatefulSet Pods in K8s, such as Pods managed by a Deployment, HTTP alert-instance testing could previously fail because the endpoint was resolved using the hostname. The fix switches to IP-based addressing, improving the reliability of alert channels in K8s deployments.&lt;/p&gt;

&lt;h3&gt;
  
  
  8. 📚 Documentation and Example Improvements — 2 PRs
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Representative PR: &lt;a href="https://github.com/apache/dolphinscheduler/pull/18478" rel="noopener noreferrer"&gt;#18478&lt;/a&gt; — [Doc-18474][Upgrade] Fix zh/en incompatible upgrade docs out of sync (#18478)&lt;/strong&gt;&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Author: @njnu-seafish&lt;/li&gt;
&lt;li&gt;Change size: +11 / -15 lines&lt;/li&gt;
&lt;li&gt;PRs in this category this month: 2&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;strong&gt;Why it matters:&lt;/strong&gt; Broken links in the datasource and configuration documentation were fixed, while the Chinese and English upgrade-incompatibility documentation was brought back into alignment. These updates help reduce confusion and prevent issues caused by outdated or inconsistent documentation.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Full list of related PRs:&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;PR #&lt;/th&gt;
&lt;th&gt;Title&lt;/th&gt;
&lt;th&gt;Author&lt;/th&gt;
&lt;th&gt;Diff&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18478" rel="noopener noreferrer"&gt;#18478&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Doc-18474][Upgrade] Fix zh/en incompatible upgrade docs out of sync (#18478)&lt;/td&gt;
&lt;td&gt;@njnu-seafish&lt;/td&gt;
&lt;td&gt;+11/-15&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;a href="https://github.com/apache/dolphinscheduler/pull/18578" rel="noopener noreferrer"&gt;#18578&lt;/a&gt;&lt;/td&gt;
&lt;td&gt;[Doc-18579] Fix malformed links in datasource and configuration docs (#18578)&lt;/td&gt;
&lt;td&gt;@kittimzhe&lt;/td&gt;
&lt;td&gt;+3/-3&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;h2&gt;
  
  
  ⚠️ Upgrade and Validation Recommendations
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Risk Assessment
&lt;/h3&gt;

&lt;p&gt;&lt;strong&gt;Overall Risk Level: 🟠 Medium-High&lt;/strong&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Risk Area&lt;/th&gt;
&lt;th&gt;Assessment&lt;/th&gt;
&lt;th&gt;Details&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;Core Module Changes&lt;/td&gt;
&lt;td&gt;⚠️ Yes&lt;/td&gt;
&lt;td&gt;The API module was touched by 15/23 PRs; permission-related APIs require particular attention during regression testing&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Number of Bug Fixes&lt;/td&gt;
&lt;td&gt;9&lt;/td&gt;
&lt;td&gt;A relatively high number of bug fixes; pay close attention to task retry and failure-recovery scenarios&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Permission/Security Changes&lt;/td&gt;
&lt;td&gt;7&lt;/td&gt;
&lt;td&gt;⚠️ Significant changes to the permission model require comprehensive authorization testing&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;K8s/Deployment Changes&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;⚠️ Deployment-related logic was changed; Helm/Docker deployments should be verified&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;UI Changes&lt;/td&gt;
&lt;td&gt;Yes&lt;/td&gt;
&lt;td&gt;⚠️ Frontend changes require smoke testing of key pages&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Database Schema&lt;/td&gt;
&lt;td&gt;⚠️ Yes&lt;/td&gt;
&lt;td&gt;DSIP-18464 adds new DDL; verify that the upgrade script is executed correctly&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;h3&gt;
  
  
  Before You Upgrade
&lt;/h3&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;📦 Back up the database:&lt;/strong&gt; Before upgrading, create a full backup of the DolphinScheduler metadata database using &lt;code&gt;mysqldump&lt;/code&gt; or &lt;code&gt;pg_dump&lt;/code&gt;.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;📋 Check the DDL scripts:&lt;/strong&gt; Verify that the new scripts under &lt;code&gt;dolphinscheduler-dao/src/main/resources/sql/upgrade/3.5.0_schema/&lt;/code&gt; are included in the upgrade process.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;📝 Review incompatible changes:&lt;/strong&gt; Pay particular attention to the four groups of API removals marked this month in &lt;a href="https://github.com/apache/dolphinscheduler/blob/dev/docs/docs/en/guide/upgrade/incompatible.md" rel="noopener noreferrer"&gt;&lt;code&gt;docs/docs/en/guide/upgrade/incompatible.md&lt;/code&gt;&lt;/a&gt;.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;🧪 Check custom integrations:&lt;/strong&gt; If you use OpenAPI, search your codebase to make sure none of these three removed APIs are still being called:&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;code&gt;cluster/query-by-code&lt;/code&gt; (PR #18584)&lt;/li&gt;
&lt;li&gt;
&lt;code&gt;task/update-with-upstream&lt;/code&gt; (PR #18569)&lt;/li&gt;
&lt;li&gt;Dynamic sub-workflow APIs (PR #18557)&lt;/li&gt;
&lt;/ul&gt;

&lt;ol&gt;
&lt;li&gt;
&lt;strong&gt;💾 Back up configuration files:&lt;/strong&gt; Back up all configuration files under the &lt;code&gt;conf/&lt;/code&gt; directory.&lt;/li&gt;
&lt;/ol&gt;

&lt;h3&gt;
  
  
  Common Issues and Quick Checks
&lt;/h3&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Symptom&lt;/th&gt;
&lt;th&gt;Possible Cause&lt;/th&gt;
&lt;th&gt;What to Check&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;A sudden 403 when saving a workflow&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;PR #18561 introduced stricter permission checks&lt;/td&gt;
&lt;td&gt;Verify that the current user has write permission for the target project and that the upstream workflow referenced by the sub-workflow is also within the user's authorized scope&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;A sub-workflow cannot be referenced&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;PR #18597 added cross-project permission checks&lt;/td&gt;
&lt;td&gt;Confirm that the user has the required permissions for the project containing the sub-workflow&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Binding a datasource returns 403&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;PR #18566 added datasource authorization checks&lt;/td&gt;
&lt;td&gt;Grant the user access to the corresponding datasource under &lt;strong&gt;Datasource Authorization&lt;/strong&gt;
&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;HTTP alert testing fails in a K8s deployment&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;Logic changed by PR #18574&lt;/td&gt;
&lt;td&gt;Check Pod network policies and verify in the logs whether the connection is being made using a hostname or an IP address&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Missed-fire behavior is not what you expected after upgrading&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;DSIP-18464 defaults to the &lt;strong&gt;Skip&lt;/strong&gt; policy&lt;/td&gt;
&lt;td&gt;Check the value of &lt;code&gt;missed_fire_policy&lt;/code&gt;; switch the policy manually if missed runs need to be executed&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Task retries trigger immediately instead of respecting the interval&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;PR #18539 changed the calculation to use &lt;code&gt;endTime&lt;/code&gt;
&lt;/td&gt;
&lt;td&gt;Verify that retry time is calculated as &lt;strong&gt;task end time + retryInterval&lt;/strong&gt;, rather than start time&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;&lt;strong&gt;Workflow-instance lists load slowly or return errors after upgrading&lt;/strong&gt;&lt;/td&gt;
&lt;td&gt;VO fields no longer match after large-field exclusions&lt;/td&gt;
&lt;td&gt;Check whether the frontend version was upgraded at the same time, or roll back temporarily to isolate the changes from #18444/#18442&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;h2&gt;
  
  
  🙏 Thank You to Our Contributors
&lt;/h2&gt;

&lt;p&gt;A huge thank-you to the &lt;strong&gt;10 contributors&lt;/strong&gt; who contributed code to Apache DolphinScheduler in August 2026:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;@ruanwenjun, @njnu-seafish, @SEPURI-SAI-KRISHNA, @kittimzhe, @nikhiln64, @hellodml, @zhang-arvin, @liang-wenjie, @hiSandog, and @SbloodyS&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Every PR helps DolphinScheduler continue to evolve in &lt;strong&gt;stability, usability, and ecosystem growth&lt;/strong&gt;. 💪&lt;/p&gt;

</description>
      <category>apachedolphinscheduler</category>
      <category>datascience</category>
      <category>dataengineering</category>
      <category>github</category>
    </item>
  </channel>
</rss>
