<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>DEV Community: Apache SeaTunnel</title>
    <description>The latest articles on DEV Community by Apache SeaTunnel (@seatunnel).</description>
    <link>https://dev.to/seatunnel</link>
    <image>
      <url>https://media2.dev.to/dynamic/image/width=90,height=90,fit=cover,gravity=auto,format=auto/https:%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg</url>
      <title>DEV Community: Apache SeaTunnel</title>
      <link>https://dev.to/seatunnel</link>
    </image>
    <atom:link rel="self" type="application/rss+xml" href="https://dev.to/feed/seatunnel"/>
    <language>en</language>
    <item>
      <title>Still maintaining hundreds of Python ETL scripts? 🤯 Build pipelines, not infrastructure. Let Apache SeaTunnel handle the runtime while you focus on data. 🚀</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 17 Jul 2026 02:29:53 +0000</pubDate>
      <link>https://dev.to/seatunnel/still-maintaining-hundreds-of-python-etl-scripts-build-pipelines-not-infrastructure-let-apache-40j0</link>
      <guid>https://dev.to/seatunnel/still-maintaining-hundreds-of-python-etl-scripts-build-pipelines-not-infrastructure-let-apache-40j0</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g" class="crayons-story__hidden-navigation-link"&gt;From Python Script Hell to a Modern Data Integration Framework&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/seatunnel" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" alt="seatunnel profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/seatunnel" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Apache SeaTunnel
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Apache SeaTunnel
                
              
              &lt;div id="story-author-preview-content-4161952" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/seatunnel" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Apache SeaTunnel&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 17&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g" id="article-link-4161952"&gt;
          From Python Script Hell to a Modern Data Integration Framework
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
          &lt;a href="https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left"&gt;
            &lt;div class="multiple_reactions_aggregate"&gt;
              &lt;span class="multiple_reactions_icons_container"&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/exploding-head-daceb38d627e6ae9b730f36a1e390fca556a4289d5a41abb2c35068ad3e2c4b5.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/multi-unicorn-b44d6f8c23cdd00964192bedc38af3e82463978aa611b4365bd33a0f1f4f3e97.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
                  &lt;span class="crayons_icon_container"&gt;
                    &lt;img src="https://assets.dev.to/assets/sparkle-heart-5f9bee3767e18deb1bb725290cb151c25234768a0e9a2bd39370c382d02920cf.svg" width="18" height="18"&gt;
                  &lt;/span&gt;
              &lt;/span&gt;
              &lt;span class="aggregate_reactions_counter"&gt;5&lt;span class="hidden s:inline"&gt;&amp;nbsp;reactions&lt;/span&gt;&lt;/span&gt;
            &lt;/div&gt;
          &lt;/a&gt;
            &lt;a href="https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            11 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>From Python Script Hell to a Modern Data Integration Framework</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 17 Jul 2026 02:29:25 +0000</pubDate>
      <link>https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g</link>
      <guid>https://dev.to/seatunnel/from-python-script-hell-to-a-modern-data-integration-framework-1a3g</guid>
      <description>&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fjo7s75ydkwg7w0gm1kcb.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fjo7s75ydkwg7w0gm1kcb.jpg" width="800" height="447"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Every Data Team Eventually Ends Up with a Collection of Python Scripts
&lt;/h2&gt;

&lt;p&gt;Almost every enterprise data platform follows a similar evolution.&lt;/p&gt;

&lt;p&gt;At the beginning of a project, data ingestion requirements are usually straightforward. Business systems need to synchronize MySQL data into a data warehouse. Marketing teams want to periodically retrieve campaign data from third-party REST APIs. Logging systems consume real-time messages from Kafka before writing them into ClickHouse or Elasticsearch. For these scenarios, Python naturally becomes the preferred language for most data engineers because of its rich ecosystem and low development cost. With just a few dozen or a few hundred lines of code, a complete data synchronization task can be implemented.&lt;/p&gt;

&lt;p&gt;At this stage, the approach works perfectly well. Development is fast, deployment is simple, and a new requirement usually means adding another Python file that can be delivered to production within a short time.&lt;/p&gt;

&lt;p&gt;The real challenge emerges as the business continues to grow.&lt;/p&gt;

&lt;p&gt;As more data sources are introduced, organizations gradually accumulate dozens, hundreds, or even thousands of synchronization jobs. Git repositories become filled with Python scripts written by different developers, following different coding styles, and using different execution models. When another synchronization task is required, the easiest solution is rarely to redesign the architecture. Instead, developers simply copy an existing script, modify the database connection, update the SQL statements and destination table, and deploy yet another program.&lt;/p&gt;

&lt;p&gt;Over time, the data platform gradually evolves into nothing more than a large collection of scripts.&lt;/p&gt;

&lt;p&gt;Many engineering teams refer to this phenomenon as &lt;strong&gt;"Python Script Hell."&lt;/strong&gt; However, the problem is not Python itself. The real issue is that organizations have unintentionally treated Python scripts as their data integration platform.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fou9g0e6ewvqs1x4mqfy4.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fou9g0e6ewvqs1x4mqfy4.jpg" width="800" height="533"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  What Enterprises Keep Rebuilding Is Actually a Runtime Framework
&lt;/h2&gt;

&lt;p&gt;If you take a closer look at these Python scripts, you'll find that only a small portion of the code is truly business-specific.&lt;/p&gt;

&lt;p&gt;A typical data synchronization task usually consists of just three steps: reading data, transforming data, and writing data. Whether the task synchronizes orders, customer information, or application logs, the business logic usually differs only in the data source, transformation rules, and destination system.&lt;/p&gt;

&lt;p&gt;However, a production-ready Python script involves much more than these three steps. Developers must establish database connections, manage thread pools, handle network failures and retry logic, persist synchronization checkpoints, generate logs and monitoring metrics, prevent duplicate writes, support task recovery, and properly release resources throughout the execution lifecycle.&lt;/p&gt;

&lt;p&gt;In other words, while organizations appear to be developing new data synchronization jobs every day, what they are actually rebuilding repeatedly is not business logic, but an entire set of runtime infrastructure.&lt;/p&gt;

&lt;p&gt;More importantly, this infrastructure is almost identical across different projects.&lt;/p&gt;

&lt;p&gt;Connection management, thread scheduling, retry mechanisms, checkpointing, state recovery, logging, monitoring, and write consistency are implemented repeatedly in one project after another. These capabilities should be provided by a common platform, yet they are instead embedded in individual Python scripts, resulting in duplicated code and increasing maintenance costs.&lt;/p&gt;

&lt;p&gt;Therefore, the real challenge is not reducing the use of Python. It is stopping the repeated implementation of runtime capabilities that should belong to a shared framework.&lt;/p&gt;

&lt;p&gt;This is precisely the problem that Apache SeaTunnel is designed to solve.&lt;/p&gt;

&lt;h2&gt;
  
  
  Apache SeaTunnel: Replacing Repetitive Frameworks, Not Python
&lt;/h2&gt;

&lt;p&gt;When people first encounter Apache SeaTunnel, they often think of it as just another ETL tool. They assume that it simply replaces Python code with configuration files.&lt;/p&gt;

&lt;p&gt;However, that is only the surface.&lt;/p&gt;

&lt;p&gt;What SeaTunnel fundamentally changes is not the programming language—it redefines the boundary between &lt;strong&gt;business logic&lt;/strong&gt; and &lt;strong&gt;runtime capabilities&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;In a traditional Python-based approach, a script is responsible not only for describing how data flows, but also for handling every aspect of execution, including connection management, task scheduling, exception recovery, state management, and many other runtime concerns. Every script effectively becomes a miniature data integration platform with its own lifecycle.&lt;/p&gt;

&lt;p&gt;SeaTunnel takes a completely different approach.&lt;/p&gt;

&lt;p&gt;Developers are responsible only for defining the data pipeline, while the runtime is responsible for executing that pipeline reliably, efficiently, and at scale.&lt;/p&gt;

&lt;p&gt;As a result, a data synchronization task only needs to answer three questions:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Where does the data come from? (&lt;strong&gt;Source&lt;/strong&gt;)&lt;/li&gt;
&lt;li&gt;How should the data be processed? (&lt;strong&gt;Transform&lt;/strong&gt;)&lt;/li&gt;
&lt;li&gt;Where should the data be written? (&lt;strong&gt;Sink&lt;/strong&gt;)&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Everything else that is related to execution is handled by the unified runtime.&lt;/p&gt;

&lt;p&gt;This is the fundamental difference between SeaTunnel and traditional Python scripts: &lt;strong&gt;developers build pipelines, while the runtime handles everything else.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Source, Transform, and Sink: Standardizing More Than Configuration
&lt;/h3&gt;

&lt;p&gt;Many articles describe Source, Transform, and Sink simply as configuration concepts. In reality, these three components are much more than that—they form the core abstraction of the entire SeaTunnel runtime architecture.&lt;/p&gt;

&lt;p&gt;Traditionally, different data synchronization tasks are implemented in completely different ways. Synchronizing data from MySQL requires database access logic, consuming Kafka messages requires managing consumers, and calling REST APIs involves authentication, pagination, and rate limiting. Although all of these tasks ultimately perform the same operation—reading data—each data source requires its own implementation.&lt;/p&gt;

&lt;p&gt;SeaTunnel does not attempt to standardize the underlying systems themselves. Instead, it standardizes the &lt;strong&gt;data flow&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Whether the data comes from MySQL, Oracle, Kafka, MongoDB, Amazon S3, or a REST API, it is represented as a &lt;strong&gt;Source&lt;/strong&gt; once it enters a SeaTunnel pipeline. Whether the data needs SQL processing, field mapping, type conversion, or data cleansing, those operations are handled uniformly through &lt;strong&gt;Transform&lt;/strong&gt;. Finally, regardless of whether the destination is ClickHouse, StarRocks, Iceberg, Kafka, or Elasticsearch, all writes are managed through &lt;strong&gt;Sink&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;The greatest value of this abstraction is that it shifts the developer's focus back to the data itself instead of the implementation details.&lt;/p&gt;

&lt;p&gt;For an enterprise, creating a new synchronization task no longer means copying an existing Python project. It simply means defining another data pipeline. While business logic continues to evolve, the development model remains consistent, providing the foundation for managing data integration at scale.&lt;/p&gt;

&lt;h3&gt;
  
  
  Connector Framework: Turning Connectivity into a Platform Capability
&lt;/h3&gt;

&lt;p&gt;In a traditional Python project, every new data source usually introduces another set of connection logic.&lt;/p&gt;

&lt;p&gt;Connecting to MySQL requires maintaining database drivers and connection pools. Calling REST APIs requires handling authentication tokens, token refresh, and rate limiting. Consuming Kafka messages requires managing consumers, partitions, and offsets. As more data sources are introduced, these implementations become scattered across different scripts. Although they solve essentially the same problem, they are difficult to reuse in practice.&lt;/p&gt;

&lt;p&gt;SeaTunnel encapsulates these capabilities within a unified &lt;strong&gt;Connector Framework&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;A connector is responsible not only for establishing connections, but also for reading data, adapting communication protocols, parsing schemas, and writing data into target systems. All connectors follow the same lifecycle and interface specifications. As a result, developers no longer need to design different connection mechanisms for different systems—they simply select the appropriate connector and provide the necessary configuration.&lt;/p&gt;

&lt;p&gt;More importantly, connector standardization makes the platform continuously extensible.&lt;/p&gt;

&lt;p&gt;When an enterprise needs to integrate another database, cloud service, or storage system, it no longer needs to create another standalone Python project. Instead, it extends the platform by implementing a new connector that operates within the existing framework.&lt;/p&gt;

&lt;p&gt;The runtime remains the same. Only the data access capability is extended.&lt;/p&gt;

&lt;p&gt;This is the fundamental distinction between a &lt;strong&gt;framework&lt;/strong&gt; and a collection of &lt;strong&gt;scripts&lt;/strong&gt;.&lt;/p&gt;

&lt;h3&gt;
  
  
  SeaTunnel Runtime: What Gets Reused Is the Runtime, Not the Code
&lt;/h3&gt;

&lt;p&gt;If Source, Transform, and Sink standardize the data flow, then the &lt;strong&gt;Runtime&lt;/strong&gt; is what transforms SeaTunnel from an ETL tool into a true &lt;strong&gt;Data Integration Framework&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fui3iejpwu32iwi7yo3qk.png" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fui3iejpwu32iwi7yo3qk.png" alt="ST runtime flow" width="800" height="502"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;When writing Python scripts, developers naturally focus on how data is read and written. However, once a synchronization task enters production, reading and writing data is no longer the most challenging part.&lt;/p&gt;

&lt;p&gt;The real challenge is ensuring that the pipeline continues to run reliably over time.&lt;/p&gt;

&lt;p&gt;A production data synchronization job must address a wide range of runtime concerns. Tasks need to be partitioned to improve throughput, multiple jobs need to be scheduled concurrently, failures must be detected and recovered automatically, execution state has to be persisted, duplicate writes must be prevented, and the entire execution process must be observable.&lt;/p&gt;

&lt;p&gt;These problems have very little to do with business logic, yet they determine whether a data platform can operate reliably in production.&lt;/p&gt;

&lt;p&gt;In the traditional model, every Python script implements these capabilities independently, leaving each developer responsible for building and maintaining their own runtime infrastructure.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;SeaTunnel takes a different approach by consolidating these responsibilities into a unified runtime.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fjuq6lzl4mincu5xers82.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fjuq6lzl4mincu5xers82.jpg" width="800" height="534"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;As a result, developers no longer need to build a separate execution environment for every synchronization task. Instead, different data pipelines run on the same runtime, sharing a common execution model and platform capabilities.&lt;/p&gt;

&lt;p&gt;In other words, what is truly reused is not application code, but the runtime itself.&lt;/p&gt;

&lt;h3&gt;
  
  
  Concurrency Scheduling: From Managing Thread Pools to Configuring Parallelism
&lt;/h3&gt;

&lt;p&gt;As data volumes continue to grow, most Python-based projects eventually encounter the same challenge: a single-threaded execution model is no longer sufficient to meet throughput requirements.&lt;/p&gt;

&lt;p&gt;To address this, some teams introduce thread pools, others adopt the &lt;code&gt;multiprocessing&lt;/code&gt; module, while some experiment with &lt;code&gt;asyncio&lt;/code&gt;. Over time, different projects evolve different concurrency models. As the number of scripts increases, thread management, resource contention, and troubleshooting become increasingly difficult.&lt;/p&gt;

&lt;p&gt;The fundamental problem is that concurrency is a runtime concern, not a business concern.&lt;/p&gt;

&lt;p&gt;From a developer's perspective, the real question is &lt;strong&gt;"How quickly should this task complete?"&lt;/strong&gt;, not &lt;strong&gt;"How many threads should be created?", "How should they be scheduled?", or "How should resources be allocated?"&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;SeaTunnel Runtime delegates task partitioning, resource scheduling, and concurrent execution to the execution engine. Developers only need to configure an appropriate level of &lt;strong&gt;Parallelism&lt;/strong&gt; based on workload requirements. The Runtime is responsible for partitioning the pipeline, allocating resources, scheduling execution, and managing the entire task lifecycle, without requiring developers to maintain thread pools or asynchronous frameworks.&lt;/p&gt;

&lt;p&gt;This design significantly reduces development complexity while providing a consistent execution model across the entire platform. When an enterprise needs to improve synchronization performance, it adjusts platform-level configurations rather than modifying hundreds of individual Python scripts.&lt;/p&gt;

&lt;h3&gt;
  
  
  Checkpoint and State: Why State Management Belongs in the Framework
&lt;/h3&gt;

&lt;p&gt;Compared with an initial full synchronization, enterprises are far more concerned with how a task recovers after a failure.&lt;/p&gt;

&lt;p&gt;Consider a synchronization job that has already processed tens of millions of records before unexpectedly terminating because of a network issue or node failure. Restarting the entire job from scratch is both inefficient and likely to produce duplicate data. On the other hand, if each developer maintains synchronization progress independently, different projects inevitably end up using different formats and recovery mechanisms.&lt;/p&gt;

&lt;p&gt;Many Python-based solutions persist the last synchronized primary key, update timestamp, Kafka offset, or file position. Although this approach works for an individual project, state management becomes increasingly fragmented as more synchronization jobs are introduced, making recovery logic difficult to standardize across the platform.&lt;/p&gt;

&lt;p&gt;SeaTunnel treats &lt;strong&gt;Checkpoint&lt;/strong&gt; and &lt;strong&gt;State&lt;/strong&gt; as core runtime capabilities rather than application logic. During execution, the framework periodically persists the execution state. If a task fails, it can automatically recover from the latest valid checkpoint and continue processing without requiring developers to manually maintain offsets, timestamps, or synchronization markers.&lt;/p&gt;

&lt;p&gt;The value of this unified state management extends beyond reducing code. More importantly, it ensures that every synchronization task follows the same recovery mechanism. State becomes a platform-managed resource instead of being scattered across databases, local files, or Redis instances.&lt;/p&gt;

&lt;h3&gt;
  
  
  Retry and Fault Tolerance: Recovery Should Be a Platform Responsibility
&lt;/h3&gt;

&lt;p&gt;Almost every Python script contains similar logic:&lt;/p&gt;

&lt;p&gt;Catch an exception, wait for a few seconds, and try again.&lt;/p&gt;

&lt;p&gt;As systems grow more complex, retry logic becomes increasingly sophisticated. Some tasks retry three times before failing, others retry indefinitely, while some require manual intervention after specific errors. Different developers inevitably implement different fault tolerance strategies.&lt;/p&gt;

&lt;p&gt;When an organization operates hundreds of synchronization jobs, these inconsistencies eventually translate into operational complexity.&lt;/p&gt;

&lt;p&gt;SeaTunnel incorporates &lt;strong&gt;Retry&lt;/strong&gt;, &lt;strong&gt;Failover&lt;/strong&gt;, and &lt;strong&gt;Fault Tolerance&lt;/strong&gt; directly into the Runtime. When a task encounters an exception, the framework determines whether the task should be retried, how it should be recovered, and how it should be rescheduled according to a unified execution policy, rather than leaving these decisions to individual applications.&lt;/p&gt;

&lt;p&gt;For developers, this eliminates the need to repeatedly implement exception handling and retry logic in every synchronization job. For platform operators, it ensures that every task follows the same execution behavior and recovery strategy.&lt;/p&gt;

&lt;p&gt;This is one of the most important distinctions between a framework and a collection of scripts: &lt;strong&gt;fault recovery is a platform capability, not application logic.&lt;/strong&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Write Semantics: Reading Data Is Easy—Writing It Correctly Is the Real Challenge
&lt;/h3&gt;

&lt;p&gt;One of the most frequently overlooked aspects of data synchronization is not reading data, but ensuring that data is written correctly to the target system.&lt;/p&gt;

&lt;p&gt;Consider a task that has written half of its records to the destination database before unexpectedly terminating. If the task restarts and writes those records again, duplicate data may be generated. If it skips the already processed portion incorrectly, data may be lost. To avoid these situations, many Python projects implement their own idempotent write logic, transaction control, or deduplication mechanisms.&lt;/p&gt;

&lt;p&gt;These implementations are not only complex but also difficult to standardize across different projects.&lt;/p&gt;

&lt;p&gt;SeaTunnel moves write semantics into the &lt;strong&gt;Sink Connector&lt;/strong&gt; and the &lt;strong&gt;Runtime&lt;/strong&gt;, allowing write consistency to be managed in a unified manner. Depending on the capabilities of the target system, different Sink connectors provide appropriate consistency guarantees, while the Runtime coordinates the writing process instead of requiring every synchronization program to implement its own strategy.&lt;/p&gt;

&lt;p&gt;As a result, developers only need to specify &lt;strong&gt;where&lt;/strong&gt; the data should be written. The Runtime is responsible for &lt;strong&gt;how&lt;/strong&gt; it is written safely and consistently.&lt;/p&gt;

&lt;p&gt;Once data consistency becomes a platform capability, organizations no longer need to maintain multiple implementations of transaction management or idempotent writes. Instead, they rely on a unified write mechanism shared across the entire platform.&lt;/p&gt;

&lt;h3&gt;
  
  
  Observability: Operating a Platform Instead of Hundreds of Scripts
&lt;/h3&gt;

&lt;p&gt;As the number of synchronization jobs continues to grow, observability becomes an essential capability of any production-grade data platform.&lt;/p&gt;

&lt;p&gt;In a traditional environment, every Python script typically has its own logging format, monitoring mechanism, and alerting strategy. When an issue occurs, operations teams often have to inspect individual log files one by one, and in some cases they may not even know which script is responsible for the failure.&lt;/p&gt;

&lt;p&gt;SeaTunnel Runtime exposes a unified set of operational metrics, including task status, throughput, latency, resource utilization, checkpoint information, and other runtime indicators. These metrics can be integrated with an organization's existing monitoring infrastructure, providing a consistent operational view across all synchronization jobs.&lt;/p&gt;

&lt;p&gt;For operations teams, the focus shifts from managing hundreds of independent programs to operating a single, unified data integration platform.&lt;/p&gt;

&lt;p&gt;This transformation goes beyond improving monitoring. It gives the platform a consistent observability model that supports large-scale operations, troubleshooting, and capacity planning.&lt;/p&gt;

&lt;h2&gt;
  
  
  From Script-Driven Development to Framework-Driven Engineering
&lt;/h2&gt;

&lt;p&gt;Looking back at the evolution of enterprise data platforms, it becomes clear that Python has never been the problem.&lt;/p&gt;

&lt;p&gt;In the early stages of a project, Python scripts enable teams to build data ingestion pipelines quickly, reduce development costs, and deliver business value with minimal overhead. They provide an efficient way to establish the first generation of a data platform.&lt;/p&gt;

&lt;p&gt;The real challenge emerges as the platform grows.&lt;/p&gt;

&lt;p&gt;When the number of synchronization jobs increases from a handful to hundreds, organizations are no longer maintaining only business logic. Instead, they are maintaining hundreds of independent implementations of connection management, concurrency scheduling, state persistence, retry mechanisms, monitoring, and other runtime capabilities.&lt;/p&gt;

&lt;p&gt;Apache SeaTunnel is valuable not because it reduces the amount of code developers need to write, but because it redefines the responsibilities of a modern data integration platform.&lt;/p&gt;

&lt;p&gt;By introducing a unified Source–Transform–Sink model, SeaTunnel standardizes how data pipelines are described. Through a unified Connector Framework, it abstracts the differences among heterogeneous data systems. More importantly, through a unified Runtime, it elevates capabilities such as parallel execution, checkpointing, state management, retry mechanisms, write semantics, and observability from application logic into platform capabilities.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F5ntg049rbjky5k80gmiw.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F5ntg049rbjky5k80gmiw.jpg" width="800" height="533"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;For developers, creating a new synchronization task no longer means copying an existing Python script and modifying it. Instead, it simply means defining another pipeline.&lt;/p&gt;

&lt;p&gt;For enterprises, the maintenance target is no longer hundreds of isolated scripts, but a single, standardized data integration framework.&lt;/p&gt;

&lt;p&gt;The number of data synchronization tasks may continue to grow, but the amount of infrastructure that must be repeatedly implemented and maintained is dramatically reduced.&lt;/p&gt;

&lt;p&gt;This is the most significant value that Apache SeaTunnel brings.&lt;/p&gt;

&lt;p&gt;It is not designed to replace Python. Rather, it allows Python to focus on solving business problems instead of carrying responsibilities that belong to the framework.&lt;/p&gt;

&lt;p&gt;As enterprise data platforms continue to evolve toward greater scale, standardization, and engineering maturity, the transition from script-driven development to framework-driven engineering is no longer simply a tooling upgrade—it is a natural evolution of modern data engineering.&lt;/p&gt;

</description>
    </item>
    <item>
      <title>Replace four data platforms with one! 🚀 See how Tongcheng Travel unified batch &amp; streaming pipelines using Apache SeaTunnel.</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 17 Jul 2026 02:22:44 +0000</pubDate>
      <link>https://dev.to/seatunnel/replace-four-data-platforms-with-one-see-how-tongcheng-travel-unified-batch-streaming-1mgk</link>
      <guid>https://dev.to/seatunnel/replace-four-data-platforms-with-one-see-how-tongcheng-travel-unified-batch-streaming-1mgk</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/seatunnel/from-four-platforms-to-one-how-tongcheng-travel-built-a-unified-data-integration-platform-with-9c3" class="crayons-story__hidden-navigation-link"&gt;From Four Platforms to One: How Tongcheng Travel Built a Unified Data Integration Platform with Apache SeaTunnel&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/seatunnel" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" alt="seatunnel profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/seatunnel" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Apache SeaTunnel
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Apache SeaTunnel
                
              
              &lt;div id="story-author-preview-content-4161912" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/seatunnel" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Apache SeaTunnel&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/seatunnel/from-four-platforms-to-one-how-tongcheng-travel-built-a-unified-data-integration-platform-with-9c3" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 17&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/seatunnel/from-four-platforms-to-one-how-tongcheng-travel-built-a-unified-data-integration-platform-with-9c3" id="article-link-4161912"&gt;
          From Four Platforms to One: How Tongcheng Travel Built a Unified Data Integration Platform with Apache SeaTunnel
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apacheseatunnel"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apacheseatunnel&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/database"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;database&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/opensource"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;opensource&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/seatunnel/from-four-platforms-to-one-how-tongcheng-travel-built-a-unified-data-integration-platform-with-9c3#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            13 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>From Four Platforms to One: How Tongcheng Travel Built a Unified Data Integration Platform with Apache SeaTunnel</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 17 Jul 2026 02:19:14 +0000</pubDate>
      <link>https://dev.to/seatunnel/from-four-platforms-to-one-how-tongcheng-travel-built-a-unified-data-integration-platform-with-9c3</link>
      <guid>https://dev.to/seatunnel/from-four-platforms-to-one-how-tongcheng-travel-built-a-unified-data-integration-platform-with-9c3</guid>
      <description>&lt;p&gt;&lt;strong&gt;From four independent data integration platforms to one unified batch-stream architecture—discover how Tongcheng Travel leveraged Apache SeaTunnel's Zeta Engine to migrate tens of thousands of production jobs with zero business impact.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;For years, Tongcheng Travel's data integration ecosystem evolved into four separate platforms: &lt;strong&gt;Data Transfer&lt;/strong&gt;, &lt;strong&gt;Data Integration&lt;/strong&gt;, &lt;strong&gt;Sqoop&lt;/strong&gt;, and &lt;strong&gt;Apache SeaTunnel&lt;/strong&gt;. While each platform served specific business needs, overlapping functionality, fragmented execution engines, and increasing operational complexity gradually became major challenges.&lt;/p&gt;

&lt;p&gt;To address these issues, Tongcheng Travel adopted &lt;strong&gt;Apache SeaTunnel Zeta Engine&lt;/strong&gt; as the unified foundation for its next-generation data integration platform.&lt;/p&gt;

&lt;p&gt;This article shares the complete migration journey—from automatically translating legacy Sqoop and FlinkSQL jobs without requiring code changes, to building an AI-powered Data Copilot capable of creating data pipelines using natural language, to designing a large-scale data consistency validation framework that guarantees migration accuracy.&lt;/p&gt;

&lt;p&gt;Ultimately, the project successfully consolidated four independent platforms into a unified batch-stream processing architecture while significantly improving operational efficiency, platform maintainability, and future scalability. Looking ahead, the team is continuing to invest in cloud-native architecture and AI-driven intelligent operations.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Meetup Recording&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;a href="https://youtu.be/b224ISVIU7A?si=eOxLZ1xyUAfdOUVs" rel="noopener noreferrer"&gt;https://youtu.be/b224ISVIU7A?si=eOxLZ1xyUAfdOUVs&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  About the Author
&lt;/h2&gt;

&lt;p&gt;&lt;strong&gt;Xiaochen Zhou&lt;/strong&gt; is a Data Platform Engineer at &lt;strong&gt;Tongcheng Travel&lt;/strong&gt; and an &lt;strong&gt;Apache SeaTunnel Committer&lt;/strong&gt;. He has been actively contributing to the SeaTunnel community, focusing on engine evolution, connector development, and large-scale enterprise adoption.&lt;/p&gt;

&lt;h1&gt;
  
  
  Building a Unified Data Integration Platform
&lt;/h1&gt;

&lt;h2&gt;
  
  
  Existing Architecture
&lt;/h2&gt;

&lt;p&gt;Tongcheng Travel previously operated &lt;strong&gt;four independent data integration services&lt;/strong&gt;:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Data Transfer&lt;/li&gt;
&lt;li&gt;Data Integration&lt;/li&gt;
&lt;li&gt;Sqoop&lt;/li&gt;
&lt;li&gt;Apache SeaTunnel&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Although these systems were built for different scenarios over time, they gradually evolved to provide overlapping capabilities, resulting in duplicated functionality, inconsistent architectures, and increasing maintenance costs.&lt;/p&gt;

&lt;p&gt;Specifically:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Data Transfer&lt;/strong&gt; and &lt;strong&gt;Sqoop&lt;/strong&gt; were built on &lt;strong&gt;Flink 1.6&lt;/strong&gt; and &lt;strong&gt;MapReduce&lt;/strong&gt;, respectively. They mainly handled offline synchronization between relational databases and big data systems. However, when processing large-scale database synchronization jobs, these engines generated significant pressure on production databases, affecting the stability of online business systems.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;Data Integration&lt;/strong&gt;, based on &lt;strong&gt;Apache Flink&lt;/strong&gt;, focused on real-time synchronization scenarios, primarily moving operational database data into the company's data lake.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;Meanwhile, the company's healthcare business had already adopted &lt;strong&gt;Apache SeaTunnel Zeta Engine&lt;/strong&gt;. Thanks to its lightweight architecture designed specifically for data integration workloads, SeaTunnel demonstrated superior resource efficiency and synchronization performance.&lt;/p&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Overall, maintaining four different data integration platforms had become increasingly unsustainable.&lt;/p&gt;

&lt;p&gt;The team decided to consolidate all existing services into &lt;strong&gt;a unified data integration platform powered entirely by Apache SeaTunnel&lt;/strong&gt;, providing a single engine capable of handling both batch and streaming workloads.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fwofqpha9i8y0pe5ty8ov.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fwofqpha9i8y0pe5ty8ov.jpg" width="800" height="145"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Design Goals and Guiding Principles
&lt;/h2&gt;

&lt;p&gt;To ensure the migration could be completed safely and efficiently, Tongcheng Travel established three core principles.&lt;/p&gt;

&lt;h3&gt;
  
  
  Zero Business Impact
&lt;/h3&gt;

&lt;p&gt;The migration had to be completely transparent to application teams.&lt;/p&gt;

&lt;p&gt;Historical jobs should continue running without requiring developers to rewrite existing scripts or learn a new execution model.&lt;/p&gt;

&lt;p&gt;Before the final cutover, the platform supports a &lt;strong&gt;dual-run validation strategy&lt;/strong&gt;, allowing both the legacy engine and Apache SeaTunnel to execute the same job simultaneously. Once validation is complete, workloads can be switched seamlessly without any business-side changes.&lt;/p&gt;

&lt;h3&gt;
  
  
  Guaranteed Data Consistency
&lt;/h3&gt;

&lt;p&gt;Data quality is the foundation of every data integration platform.&lt;/p&gt;

&lt;p&gt;A comprehensive validation and fallback mechanism must ensure that migrated jobs produce results identical to those generated by the legacy engines, guaranteeing consistency before, during, and after migration.&lt;/p&gt;

&lt;h3&gt;
  
  
  Better Performance and Stability
&lt;/h3&gt;

&lt;p&gt;Legacy MapReduce-based pipelines were heavyweight, while Flink 1.6 introduced significant load on production databases and lacked many modern optimizations.&lt;/p&gt;

&lt;p&gt;By standardizing on &lt;strong&gt;Apache SeaTunnel Zeta Engine&lt;/strong&gt;, Tongcheng Travel aimed to leverage its lightweight execution model and high-performance architecture to improve synchronization efficiency while reducing infrastructure costs.&lt;/p&gt;

&lt;h1&gt;
  
  
  Automatic Task Migration
&lt;/h1&gt;

&lt;p&gt;After defining the overall architecture, the team's biggest challenge became clear:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;How can tens of thousands of production jobs be migrated from Sqoop and FlinkSQL to Apache SeaTunnel without requiring developers to modify a single line of code?&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Migrating Sqoop and FlinkSQL to Apache SeaTunnel
&lt;/h2&gt;

&lt;p&gt;For most organizations, upgrading the underlying data integration engine usually requires significant cooperation from application teams.&lt;/p&gt;

&lt;p&gt;Tongcheng Travel took a completely different approach.&lt;/p&gt;

&lt;p&gt;Their guiding principle was:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Zero business awareness. Zero code changes.&lt;/strong&gt;&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;Existing Sqoop scripts and FlinkSQL jobs continue to be submitted exactly as before.&lt;/p&gt;

&lt;p&gt;Behind the scenes, the platform transparently converts them into standard Apache SeaTunnel jobs and executes them on the SeaTunnel engine.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ffn7271o63w4g2s35cx8j.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Ffn7271o63w4g2s35cx8j.jpg" width="800" height="159"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Skill Layer
&lt;/h3&gt;

&lt;p&gt;The first layer of the architecture consists of a collection of intelligent "Skill" modules responsible for understanding different job formats.&lt;/p&gt;

&lt;p&gt;Legacy production jobs—including tens of thousands of Sqoop scripts and FlinkSQL statements—remain completely unchanged.&lt;/p&gt;

&lt;p&gt;Instead, specialized translation components interpret each job type:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Sqoop Skill&lt;/li&gt;
&lt;li&gt;FlinkSQL Skill&lt;/li&gt;
&lt;li&gt;SeaTunnel Skill&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These Skills recognize syntax, configuration parameters, execution semantics, and connector definitions before mapping them into a standard Apache SeaTunnel configuration.&lt;/p&gt;

&lt;p&gt;This abstraction layer isolates application developers from changes in the underlying execution engine.&lt;/p&gt;

&lt;h3&gt;
  
  
  Execution Layer
&lt;/h3&gt;

&lt;p&gt;After translation is complete, the unified submission service generates a standard SeaTunnel job configuration.&lt;/p&gt;

&lt;p&gt;To guarantee migration safety, Tongcheng Travel introduced a &lt;strong&gt;dual-run execution mechanism&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Whenever a workflow is triggered, the platform actually launches &lt;strong&gt;two independent jobs&lt;/strong&gt;.&lt;/p&gt;

&lt;h4&gt;
  
  
  Production Baseline Job
&lt;/h4&gt;

&lt;p&gt;The original Sqoop or FlinkSQL task continues running exactly as before.&lt;/p&gt;

&lt;p&gt;This guarantees that existing production data delivery remains completely unaffected throughout the migration process.&lt;/p&gt;

&lt;h4&gt;
  
  
  Apache SeaTunnel Canary Job
&lt;/h4&gt;

&lt;p&gt;At the same time, the translated Apache SeaTunnel job is submitted to a separate execution cluster.&lt;/p&gt;

&lt;p&gt;Instead of writing into production storage, the output is written into a validation environment for comparison.&lt;/p&gt;

&lt;h4&gt;
  
  
  Intelligent Validation and Seamless Cutover
&lt;/h4&gt;

&lt;p&gt;Once both jobs complete, an automated validation framework compares their outputs.&lt;/p&gt;

&lt;p&gt;If every validation rule passes successfully, future executions of that workload are automatically routed to Apache SeaTunnel.&lt;/p&gt;

&lt;p&gt;From the application's perspective, nothing changes.&lt;/p&gt;

&lt;p&gt;The migration happens entirely behind the scenes.&lt;/p&gt;

&lt;h2&gt;
  
  
  From Natural Language to Apache SeaTunnel
&lt;/h2&gt;

&lt;p&gt;Migrating historical workloads solves yesterday's problems.&lt;/p&gt;

&lt;p&gt;The next challenge is enabling future development to become dramatically easier.&lt;/p&gt;

&lt;p&gt;Tongcheng Travel wanted business users—not just experienced data engineers—to create data synchronization tasks without needing to understand connectors, SQL dialects, or execution engines.&lt;/p&gt;

&lt;p&gt;Their answer is &lt;strong&gt;Data Copilot&lt;/strong&gt;, an AI-powered pipeline generation framework built on large language models.&lt;/p&gt;

&lt;h3&gt;
  
  
  Intelligent Information Completion
&lt;/h3&gt;

&lt;p&gt;Real-world user requests are rarely complete.&lt;/p&gt;

&lt;p&gt;A request such as:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;"Synchronize yesterday's core product sales data into the data warehouse."&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;contains insufficient information for execution.&lt;/p&gt;

&lt;p&gt;It doesn't specify:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;the source database&lt;/li&gt;
&lt;li&gt;the destination table&lt;/li&gt;
&lt;li&gt;partition strategy&lt;/li&gt;
&lt;li&gt;synchronization mode&lt;/li&gt;
&lt;li&gt;connector configuration&lt;/li&gt;
&lt;li&gt;execution parallelism&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;To bridge this gap, Tongcheng Travel implemented an intelligent information completion framework.&lt;/p&gt;

&lt;p&gt;The system first extracts the user's intent.&lt;/p&gt;

&lt;p&gt;It then combines:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;historical user behavior,&lt;/li&gt;
&lt;li&gt;enterprise metadata,&lt;/li&gt;
&lt;li&gt;business semantics,&lt;/li&gt;
&lt;li&gt;existing synchronization patterns,&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;to automatically infer the missing information.&lt;/p&gt;

&lt;p&gt;Finally, the system selects the appropriate &lt;strong&gt;Source Connector&lt;/strong&gt; (for example, the MySQL Core Product table) and &lt;strong&gt;Sink Connector&lt;/strong&gt; (such as Hive), while the SeaTunnel Skill automatically generates low-level execution parameters, including partitioning strategy, parallelism, and connector configuration.&lt;/p&gt;

&lt;p&gt;The result is a fully executable Apache SeaTunnel job generated entirely from natural language.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F8qchtrf0wksdt8r2q5hw.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F8qchtrf0wksdt8r2q5hw.jpg" width="798" height="333"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Text2SQL: From Natural Language to SQL
&lt;/h3&gt;

&lt;p&gt;Building complete data pipelines from natural language requires more than simply understanding user intent. The platform must also generate accurate, executable SQL statements that can be seamlessly embedded into Apache SeaTunnel jobs.&lt;/p&gt;

&lt;p&gt;To achieve this, Tongcheng Travel designed a multi-stage Text2SQL architecture consisting of &lt;strong&gt;Schema Linking&lt;/strong&gt;, &lt;strong&gt;Candidate SQL Generation&lt;/strong&gt;, and &lt;strong&gt;Candidate Selection&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fuas9t1yvfwi9hoi0ithc.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fuas9t1yvfwi9hoi0ithc.jpg" width="799" height="358"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h4&gt;
  
  
  Schema Linking
&lt;/h4&gt;

&lt;p&gt;Schema Linking identifies references to database schemas, tables, columns, and filtering conditions within a user's natural language request.&lt;/p&gt;

&lt;p&gt;This step significantly improves both cross-domain generalization and the accuracy of complex SQL generation, making it a foundational component of nearly every modern Text2SQL system.&lt;/p&gt;

&lt;p&gt;For example, when a user asks, "Synchronize yesterday's hotel orders," the system automatically identifies the corresponding business tables, date fields, and filtering conditions based on enterprise metadata and schema information.&lt;/p&gt;

&lt;h4&gt;
  
  
  Candidate SQL Generation
&lt;/h4&gt;

&lt;p&gt;Rather than relying on the output of a single large language model, Tongcheng Travel adopts a &lt;strong&gt;multi-generator strategy&lt;/strong&gt; to maximize SQL generation quality and reliability.&lt;/p&gt;

&lt;p&gt;Three independent generators work in parallel.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Reasoning Generator&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The Reasoning Generator leverages the zero-shot reasoning capabilities of large language models to generate SQL directly from the user's intent and the underlying database schema.&lt;/p&gt;

&lt;p&gt;This approach performs well when encountering entirely new business scenarios without requiring historical examples.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;In-Context Learning (ICL) Generator&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The ICL Generator adopts a Few-shot Learning strategy.&lt;/p&gt;

&lt;p&gt;Instead of generating SQL from scratch, it retrieves previously executed high-quality SQL statements that closely resemble the current request and provides them as examples to the language model.&lt;/p&gt;

&lt;p&gt;This significantly improves generation accuracy for recurring business scenarios.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Divide-and-Conquer Generator&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Some enterprise SQL statements span hundreds or even thousands of lines and include numerous nested subqueries.&lt;/p&gt;

&lt;p&gt;Instead of generating one massive SQL statement, the Divide-and-Conquer Generator decomposes the query into multiple Common Table Expressions (CTEs), generates each subquery independently, and finally assembles them into a complete SQL statement.&lt;/p&gt;

&lt;p&gt;This approach dramatically improves generation quality for highly complex analytical queries.&lt;/p&gt;

&lt;h4&gt;
  
  
  Candidate SQL Selection
&lt;/h4&gt;

&lt;p&gt;Once multiple candidate SQL statements have been generated, they undergo an intelligent evaluation process.&lt;/p&gt;

&lt;p&gt;Each candidate is scored based on several dimensions, including:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;SQL syntax correctness&lt;/li&gt;
&lt;li&gt;Abstract Syntax Tree (AST) complexity&lt;/li&gt;
&lt;li&gt;Estimated execution cost&lt;/li&gt;
&lt;li&gt;Semantic consistency with the original request&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Finally, a dedicated &lt;strong&gt;SQL Selector&lt;/strong&gt; chooses the optimal SQL statement and embeds it directly into the Apache SeaTunnel Transform stage.&lt;/p&gt;

&lt;p&gt;This multi-path generation strategy substantially improves both SQL accuracy and execution stability compared to traditional single-model approaches.&lt;/p&gt;

&lt;h3&gt;
  
  
  Text2ETL: Generating Data Pipelines from Natural Language
&lt;/h3&gt;

&lt;p&gt;Beyond SQL generation, Tongcheng Travel also enables complete ETL workflows to be created through natural language.&lt;/p&gt;

&lt;h4&gt;
  
  
  Interactive Preview
&lt;/h4&gt;

&lt;p&gt;After processing a user's request, the platform immediately generates a preview of the transformed dataset.&lt;/p&gt;

&lt;p&gt;Users can verify whether the generated pipeline matches their expectations before execution, significantly reducing trial-and-error during development.&lt;/p&gt;

&lt;h4&gt;
  
  
  Intelligent Transform Routing
&lt;/h4&gt;

&lt;p&gt;For standard data transformation requirements, the system automatically maps user intent to Apache SeaTunnel's native Transform plugins, including operations such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Filter&lt;/li&gt;
&lt;li&gt;Replace&lt;/li&gt;
&lt;li&gt;Split&lt;/li&gt;
&lt;li&gt;Additional built-in Transform components&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Whenever possible, the platform prioritizes native SeaTunnel capabilities to maximize execution performance and maintainability.&lt;/p&gt;

&lt;h4&gt;
  
  
  Dynamic Compilation as a Fallback
&lt;/h4&gt;

&lt;p&gt;Some enterprise data cleansing scenarios are too complex to be expressed using combinations of built-in Transform operators.&lt;/p&gt;

&lt;p&gt;Examples include sophisticated data quality rules, custom parsing logic, or highly specialized business transformations.&lt;/p&gt;

&lt;p&gt;In these situations, the platform automatically generates Java, Scala, or Groovy code implementing a User Defined Function (UDF).&lt;/p&gt;

&lt;p&gt;The generated code is dynamically compiled in the background and loaded into the Apache SeaTunnel runtime as a plugin without requiring manual packaging or deployment.&lt;/p&gt;

&lt;p&gt;This mechanism allows developers to benefit from AI-generated pipeline creation while preserving the flexibility needed for complex enterprise workloads.&lt;/p&gt;

&lt;h1&gt;
  
  
  Data Consistency Validation During Migration
&lt;/h1&gt;

&lt;p&gt;Automatically converting jobs is only the first step in a successful migration.&lt;/p&gt;

&lt;p&gt;The true challenge lies in ensuring that both the legacy engine and Apache SeaTunnel produce exactly the same results.&lt;/p&gt;

&lt;p&gt;Because different execution engines may implement serialization formats, timestamp handling, numeric precision, and connector behavior differently, guaranteeing &lt;strong&gt;zero data loss and zero data deviation&lt;/strong&gt; becomes the most critical requirement of the entire migration project.&lt;/p&gt;

&lt;p&gt;Tongcheng Travel therefore designed a comprehensive validation framework capable of verifying large-scale production datasets efficiently.&lt;/p&gt;

&lt;h2&gt;
  
  
  Validation for File-Based Engines
&lt;/h2&gt;

&lt;p&gt;Unlike relational databases, file systems do not provide built-in query capabilities.&lt;/p&gt;

&lt;p&gt;Loading terabytes of files into memory for full comparison would be both time-consuming and extremely memory-intensive, potentially leading to OutOfMemory (OOM) errors.&lt;/p&gt;

&lt;p&gt;To address this challenge, Tongcheng Travel designed a highly efficient comparison framework based on &lt;strong&gt;side indexes&lt;/strong&gt; and &lt;strong&gt;multi-level filtering&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F4mmmtpdw7zy32pgadbt4.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F4mmmtpdw7zy32pgadbt4.jpg" width="799" height="370"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Step 1. Data Reading and Normalization
&lt;/h3&gt;

&lt;p&gt;The system concurrently scans both output directories:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Legacy XData/Sqoop output (baseline)&lt;/li&gt;
&lt;li&gt;Apache SeaTunnel Zeta output (candidate)&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Each record is read line by line and first normalized to eliminate formatting differences.&lt;/p&gt;

&lt;p&gt;After normalization, the entire row is hashed using a deterministic hash algorithm.&lt;/p&gt;

&lt;h3&gt;
  
  
  Step 2. Building Side Indexes
&lt;/h3&gt;

&lt;p&gt;To quickly locate discrepancies without rescanning original files, the platform builds an ordered external index while computing hash values.&lt;/p&gt;

&lt;p&gt;The side index consists of two layers.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;File-Level Index&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;The header records the mapping between a logical file index and its physical storage location.&lt;/p&gt;

&lt;p&gt;For example:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;fileIndex = 0 → .../part-00000
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;&lt;strong&gt;Record-Level Index&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Each normalized record generates an index entry containing:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;[hash, fileIndex, lineNumber]
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;These index entries are globally sorted by hash value using external sorting algorithms.&lt;/p&gt;

&lt;p&gt;This preprocessing lays the foundation for highly efficient large-scale comparison.&lt;/p&gt;

&lt;h3&gt;
  
  
  Step 3. Multi-Level Comparison Algorithm
&lt;/h3&gt;

&lt;p&gt;Instead of comparing every record directly, the validation framework performs comparison in multiple stages.&lt;/p&gt;

&lt;h4&gt;
  
  
  Level 1: Bloom Filter Screening
&lt;/h4&gt;

&lt;p&gt;Each dataset first generates an independent Bloom Filter.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;BloomFilter(XData)

BloomFilter(Zeta)
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Because Bloom Filters require very little memory while providing extremely fast lookup performance, they quickly eliminate the vast majority of matching records.&lt;/p&gt;

&lt;p&gt;Only potentially inconsistent data proceeds to the next stage.&lt;/p&gt;

&lt;h4&gt;
  
  
  Level 2: Two-Pointer Comparison
&lt;/h4&gt;

&lt;p&gt;For records that cannot be confirmed through Bloom Filter screening, the system compares the two sorted hash indexes using a classic two-pointer algorithm.&lt;/p&gt;

&lt;p&gt;If:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;h1 &amp;lt; h2
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;the baseline dataset contains records missing from Apache SeaTunnel.&lt;/p&gt;

&lt;p&gt;The pointer for the baseline dataset advances.&lt;/p&gt;

&lt;p&gt;If:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;h1 &amp;gt; h2
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Apache SeaTunnel contains additional records.&lt;/p&gt;

&lt;p&gt;The pointer for the candidate dataset advances.&lt;/p&gt;

&lt;p&gt;If:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;h1 == h2
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;the framework performs an additional comparison of the occurrence count for that hash value.&lt;/p&gt;

&lt;p&gt;This effectively handles duplicate records while maintaining comparison accuracy.&lt;/p&gt;

&lt;h3&gt;
  
  
  Step 4. Difference Localization
&lt;/h3&gt;

&lt;p&gt;Whenever inconsistencies are detected, the framework performs a binary search on the side index to rapidly locate the offending hash value.&lt;/p&gt;

&lt;p&gt;Using the corresponding:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;fileIndex&lt;/li&gt;
&lt;li&gt;lineNumber&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;developers can immediately identify the exact row in the original source file where corruption, truncation, or encoding issues occurred.&lt;/p&gt;

&lt;p&gt;This reduces troubleshooting time from hours to seconds.&lt;/p&gt;

&lt;h3&gt;
  
  
  Step 5. Automatic Cutover
&lt;/h3&gt;

&lt;p&gt;Only after every validation rule passes successfully across the complete dataset does the platform perform the final production cutover, routing future executions entirely to Apache SeaTunnel.&lt;/p&gt;

&lt;h2&gt;
  
  
  Validation for Database Engines
&lt;/h2&gt;

&lt;p&gt;For relational databases and data warehouses, Tongcheng Travel takes advantage of the database engine itself to calculate data fingerprints, eliminating the need to export large datasets for comparison.&lt;/p&gt;

&lt;p&gt;Instead of comparing records one by one, the platform computes hash-based feature values for every column and compares the aggregated results between the legacy engine and Apache SeaTunnel.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;SELECT&lt;/span&gt;
  &lt;span class="k"&gt;sum&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;murmur_hash3_32&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;coalesce&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="k"&gt;CAST&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;order_serial_no&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;string&lt;/span&gt;&lt;span class="p"&gt;),&lt;/span&gt; &lt;span class="s1"&gt;'NULL'&lt;/span&gt;&lt;span class="p"&gt;)))&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;order_serial_no_hash&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="k"&gt;sum&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;murmur_hash3_32&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;coalesce&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="k"&gt;CAST&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;platform_code&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;string&lt;/span&gt;&lt;span class="p"&gt;),&lt;/span&gt; &lt;span class="s1"&gt;'NULL'&lt;/span&gt;&lt;span class="p"&gt;)))&lt;/span&gt; &lt;span class="k"&gt;AS&lt;/span&gt; &lt;span class="n"&gt;platform_code_hash&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="p"&gt;...&lt;/span&gt;
&lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="p"&gt;...&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;By comparing the hash signatures of each column, the platform can quickly determine whether two datasets are identical with minimal computational overhead.&lt;/p&gt;

&lt;p&gt;Compared with traditional full-table comparison, this approach dramatically improves validation efficiency while maintaining high accuracy, making it well suited for large-scale production migrations.&lt;/p&gt;

&lt;h1&gt;
  
  
  Improving Performance and Stability
&lt;/h1&gt;

&lt;p&gt;Data consistency guarantees migration correctness, but long-term platform success also depends on execution efficiency and operational stability.&lt;/p&gt;

&lt;p&gt;To maximize the performance of the unified data integration platform, Tongcheng Travel optimized Apache SeaTunnel across three key areas:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Connector enhancements&lt;/li&gt;
&lt;li&gt;Intelligent parallelism estimation&lt;/li&gt;
&lt;li&gt;Improved observability&lt;/li&gt;
&lt;/ul&gt;

&lt;h2&gt;
  
  
  Connector Enhancements
&lt;/h2&gt;

&lt;p&gt;Since connectors serve as the bridge between Apache SeaTunnel and enterprise data systems, improving connector reliability directly improves overall platform stability.&lt;/p&gt;

&lt;p&gt;The team contributed a series of enhancements covering multiple ecosystems.&lt;/p&gt;

&lt;h3&gt;
  
  
  Enhanced Paimon Connector
&lt;/h3&gt;

&lt;p&gt;The Paimon connector received significant improvements, including:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Full support for the &lt;strong&gt;TIME&lt;/strong&gt; data type&lt;/li&gt;
&lt;li&gt;Predicate pushdown for &lt;strong&gt;LIKE&lt;/strong&gt; and &lt;strong&gt;BETWEEN&lt;/strong&gt; operations, reducing unnecessary data scanning&lt;/li&gt;
&lt;li&gt;Dynamic table option discovery&lt;/li&gt;
&lt;li&gt;Branch writing support for Sink operations&lt;/li&gt;
&lt;li&gt;Fixes for DECIMAL precision loss&lt;/li&gt;
&lt;li&gt;Resolution of missing fields during DataType conversion&lt;/li&gt;
&lt;li&gt;Parallel reading from multiple Paimon Source tables&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These enhancements significantly improve both compatibility and execution efficiency for Paimon workloads.&lt;/p&gt;

&lt;h3&gt;
  
  
  High Availability for OLAP Connectors
&lt;/h3&gt;

&lt;p&gt;The StarRocks and Doris connectors were enhanced to improve availability and reliability.&lt;/p&gt;

&lt;p&gt;For StarRocks:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Added Frontend (FE) High Availability support&lt;/li&gt;
&lt;li&gt;Randomized FE endpoint selection to eliminate single points of failure&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;For Doris:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Fixed data loss issues occurring when
&lt;/li&gt;
&lt;/ul&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;request_table_size &amp;lt; BUCKETS
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;These improvements greatly increase connector resilience in production environments.&lt;/p&gt;

&lt;h3&gt;
  
  
  Improved Streaming Stability
&lt;/h3&gt;

&lt;p&gt;Several optimizations were introduced for streaming connectors.&lt;/p&gt;

&lt;p&gt;For Kafka:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Added partition filtering in Stream Mode to prevent consumer blocking&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;For RocketMQ:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Added support for skipping malformed records instead of terminating the entire synchronization job&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;For Milvus:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Fixed missing partition-level Load State validation&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These changes improve fault tolerance without sacrificing throughput.&lt;/p&gt;

&lt;h3&gt;
  
  
  HBase Improvements
&lt;/h3&gt;

&lt;p&gt;The HBase Source connector now supports row-range boundary queries, allowing more efficient incremental extraction and partitioned reads.&lt;/p&gt;

&lt;h3&gt;
  
  
  HDFS ViewFs Compatibility
&lt;/h3&gt;

&lt;p&gt;Apache SeaTunnel now supports the ViewFs filesystem schema, enabling seamless access to HDFS deployments spanning multiple namespace mount points.&lt;/p&gt;

&lt;p&gt;This enhancement improves compatibility with large enterprise Hadoop clusters.&lt;/p&gt;

&lt;h2&gt;
  
  
  Intelligent Parallelism Estimation
&lt;/h2&gt;

&lt;p&gt;Inspired by Apache Flink's Adaptive Batch Scheduler, Tongcheng Travel implemented an &lt;strong&gt;Intelligent Parallelism Estimation&lt;/strong&gt; mechanism for Apache SeaTunnel Zeta Engine.&lt;/p&gt;

&lt;p&gt;Before a job is submitted, the engine performs lightweight metadata analysis to estimate the actual workload.&lt;/p&gt;

&lt;p&gt;Depending on the source type, it collects information such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Total table row count&lt;/li&gt;
&lt;li&gt;Directory size&lt;/li&gt;
&lt;li&gt;Kafka partition count&lt;/li&gt;
&lt;li&gt;Paimon bucket distribution&lt;/li&gt;
&lt;li&gt;Physical storage topology&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These physical characteristics are then combined with runtime information, including:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Available cluster resources&lt;/li&gt;
&lt;li&gt;Destination throughput limits&lt;/li&gt;
&lt;li&gt;Current cluster workload&lt;/li&gt;
&lt;li&gt;Resource utilization&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Using these inputs, the scheduler dynamically calculates the optimal degree of parallelism and determines the most efficient task partitioning strategy.&lt;/p&gt;

&lt;p&gt;Rather than relying on static configuration, computing resources are allocated on demand, allowing Apache SeaTunnel to align task execution with the underlying storage topology.&lt;/p&gt;

&lt;p&gt;The result is higher resource utilization, shorter execution times, and improved overall cluster efficiency.&lt;/p&gt;

&lt;h2&gt;
  
  
  Improved Observability
&lt;/h2&gt;

&lt;p&gt;As the unified platform continued to grow, comprehensive observability became increasingly important.&lt;/p&gt;

&lt;p&gt;Tongcheng Travel therefore introduced several monitoring capabilities to improve operational visibility.&lt;/p&gt;

&lt;h3&gt;
  
  
  Checkpoint Monitoring
&lt;/h3&gt;

&lt;p&gt;Inspired by Apache Flink's mature checkpoint monitoring system, the team built a comprehensive observability framework for Apache SeaTunnel Checkpoints.&lt;/p&gt;

&lt;p&gt;Key metrics include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;End-to-end Checkpoint duration&lt;/li&gt;
&lt;li&gt;State size&lt;/li&gt;
&lt;li&gt;Checkpoint failure frequency&lt;/li&gt;
&lt;li&gt;Checkpoint success rate&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These metrics help engineers quickly identify bottlenecks affecting fault tolerance and recovery performance.&lt;/p&gt;

&lt;h3&gt;
  
  
  Master Election Metrics
&lt;/h3&gt;

&lt;p&gt;Borrowing ideas from Apache Kafka's controller election monitoring, Tongcheng Travel also implemented detailed monitoring for Apache SeaTunnel Master elections.&lt;/p&gt;

&lt;p&gt;The platform continuously tracks:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Election duration&lt;/li&gt;
&lt;li&gt;Master switch frequency&lt;/li&gt;
&lt;li&gt;Active Master availability&lt;/li&gt;
&lt;li&gt;Election success rate&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;When abnormal conditions occur—such as split-brain scenarios or frequent Master failovers—the monitoring system immediately generates alerts, helping engineers diagnose underlying network issues or cluster resource bottlenecks before they impact production workloads.&lt;/p&gt;

&lt;h1&gt;
  
  
  Future Roadmap
&lt;/h1&gt;

&lt;p&gt;With the unified data integration platform now running reliably in production, Tongcheng Travel is focusing on the next stage of its evolution.&lt;/p&gt;

&lt;p&gt;Over the next one to two years, development will center around two strategic directions:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;strong&gt;Cloud-Native Architecture&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;AI-Powered Data Integration&lt;/strong&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;h2&gt;
  
  
  Cloud-Native Architecture
&lt;/h2&gt;

&lt;p&gt;The team plans to fully embrace Kubernetes-native resource scheduling.&lt;/p&gt;

&lt;p&gt;Future enhancements include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;On-demand Worker provisioning&lt;/li&gt;
&lt;li&gt;Automatic Worker lifecycle management&lt;/li&gt;
&lt;li&gt;Elastic scaling based on workload&lt;/li&gt;
&lt;li&gt;Dynamic resource allocation&lt;/li&gt;
&lt;li&gt;Remote Job Log storage following cloud-native logging best practices&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These capabilities will further improve infrastructure efficiency while reducing operational overhead.&lt;/p&gt;

&lt;h2&gt;
  
  
  AI-Powered Data Integration
&lt;/h2&gt;

&lt;p&gt;Artificial intelligence will become a core component of the next-generation data platform.&lt;/p&gt;

&lt;p&gt;The long-term vision is to build a closed-loop intelligent operations system capable of:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Generating complete synchronization jobs from natural language&lt;/li&gt;
&lt;li&gt;Automatically identifying root causes of failed jobs&lt;/li&gt;
&lt;li&gt;Optimizing parallelism based on runtime performance and data latency&lt;/li&gt;
&lt;li&gt;Continuously tuning execution strategies without manual intervention&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The goal is to transform data integration from a manually operated platform into a self-healing, self-optimizing system powered by AI.&lt;/p&gt;

&lt;h1&gt;
  
  
  Final Thoughts
&lt;/h1&gt;

&lt;p&gt;A unified data platform is more than an engine upgrade—it's a foundation for future innovation. With Apache SeaTunnel at its core, Tongcheng Travel is building a cloud-native, AI-powered data integration platform ready for the next generation of enterprise data engineering.&lt;/p&gt;

</description>
      <category>apacheseatunnel</category>
      <category>datascience</category>
      <category>database</category>
      <category>opensource</category>
    </item>
    <item>
      <title>🚀 Apache SeaTunnel merged 125 PRs in June! From Control Plane evolution and StateStore abstraction to BigQuery, MQTT &amp; Calcite SQL, the platform keeps getting smarter and stronger. 💪</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 10 Jul 2026 02:57:20 +0000</pubDate>
      <link>https://dev.to/seatunnel/apache-seatunnel-merged-125-prs-in-june-from-control-plane-evolution-and-statestore-abstraction-46j3</link>
      <guid>https://dev.to/seatunnel/apache-seatunnel-merged-125-prs-in-june-from-control-plane-evolution-and-statestore-abstraction-46j3</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/seatunnel/apache-seatunnel-june-2026-roundup-control-plane-evolution-smarter-governance-and-a-growing-5a3n" class="crayons-story__hidden-navigation-link"&gt;Apache SeaTunnel June 2026 Roundup: Control Plane Evolution, Smarter Governance, and a Growing Connector Ecosystem&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/seatunnel" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" alt="seatunnel profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/seatunnel" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Apache SeaTunnel
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Apache SeaTunnel
                
              
              &lt;div id="story-author-preview-content-4109768" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/seatunnel" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Apache SeaTunnel&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/seatunnel/apache-seatunnel-june-2026-roundup-control-plane-evolution-smarter-governance-and-a-growing-5a3n" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 10&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/seatunnel/apache-seatunnel-june-2026-roundup-control-plane-evolution-smarter-governance-and-a-growing-5a3n" id="article-link-4109768"&gt;
          Apache SeaTunnel June 2026 Roundup: Control Plane Evolution, Smarter Governance, and a Growing Connector Ecosystem
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apacheseatunnel"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apacheseatunnel&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/opensource"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;opensource&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/data"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;data&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/seatunnel/apache-seatunnel-june-2026-roundup-control-plane-evolution-smarter-governance-and-a-growing-5a3n#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            13 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>Apache SeaTunnel June 2026 Roundup: Control Plane Evolution, Smarter Governance, and a Growing Connector Ecosystem</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 10 Jul 2026 02:56:52 +0000</pubDate>
      <link>https://dev.to/seatunnel/apache-seatunnel-june-2026-roundup-control-plane-evolution-smarter-governance-and-a-growing-5a3n</link>
      <guid>https://dev.to/seatunnel/apache-seatunnel-june-2026-roundup-control-plane-evolution-smarter-governance-and-a-growing-5a3n</guid>
      <description>&lt;p&gt;Hi, SeaTunnel Community!&lt;/p&gt;

&lt;p&gt;The Apache SeaTunnel June 2026 monthly roundup is here. Throughout June, the project merged &lt;strong&gt;125 pull requests&lt;/strong&gt;, a significant increase from &lt;strong&gt;87 PRs in May&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;More importantly, June wasn't simply about fixing bugs or adding new connectors. Development across the project clearly converged around three major strategic directions:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;The control plane and engine abstractions continued to evolve.&lt;/strong&gt; Features such as table-level fault isolation for multi-table synchronization, the introduction of the StateStore abstraction, task workload monitoring, and reporting of non-terminal job states indicate that Zeta is evolving beyond simply executing jobs toward becoming a platform that is easier to govern, recover, and observe in production.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;The Connector-V2 ecosystem continued to expand.&lt;/strong&gt; New connectors and capabilities—including BigQuery Sink, MQTT Source, Salesforce Source, Vitess CDC, and Google Cloud Bigtable integration—further broadened SeaTunnel's coverage across modern data infrastructure.&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;&lt;strong&gt;The data processing layer became more powerful.&lt;/strong&gt; Features such as Schema Evolution for file connectors, the Calcite SQL Transform Plugin, and built-in Base64 SQL functions demonstrate that SeaTunnel is no longer focused solely on moving data. It is steadily evolving into a platform capable of performing increasingly sophisticated data transformation and processing.&lt;/p&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;h1&gt;
  
  
  1. Project Overview
&lt;/h1&gt;

&lt;h2&gt;
  
  
  1.1 Development Statistics
&lt;/h2&gt;

&lt;p&gt;Development activity in June can be categorized into four major areas:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;New Features:&lt;/strong&gt; 18&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Performance Improvements:&lt;/strong&gt; 0&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Bug Fixes:&lt;/strong&gt; 40&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Architecture Improvements:&lt;/strong&gt; 67&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzces49fs64bbjzjmykud.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzces49fs64bbjzjmykud.jpg" width="799" height="370"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Compared with May, the most notable change was the substantial increase in &lt;strong&gt;architecture-related pull requests&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;This signals an important shift in the project's priorities. Rather than focusing solely on shipping new connectors, the community invested heavily in strengthening SeaTunnel's underlying architecture, engineering quality, documentation, stability, and system abstractions. These foundational improvements lay the groundwork for long-term scalability and maintainability.&lt;/p&gt;

&lt;h2&gt;
  
  
  1.2 Module Distribution
&lt;/h2&gt;

&lt;p&gt;PR distribution across major modules:&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Module&lt;/th&gt;
&lt;th&gt;PRs&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;seatunnel-connectors-v2&lt;/td&gt;
&lt;td&gt;35&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;docs&lt;/td&gt;
&lt;td&gt;27&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;seatunnel-engine&lt;/td&gt;
&lt;td&gt;22&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;seatunnel-connectors-v2/connector-cdc&lt;/td&gt;
&lt;td&gt;14&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;seatunnel-e2e&lt;/td&gt;
&lt;td&gt;12&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;seatunnel-api&lt;/td&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;.github/ci&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;Other&lt;/td&gt;
&lt;td&gt;10&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;One particularly noteworthy trend is the significant increase in documentation-related contributions.&lt;/p&gt;

&lt;p&gt;Throughout June, the community invested considerable effort into transforming operational knowledge into well-structured documentation. Examples include documentation covering:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Zeta StateStore and recovery mechanisms&lt;/li&gt;
&lt;li&gt;CDC production deployment cookbooks&lt;/li&gt;
&lt;li&gt;REST API lifecycle documentation&lt;/li&gt;
&lt;li&gt;Operational best practices for production environments&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;While documentation doesn't directly improve throughput or latency, it dramatically enhances SeaTunnel's usability, maintainability, and adoption in enterprise environments. Making operational knowledge explicit lowers the learning curve for new users and enables teams to deploy SeaTunnel more confidently in production.&lt;/p&gt;

&lt;h1&gt;
  
  
  2. Top Contributors
&lt;/h1&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F9iior8y4d1ci6vowi0an.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F9iior8y4d1ci6vowi0an.jpg" width="799" height="450"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;div class="table-wrapper-paragraph"&gt;&lt;table&gt;
&lt;thead&gt;
&lt;tr&gt;
&lt;th&gt;Rank&lt;/th&gt;
&lt;th&gt;GitHub User&lt;/th&gt;
&lt;th&gt;Merged PRs&lt;/th&gt;
&lt;th&gt;Primary Contribution Categories&lt;/th&gt;
&lt;/tr&gt;
&lt;/thead&gt;
&lt;tbody&gt;
&lt;tr&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;DanielLeens&lt;/td&gt;
&lt;td&gt;27&lt;/td&gt;
&lt;td&gt;Features ×1 · Bug Fixes ×5 · Architecture ×21&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;zhangshenghang&lt;/td&gt;
&lt;td&gt;26&lt;/td&gt;
&lt;td&gt;Features ×1 · Bug Fixes ×11 · Architecture ×14&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;3&lt;/td&gt;
&lt;td&gt;nzw921rx&lt;/td&gt;
&lt;td&gt;19&lt;/td&gt;
&lt;td&gt;Features ×2 · Bug Fixes ×6 · Architecture ×11&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;davidzollo&lt;/td&gt;
&lt;td&gt;7&lt;/td&gt;
&lt;td&gt;Bug Fixes ×2 · Architecture ×5&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;5&lt;/td&gt;
&lt;td&gt;dybyte&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;Features ×3 · Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;6&lt;/td&gt;
&lt;td&gt;yzeng1618&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;Features ×1 · Bug Fixes ×1 · Architecture ×2&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;7&lt;/td&gt;
&lt;td&gt;QuakeWang&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;Bug Fixes ×4&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;8&lt;/td&gt;
&lt;td&gt;JeremyXin&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;Bug Fixes ×2 · Architecture ×2&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;9&lt;/td&gt;
&lt;td&gt;DanielCarter-stack&lt;/td&gt;
&lt;td&gt;4&lt;/td&gt;
&lt;td&gt;Bug Fixes ×1 · Architecture ×3&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;10&lt;/td&gt;
&lt;td&gt;yuluo-yx&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;Bug Fixes ×2&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;11&lt;/td&gt;
&lt;td&gt;ricky2129&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;Features ×1 · Bug Fixes ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;12&lt;/td&gt;
&lt;td&gt;MyeoungDev&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;Features ×2&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;13&lt;/td&gt;
&lt;td&gt;ss666&lt;/td&gt;
&lt;td&gt;2&lt;/td&gt;
&lt;td&gt;Features ×1 · Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;14&lt;/td&gt;
&lt;td&gt;CosmosNi&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;15&lt;/td&gt;
&lt;td&gt;LeonYoah&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Bug Fix ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;16&lt;/td&gt;
&lt;td&gt;Marx-Carvalho&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;17&lt;/td&gt;
&lt;td&gt;hawk9821&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Bug Fix ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;18&lt;/td&gt;
&lt;td&gt;77amyfly&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Bug Fix ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;19&lt;/td&gt;
&lt;td&gt;JAEKWANG97&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Feature ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;20&lt;/td&gt;
&lt;td&gt;NixonWahome&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Feature ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;21&lt;/td&gt;
&lt;td&gt;GabrielBBaldez&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Feature ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;22&lt;/td&gt;
&lt;td&gt;niumy0701&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;23&lt;/td&gt;
&lt;td&gt;15037143579&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;24&lt;/td&gt;
&lt;td&gt;CloverDew&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Bug Fix ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;25&lt;/td&gt;
&lt;td&gt;xxzuo&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;26&lt;/td&gt;
&lt;td&gt;Muktha9491&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Feature ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;27&lt;/td&gt;
&lt;td&gt;programmerloverun&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Feature ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;28&lt;/td&gt;
&lt;td&gt;zooo-code&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Bug Fix ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;29&lt;/td&gt;
&lt;td&gt;zhiliang-wu&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Feature ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;30&lt;/td&gt;
&lt;td&gt;doyong365&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;tr&gt;
&lt;td&gt;31&lt;/td&gt;
&lt;td&gt;loupipalien&lt;/td&gt;
&lt;td&gt;1&lt;/td&gt;
&lt;td&gt;Architecture ×1&lt;/td&gt;
&lt;/tr&gt;
&lt;/tbody&gt;
&lt;/table&gt;&lt;/div&gt;

&lt;p&gt;June also highlighted the growing diversity of the Apache SeaTunnel community.&lt;/p&gt;

&lt;p&gt;Contributors from different organizations and regions collaborated across connector development, engine architecture, documentation, testing, and ecosystem improvements. Beyond the number of merged PRs, the sustained focus on architectural evolution demonstrates the community's commitment to building a more robust, scalable, and enterprise-ready data integration platform.&lt;/p&gt;

&lt;h1&gt;
  
  
  3. Key Technical Highlights
&lt;/h1&gt;

&lt;p&gt;Looking across all the pull requests merged in June, the major technical work can be grouped into four strategic areas.&lt;/p&gt;

&lt;h2&gt;
  
  
  3.1 Multi-Table Synchronization Enters a New Stage of Reliability
&lt;/h2&gt;

&lt;p&gt;One of the most significant improvements is the introduction of &lt;strong&gt;table-level fault isolation&lt;/strong&gt; for multi-table synchronization. (#10600)&lt;/p&gt;

&lt;p&gt;Previously, a failure in a single table could cause the entire synchronization job to fail. With the new fault isolation mechanism, SeaTunnel moves away from an all-or-nothing execution model toward more fine-grained failure handling.&lt;/p&gt;

&lt;p&gt;This is an important milestone because multi-table synchronization has become one of SeaTunnel's primary production use cases. As deployments continue to scale, the ability to isolate failures at the table level becomes essential for improving reliability and reducing operational costs.&lt;/p&gt;

&lt;h2&gt;
  
  
  3.2 File Connectors and SQL Processing Become More Powerful
&lt;/h2&gt;

&lt;p&gt;June also brought substantial improvements to SeaTunnel's data processing capabilities.&lt;/p&gt;

&lt;p&gt;The file connector now supports &lt;strong&gt;Schema Evolution&lt;/strong&gt;, enabling runtime handling of schema changes such as column additions, removals, renames, and updates. (#10744)&lt;/p&gt;

&lt;p&gt;At the same time, the introduction of the &lt;strong&gt;Calcite SQL Transform Plugin&lt;/strong&gt; provides a far more powerful SQL transformation framework with extensible UDF support. (#11062)&lt;/p&gt;

&lt;p&gt;Built-in SQL capabilities were also expanded through new &lt;strong&gt;Base64 functions&lt;/strong&gt;, demonstrating that SQL Transform is evolving beyond simple projection and filtering into a more comprehensive data processing layer. (#11114)&lt;/p&gt;

&lt;h2&gt;
  
  
  3.3 Zeta Continues Decoupling from Hazelcast IMap
&lt;/h2&gt;

&lt;p&gt;Another major architectural milestone is the introduction of the new &lt;strong&gt;StateStore abstraction&lt;/strong&gt;. (#10812)&lt;/p&gt;

&lt;p&gt;Historically, parts of the engine relied directly on Hazelcast's IMap implementation for state management. The new abstraction layer removes this tight coupling by introducing generic state storage interfaces.&lt;/p&gt;

&lt;p&gt;This architectural evolution opens the door for future support of alternative state backends, hybrid storage strategies, and more flexible runtime implementations.&lt;/p&gt;

&lt;h2&gt;
  
  
  3.4 Connector Expansion Continues with a Stronger Focus on Production Readiness
&lt;/h2&gt;

&lt;p&gt;The Connector-V2 ecosystem continued to grow throughout June.&lt;/p&gt;

&lt;p&gt;New connectors and integrations include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;BigQuery Sink&lt;/li&gt;
&lt;li&gt;MQTT Source&lt;/li&gt;
&lt;li&gt;Salesforce Source&lt;/li&gt;
&lt;li&gt;Vitess CDC&lt;/li&gt;
&lt;li&gt;Google Cloud Bigtable Source&lt;/li&gt;
&lt;li&gt;Google Cloud Bigtable Sink&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Alongside these additions, the community delivered numerous stability improvements across existing connectors and infrastructure, including fixes for:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Apache Paimon&lt;/li&gt;
&lt;li&gt;CDC connectors&lt;/li&gt;
&lt;li&gt;JdbcHive integration tests&lt;/li&gt;
&lt;li&gt;UI NaN display issues&lt;/li&gt;
&lt;li&gt;End-to-End testing stability&lt;/li&gt;
&lt;li&gt;JDK 8 compatibility&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Rather than simply increasing the number of supported systems, the project is placing greater emphasis on connector maturity, compatibility, and production readiness.&lt;/p&gt;

&lt;h1&gt;
  
  
  4. In-Depth Analysis of Major Technical Changes
&lt;/h1&gt;

&lt;p&gt;Among all the work completed in June, five pull requests best represent the project's technical direction. Below, we examine each of them from three perspectives:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;strong&gt;Background&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Implementation&lt;/strong&gt;&lt;/li&gt;
&lt;li&gt;&lt;strong&gt;Impact&lt;/strong&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;h1&gt;
  
  
  4.1 PR #10600 — Table-Level Fault Isolation: From Global Failure to Granular Recovery
&lt;/h1&gt;

&lt;p&gt;&lt;strong&gt;PR Size&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;23 files changed&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;1,648 insertions&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;144 deletions&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Background
&lt;/h2&gt;

&lt;p&gt;Multi-table synchronization has become one of Apache SeaTunnel's core production scenarios.&lt;/p&gt;

&lt;p&gt;Previously, if a single table encountered a write failure, the entire synchronization job would often fail.&lt;/p&gt;

&lt;p&gt;In large-scale production environments, this creates a classic "long-tail" problem:&lt;/p&gt;

&lt;p&gt;Imagine synchronizing 100 tables. If 99 complete successfully while just one fails, the entire job must be rerun, wasting considerable time and computing resources.&lt;/p&gt;

&lt;p&gt;As deployments continue to grow, this all-or-nothing failure model becomes increasingly expensive.&lt;/p&gt;

&lt;h2&gt;
  
  
  Core Design
&lt;/h2&gt;

&lt;p&gt;This pull request introduces new abstractions, including:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;code&gt;MultiTableCommonOptions&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;&lt;code&gt;MultiTableFailurePolicy&lt;/code&gt;&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These abstractions expose failure handling as an explicit configuration rather than hiding it inside runtime logic.&lt;/p&gt;

&lt;p&gt;Key implementation:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;class&lt;/span&gt; &lt;span class="nc"&gt;MultiTableCommonOptions&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;

    &lt;span class="nd"&gt;@Experimental&lt;/span&gt;
    &lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;static&lt;/span&gt; &lt;span class="kd"&gt;final&lt;/span&gt; &lt;span class="nc"&gt;Option&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;MultiTableFailurePolicy&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="no"&gt;MULTI_TABLE_FAILURE_POLICY&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt;
            &lt;span class="nc"&gt;Options&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;key&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"multi_table.failure_policy"&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
                    &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;enumType&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="nc"&gt;MultiTableFailurePolicy&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;class&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  How It Works
&lt;/h2&gt;

&lt;p&gt;Instead of treating every failure as a job-ending event, the runtime can now make decisions based on configurable policies.&lt;/p&gt;

&lt;p&gt;Possible behaviors include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Recording failed tables&lt;/li&gt;
&lt;li&gt;Continuing synchronization for healthy tables&lt;/li&gt;
&lt;li&gt;Preserving failure context for later recovery&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;This transforms failure handling from a single execution path into policy-driven runtime behavior.&lt;/p&gt;

&lt;h2&gt;
  
  
  Impact
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Connector Layer
&lt;/h3&gt;

&lt;p&gt;Multi-table sink implementations can now understand failure policies and receive table-level failure metadata.&lt;/p&gt;

&lt;h3&gt;
  
  
  Engine Layer
&lt;/h3&gt;

&lt;p&gt;The coordinator propagates failure context throughout the job lifecycle, enabling more intelligent recovery behavior.&lt;/p&gt;

&lt;h3&gt;
  
  
  Users
&lt;/h3&gt;

&lt;p&gt;For production workloads involving hundreds or thousands of similarly structured tables, this significantly reduces rerun costs and improves operational efficiency.&lt;/p&gt;

&lt;p&gt;This feature represents a major enhancement to SeaTunnel's production governance capabilities.&lt;/p&gt;

&lt;h1&gt;
  
  
  4.2 PR #10744 — Schema Evolution for File Connectors
&lt;/h1&gt;

&lt;p&gt;&lt;strong&gt;PR Size&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;4 files changed&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;168 insertions&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;2 deletions&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Background
&lt;/h2&gt;

&lt;p&gt;Traditional file-based sinks generally have poor tolerance for schema changes.&lt;/p&gt;

&lt;p&gt;In CDC pipelines, DDL operations such as adding or renaming columns frequently cause downstream writes to fail or produce unreadable files.&lt;/p&gt;

&lt;p&gt;By introducing Schema Evolution into the file connector framework, SeaTunnel enables file-based pipelines to adapt to schema changes automatically instead of treating files as static outputs.&lt;/p&gt;

&lt;h2&gt;
  
  
  Core Design
&lt;/h2&gt;

&lt;p&gt;A new configuration option has been added:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;static&lt;/span&gt; &lt;span class="kd"&gt;final&lt;/span&gt; &lt;span class="nc"&gt;Option&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;Boolean&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="no"&gt;SCHEMA_EVOLUTION_ENABLED&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt;
        &lt;span class="nc"&gt;Options&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;key&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"schema_evolution_enabled"&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
                &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;booleanType&lt;/span&gt;&lt;span class="o"&gt;()&lt;/span&gt;
                &lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;defaultValue&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="kc"&gt;false&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;When enabled, runtime DDL events—including:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;ADD COLUMN&lt;/li&gt;
&lt;li&gt;DROP COLUMN&lt;/li&gt;
&lt;li&gt;RENAME COLUMN&lt;/li&gt;
&lt;li&gt;UPDATE COLUMN&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;are propagated directly to the file sink.&lt;/p&gt;

&lt;p&gt;Rather than mixing incompatible schemas into a single file, SeaTunnel automatically rotates output files whenever a schema boundary is detected.&lt;/p&gt;

&lt;h2&gt;
  
  
  How It Works
&lt;/h2&gt;

&lt;p&gt;Schema evolution is introduced as an explicit runtime capability through the &lt;code&gt;schema_evolution_enabled&lt;/code&gt; option.&lt;/p&gt;

&lt;p&gt;Whenever the CDC source emits an ALTER TABLE event, the file sink:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;Detects the schema change.&lt;/li&gt;
&lt;li&gt;Finalizes the current output file.&lt;/li&gt;
&lt;li&gt;Creates a new file using the updated schema.&lt;/li&gt;
&lt;li&gt;Continues processing without interrupting the pipeline.&lt;/li&gt;
&lt;/ol&gt;

&lt;p&gt;This strategy preserves schema consistency while allowing long-running CDC jobs to continue uninterrupted.&lt;/p&gt;

&lt;h2&gt;
  
  
  Impact
&lt;/h2&gt;

&lt;p&gt;This feature delivers the greatest value for CDC-to-file scenarios, such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;MySQL CDC → Parquet&lt;/li&gt;
&lt;li&gt;MySQL CDC → ORC&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;It also makes downstream analytics significantly easier, since each file now corresponds to a single schema version rather than containing mixed record formats.&lt;/p&gt;

&lt;p&gt;More broadly, this enhancement brings file connectors much closer to the schema-aware behavior typically associated with modern streaming data platforms.&lt;/p&gt;

&lt;h1&gt;
  
  
  4.3 PR #11062: Calcite SQL Transform Plugin Brings a More Powerful Transformation Layer to SeaTunnel
&lt;/h1&gt;

&lt;p&gt;&lt;strong&gt;PR Size:&lt;/strong&gt; &lt;strong&gt;40 files changed, 8,257 insertions(+), 1 deletion(-)&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;This is one of the largest feature contributions merged in June.&lt;/p&gt;

&lt;h2&gt;
  
  
  Background
&lt;/h2&gt;

&lt;p&gt;SeaTunnel has long provided transformation capabilities between sources and sinks. However, previous transform operators primarily focused on predefined functionality, making it difficult to express complex business logic or extend SQL capabilities through reusable functions.&lt;/p&gt;

&lt;p&gt;As enterprise data pipelines become increasingly sophisticated, users expect a declarative SQL layer capable of handling rich transformations, custom business logic, and extensible function libraries without writing additional processing code.&lt;/p&gt;

&lt;p&gt;The introduction of the Calcite SQL Transform Plugin represents a significant step toward that goal.&lt;/p&gt;

&lt;p&gt;Rather than adding another transform operator, this PR introduces a unified SQL processing layer powered by Apache Calcite, allowing users to perform more sophisticated data transformations using familiar SQL syntax.&lt;/p&gt;

&lt;h2&gt;
  
  
  Core Design
&lt;/h2&gt;

&lt;p&gt;At the heart of this implementation is the new &lt;code&gt;CalciteUdf&lt;/code&gt; SPI, which defines a standard extension mechanism for user-defined functions.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="cm"&gt;/**
 * SPI for Calcite SQL transform UDFs. Implementations must provide a public static
 * eval method whose signature determines the SQL function's input/output types.
 */&lt;/span&gt;
&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;interface&lt;/span&gt; &lt;span class="nc"&gt;CalciteUdf&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The SPI specifies that every UDF must expose a &lt;strong&gt;public static &lt;code&gt;eval()&lt;/code&gt;&lt;/strong&gt; method, allowing Calcite's code generation engine to invoke functions directly without creating object instances.&lt;/p&gt;

&lt;p&gt;This design minimizes runtime overhead while providing a clean and standardized extension mechanism.&lt;/p&gt;

&lt;p&gt;In addition to the SPI itself, the PR also includes comprehensive documentation, end-to-end tests, sample configurations, and supporting infrastructure, making the feature production-ready rather than an experimental prototype.&lt;/p&gt;

&lt;h2&gt;
  
  
  How It Works
&lt;/h2&gt;

&lt;p&gt;Instead of hardcoding transformation logic into the engine, SeaTunnel now delegates SQL parsing, optimization, and execution to Apache Calcite.&lt;/p&gt;

&lt;p&gt;The plugin architecture enables users to:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Write more expressive SQL transformations&lt;/li&gt;
&lt;li&gt;Register custom UDFs through the SPI&lt;/li&gt;
&lt;li&gt;Extend built-in SQL functions without modifying the engine&lt;/li&gt;
&lt;li&gt;Share reusable function libraries across projects&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;By separating SQL execution from connector logic, SeaTunnel establishes a cleaner architecture in which data movement and data transformation evolve independently.&lt;/p&gt;

&lt;h2&gt;
  
  
  Impact
&lt;/h2&gt;

&lt;p&gt;For users, this dramatically improves the expressiveness of SQL-based data processing. Complex transformations that previously required custom Java development can now be implemented directly in SQL.&lt;/p&gt;

&lt;p&gt;For organizations, the standardized UDF extension mechanism makes it easier to encapsulate business logic into reusable function libraries, improving maintainability across multiple data pipelines.&lt;/p&gt;

&lt;p&gt;More importantly, this plugin lays the foundation for future enhancements, including richer built-in functions, enterprise UDF ecosystems, and more advanced SQL optimization capabilities.&lt;/p&gt;

&lt;p&gt;It represents another important step in SeaTunnel's evolution from a data synchronization framework to a comprehensive data integration platform.&lt;/p&gt;

&lt;h1&gt;
  
  
  4.4 PR #10812: StateStore Abstraction Marks a Key Architectural Milestone for Zeta
&lt;/h1&gt;

&lt;p&gt;&lt;strong&gt;PR Size:&lt;/strong&gt; &lt;strong&gt;28 files changed, 2,071 insertions(+), 92 deletions(-)&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Background
&lt;/h2&gt;

&lt;p&gt;As distributed execution engines evolve, state management becomes one of the most critical architectural components.&lt;/p&gt;

&lt;p&gt;Previously, portions of the Zeta engine relied directly on Hazelcast's &lt;code&gt;IMap&lt;/code&gt; implementation for state storage. While this approach simplified the initial implementation, it also introduced tight coupling between engine logic and a specific storage technology.&lt;/p&gt;

&lt;p&gt;Over time, this coupling becomes architectural debt.&lt;/p&gt;

&lt;p&gt;Supporting alternative state backends, introducing hybrid storage strategies, or implementing specialized capabilities such as TTL management becomes increasingly difficult when upper-layer components depend directly on Hazelcast APIs.&lt;/p&gt;

&lt;p&gt;This PR addresses that challenge by introducing a generic StateStore abstraction.&lt;/p&gt;

&lt;h2&gt;
  
  
  Core Design
&lt;/h2&gt;

&lt;p&gt;Rather than exposing Hazelcast semantics throughout the engine, the implementation introduces a hierarchy of capability-oriented interfaces.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;interface&lt;/span&gt; &lt;span class="nc"&gt;ExpiringStateStore&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="no"&gt;K&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="no"&gt;V&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="kd"&gt;extends&lt;/span&gt; &lt;span class="nc"&gt;StateStore&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="no"&gt;K&lt;/span&gt;&lt;span class="o"&gt;,&lt;/span&gt; &lt;span class="no"&gt;V&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Additional interfaces include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;&lt;code&gt;StateStore&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;&lt;code&gt;ExpiringStateStore&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;&lt;code&gt;IterableStateStore&lt;/code&gt;&lt;/li&gt;
&lt;li&gt;Other capability-specific abstractions&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Instead of programming against a concrete storage implementation, upper-layer components now depend only on the capabilities they require.&lt;/p&gt;

&lt;p&gt;This follows a classic interface-oriented architecture that significantly improves extensibility.&lt;/p&gt;

&lt;h2&gt;
  
  
  How It Works
&lt;/h2&gt;

&lt;p&gt;The new abstraction separates storage capabilities into independent interfaces.&lt;/p&gt;

&lt;p&gt;For example:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Basic key-value storage is defined by &lt;code&gt;StateStore&lt;/code&gt;
&lt;/li&gt;
&lt;li&gt;TTL-aware storage is represented by &lt;code&gt;ExpiringStateStore&lt;/code&gt;
&lt;/li&gt;
&lt;li&gt;Iterable storage capabilities are provided independently&lt;/li&gt;
&lt;li&gt;Additional storage features can be introduced without affecting existing implementations&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;This modular design prevents Hazelcast-specific concepts from leaking into the rest of the execution engine.&lt;/p&gt;

&lt;p&gt;As a result, storage implementations become interchangeable while the engine itself remains largely unchanged.&lt;/p&gt;

&lt;h2&gt;
  
  
  Impact
&lt;/h2&gt;

&lt;p&gt;Although end users may not notice immediate behavioral differences, this is one of the most strategically important architectural improvements merged in June.&lt;/p&gt;

&lt;p&gt;The new abstraction establishes the foundation for future capabilities such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Alternative state storage backends&lt;/li&gt;
&lt;li&gt;Hybrid storage architectures&lt;/li&gt;
&lt;li&gt;Independent state services&lt;/li&gt;
&lt;li&gt;More sophisticated TTL management&lt;/li&gt;
&lt;li&gt;Distributed counters&lt;/li&gt;
&lt;li&gt;Enhanced recovery mechanisms&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;In other words, this PR is less about adding visible features and more about enabling the next generation of Zeta's architecture.&lt;/p&gt;

&lt;h1&gt;
  
  
  4.5 PR #10485: BigQuery Sink Further Expands SeaTunnel's Cloud Data Warehouse Ecosystem
&lt;/h1&gt;

&lt;p&gt;&lt;strong&gt;PR Size:&lt;/strong&gt; &lt;strong&gt;40 files changed, 3,169 insertions(+), 1 deletion(-)&lt;/strong&gt;&lt;/p&gt;

&lt;h2&gt;
  
  
  Background
&lt;/h2&gt;

&lt;p&gt;As cloud-native data platforms continue to gain adoption, seamless integration with major cloud data warehouses has become increasingly important.&lt;/p&gt;

&lt;p&gt;For organizations running workloads on Google Cloud Platform, BigQuery is often the analytical database of choice. Until now, exporting data into BigQuery required additional tooling or custom integrations.&lt;/p&gt;

&lt;p&gt;The new BigQuery Sink closes this gap by making BigQuery a first-class destination within the SeaTunnel ecosystem.&lt;/p&gt;

&lt;p&gt;Importantly, this PR goes far beyond simply adding another connector.&lt;/p&gt;

&lt;p&gt;It includes configuration definitions, serialization logic, error handling, documentation, plugin registration, and integration support, providing a complete production-ready implementation.&lt;/p&gt;

&lt;h2&gt;
  
  
  Core Design
&lt;/h2&gt;

&lt;p&gt;The connector introduces a dedicated configuration class:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight java"&gt;&lt;code&gt;&lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;class&lt;/span&gt; &lt;span class="nc"&gt;BigQuerySinkOptions&lt;/span&gt; &lt;span class="o"&gt;{&lt;/span&gt;

    &lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;static&lt;/span&gt; &lt;span class="kd"&gt;final&lt;/span&gt; &lt;span class="nc"&gt;String&lt;/span&gt; &lt;span class="no"&gt;IDENTIFIER&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt; &lt;span class="s"&gt;"BigQuery"&lt;/span&gt;&lt;span class="o"&gt;;&lt;/span&gt;

    &lt;span class="kd"&gt;public&lt;/span&gt; &lt;span class="kd"&gt;static&lt;/span&gt; &lt;span class="kd"&gt;final&lt;/span&gt; &lt;span class="nc"&gt;Option&lt;/span&gt;&lt;span class="o"&gt;&amp;lt;&lt;/span&gt;&lt;span class="nc"&gt;String&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="no"&gt;PROJECT_ID&lt;/span&gt; &lt;span class="o"&gt;=&lt;/span&gt;
            &lt;span class="nc"&gt;Options&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="na"&gt;key&lt;/span&gt;&lt;span class="o"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"project_id"&lt;/span&gt;&lt;span class="o"&gt;)&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The configuration allows users to specify essential BigQuery settings such as project information while integrating seamlessly with SeaTunnel's existing connector framework.&lt;/p&gt;

&lt;h2&gt;
  
  
  How It Works
&lt;/h2&gt;

&lt;p&gt;The connector leverages SeaTunnel's unified Connector-V2 architecture, allowing users to write data into BigQuery using the same configuration model shared across the entire connector ecosystem.&lt;/p&gt;

&lt;p&gt;This consistent design minimizes the learning curve while simplifying deployment and maintenance across heterogeneous data platforms.&lt;/p&gt;

&lt;h2&gt;
  
  
  Impact
&lt;/h2&gt;

&lt;p&gt;The addition of the BigQuery Sink further strengthens SeaTunnel's support for modern cloud-native data architectures.&lt;/p&gt;

&lt;p&gt;Organizations operating in multi-cloud or hybrid-cloud environments can now integrate BigQuery into their data pipelines more naturally, reducing the need for custom development and improving interoperability across cloud services.&lt;/p&gt;

&lt;p&gt;Beyond the connector itself, this contribution reflects the community's continued investment in expanding SeaTunnel's cloud ecosystem while maintaining a consistent and unified user experience.&lt;/p&gt;

&lt;h1&gt;
  
  
  5. How Should These Improvements Be Evaluated?
&lt;/h1&gt;

&lt;p&gt;Unlike traditional performance-oriented releases, most of the major changes merged in June were not designed to increase throughput or reduce latency. Instead, they focused on improving &lt;strong&gt;governance, reliability, extensibility, compatibility, and operational resilience&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;As a result, evaluating these enhancements requires a different set of metrics. Looking only at rows per second or execution latency would fail to capture their real value.&lt;/p&gt;

&lt;p&gt;Instead, each feature should be validated according to its intended purpose.&lt;/p&gt;

&lt;h2&gt;
  
  
  Table-Level Fault Isolation (PR #10600)
&lt;/h2&gt;

&lt;p&gt;For table-level fault isolation, the focus should be on operational efficiency rather than raw performance.&lt;/p&gt;

&lt;p&gt;Recommended evaluation metrics include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;The percentage of tables that continue to complete successfully when one or more tables fail&lt;/li&gt;
&lt;li&gt;The completeness and accuracy of recorded failure information&lt;/li&gt;
&lt;li&gt;The reduction in recovery time and rerun costs&lt;/li&gt;
&lt;li&gt;The effectiveness of failure isolation in large-scale multi-table synchronization jobs&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These indicators better reflect how the feature improves production reliability.&lt;/p&gt;

&lt;h2&gt;
  
  
  Schema Evolution for File Connectors (PR #10744)
&lt;/h2&gt;

&lt;p&gt;Schema evolution should be evaluated based on pipeline continuity and data correctness.&lt;/p&gt;

&lt;p&gt;Key validation metrics include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Whether CDC pipelines continue running after DDL changes&lt;/li&gt;
&lt;li&gt;Whether output files are correctly rotated at schema boundaries&lt;/li&gt;
&lt;li&gt;Whether downstream analytics systems can continuously consume generated files&lt;/li&gt;
&lt;li&gt;Whether schema versions remain consistent across file partitions&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The primary objective is ensuring uninterrupted data delivery while preserving schema consistency.&lt;/p&gt;

&lt;h2&gt;
  
  
  Calcite SQL Transform Plugin (PR #11062)
&lt;/h2&gt;

&lt;p&gt;For the SQL transformation framework, evaluation should focus on functionality and extensibility rather than execution speed alone.&lt;/p&gt;

&lt;p&gt;Important validation areas include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Expressiveness of complex SQL transformations&lt;/li&gt;
&lt;li&gt;Ease of developing and registering custom UDFs&lt;/li&gt;
&lt;li&gt;Success rate of end-to-end SQL execution&lt;/li&gt;
&lt;li&gt;Compatibility with existing transformation workflows&lt;/li&gt;
&lt;li&gt;Stability of plugin loading and function discovery&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;These measurements demonstrate whether the new SQL layer can support increasingly sophisticated business scenarios.&lt;/p&gt;

&lt;h2&gt;
  
  
  StateStore Abstraction (PR #10812)
&lt;/h2&gt;

&lt;p&gt;Since the StateStore abstraction is primarily an architectural enhancement, validation should emphasize compatibility and maintainability.&lt;/p&gt;

&lt;p&gt;Recommended evaluation metrics include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Stability of state recovery workflows&lt;/li&gt;
&lt;li&gt;Compatibility across different StateStore implementations&lt;/li&gt;
&lt;li&gt;Regression testing results after replacing storage backends&lt;/li&gt;
&lt;li&gt;Correctness of capability-oriented interface implementations&lt;/li&gt;
&lt;li&gt;Overall engine stability during failover and recovery&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Ultimately, the success of this change lies in enabling future evolution without disrupting existing functionality.&lt;/p&gt;

&lt;p&gt;Taken together, these improvements highlight an important shift in how SeaTunnel should be evaluated.&lt;/p&gt;

&lt;p&gt;Rather than focusing exclusively on throughput benchmarks, June's work demonstrates growing maturity in areas such as reliability testing, disaster recovery, configuration compatibility, end-to-end validation, and production governance. These qualities are often the deciding factors for enterprise adoption.&lt;/p&gt;

&lt;h1&gt;
  
  
  6. Looking Ahead: Where Is Apache SeaTunnel Heading?
&lt;/h1&gt;

&lt;p&gt;The technical work completed in June reveals a clear direction for the project's evolution.&lt;/p&gt;

&lt;p&gt;While connector expansion remains important, the community is increasingly investing in the foundational capabilities required by modern enterprise data platforms.&lt;/p&gt;

&lt;h2&gt;
  
  
  Evolving Beyond a Connector Framework
&lt;/h2&gt;

&lt;p&gt;SeaTunnel has long been recognized for its broad connector ecosystem. However, recent developments show that the project is expanding well beyond data movement.&lt;/p&gt;

&lt;p&gt;Features such as Schema Evolution, the Calcite SQL Transform Plugin, table-level fault isolation, and the StateStore abstraction all extend beyond the scope of individual connectors.&lt;/p&gt;

&lt;p&gt;Together, they form the building blocks of a more comprehensive data integration platform capable of handling not only data ingestion, but also transformation, governance, recovery, and lifecycle management.&lt;/p&gt;

&lt;h2&gt;
  
  
  Strengthening Zeta as an Enterprise Execution Engine
&lt;/h2&gt;

&lt;p&gt;The Zeta engine continues to mature into a more intelligent and manageable execution platform.&lt;/p&gt;

&lt;p&gt;Recent improvements—including reporting of non-terminal job states, task workload monitoring, StateStore abstraction, enhanced recovery mechanisms, and expanded operational documentation—demonstrate a growing emphasis on observability, resilience, and operational governance.&lt;/p&gt;

&lt;p&gt;These capabilities are essential for organizations running large-scale production workloads, where stability and recoverability are just as important as execution performance.&lt;/p&gt;

&lt;h2&gt;
  
  
  Documentation Is Becoming a Strategic Asset
&lt;/h2&gt;

&lt;p&gt;Another notable trend is the community's increased investment in documentation.&lt;/p&gt;

&lt;p&gt;The high proportion of documentation-related pull requests in June does not simply reflect more written content. Instead, it represents a deliberate effort to capture and share operational knowledge that was previously scattered across code, discussions, and individual experience.&lt;/p&gt;

&lt;p&gt;Topics such as StateStore architecture, CDC production best practices, recovery workflows, and REST API lifecycle management are now documented in a more systematic way.&lt;/p&gt;

&lt;p&gt;For users, this reduces the learning curve and accelerates production adoption.&lt;/p&gt;

&lt;p&gt;For contributors, it establishes a stronger foundation for future collaboration.&lt;/p&gt;

&lt;p&gt;For the community as a whole, it makes Apache SeaTunnel more accessible, maintainable, and sustainable.&lt;/p&gt;

&lt;h2&gt;
  
  
  Final Thoughts
&lt;/h2&gt;

&lt;p&gt;With continued innovation across the engine, connectors, and developer experience, Apache SeaTunnel is becoming more powerful, more reliable, and easier to adopt. Thanks to every contributor who helped shape the project throughout June—we look forward to building an even stronger ecosystem together.&lt;/p&gt;

</description>
      <category>apacheseatunnel</category>
      <category>opensource</category>
      <category>datascience</category>
      <category>data</category>
    </item>
    <item>
      <title>🚀 Stream MySQL CDC to any HTTP API with Apache SeaTunnel! Build real-time integrations without direct DB access. ⚡🌐📦 #ApacheSeaTunnel #CDC #DataEngineering #OpenSource #MySQL</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 10 Jul 2026 02:17:53 +0000</pubDate>
      <link>https://dev.to/seatunnel/stream-mysql-cdc-to-any-http-api-with-apache-seatunnel-build-real-time-integrations-without-2mn1</link>
      <guid>https://dev.to/seatunnel/stream-mysql-cdc-to-any-http-api-with-apache-seatunnel-build-real-time-integrations-without-2mn1</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/seatunnel/how-to-stream-mysql-cdc-data-to-any-http-api-with-apache-seatunnel-52ej" class="crayons-story__hidden-navigation-link"&gt;How to Stream MySQL CDC Data to Any HTTP API with Apache SeaTunnel&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/seatunnel" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" alt="seatunnel profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/seatunnel" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Apache SeaTunnel
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Apache SeaTunnel
                
              
              &lt;div id="story-author-preview-content-4109602" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/seatunnel" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Apache SeaTunnel&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/seatunnel/how-to-stream-mysql-cdc-data-to-any-http-api-with-apache-seatunnel-52ej" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 10&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/seatunnel/how-to-stream-mysql-cdc-data-to-any-http-api-with-apache-seatunnel-52ej" id="article-link-4109602"&gt;
          How to Stream MySQL CDC Data to Any HTTP API with Apache SeaTunnel
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/mysql"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;mysql&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/apacheseatunnel"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;apacheseatunnel&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/http"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;http&lt;/a&gt;
            &lt;a class="crayons-tag  crayons-tag--monochrome " href="/t/datascience"&gt;&lt;span class="crayons-tag__prefix"&gt;#&lt;/span&gt;datascience&lt;/a&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/seatunnel/how-to-stream-mysql-cdc-data-to-any-http-api-with-apache-seatunnel-52ej#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>How to Stream MySQL CDC Data to Any HTTP API with Apache SeaTunnel</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Fri, 10 Jul 2026 02:10:37 +0000</pubDate>
      <link>https://dev.to/seatunnel/how-to-stream-mysql-cdc-data-to-any-http-api-with-apache-seatunnel-52ej</link>
      <guid>https://dev.to/seatunnel/how-to-stream-mysql-cdc-data-to-any-http-api-with-apache-seatunnel-52ej</guid>
      <description>&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fsgp3epuougnssxoeopvv.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fsgp3epuougnssxoeopvv.jpg" width="800" height="418"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;If your target system &lt;strong&gt;can't connect directly to your database&lt;/strong&gt; and only accepts data through &lt;strong&gt;HTTP APIs&lt;/strong&gt;, Apache SeaTunnel has you covered.&lt;/p&gt;

&lt;p&gt;In this tutorial, you'll learn how to build a real-time data pipeline that captures MySQL changes and pushes them directly to a web application's HTTP endpoint—no custom synchronization service required.&lt;/p&gt;

&lt;h2&gt;
  
  
  Prerequisites
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Install the Required Connectors
&lt;/h3&gt;

&lt;p&gt;Edit the &lt;code&gt;config/plugin_config&lt;/code&gt; file and add the following connectors:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;connector-http-base
connector-cdc-mysql
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Then install the plugins:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;sh bin/install-plugin.sh
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Tip:&lt;/strong&gt; If your network environment prevents automatic downloads, you can manually download the corresponding connector JARs from Maven Central and place them in the &lt;code&gt;connectors/&lt;/code&gt; directory.&lt;/p&gt;
&lt;/blockquote&gt;

&lt;h3&gt;
  
  
  Add the MySQL JDBC Driver
&lt;/h3&gt;

&lt;p&gt;Copy &lt;code&gt;mysql-connector-java-8.0.28.jar&lt;/code&gt; (or any compatible 8.x version) into the SeaTunnel &lt;code&gt;lib/&lt;/code&gt; directory.&lt;/p&gt;

&lt;h2&gt;
  
  
  Configure MySQL for CDC
&lt;/h2&gt;

&lt;p&gt;SeaTunnel's MySQL CDC connector reads data from the &lt;strong&gt;MySQL binary log (Binlog)&lt;/strong&gt;, so Binlog must be enabled before you begin.&lt;/p&gt;

&lt;p&gt;Open your MySQL configuration file (&lt;code&gt;my.cnf&lt;/code&gt; or &lt;code&gt;my.ini&lt;/code&gt;), add the following settings, and restart MySQL:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight ini"&gt;&lt;code&gt;&lt;span class="nn"&gt;[mysqld]&lt;/span&gt;
&lt;span class="py"&gt;server-id&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="s"&gt;1&lt;/span&gt;
&lt;span class="py"&gt;log-bin&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="s"&gt;mysql-bin&lt;/span&gt;
&lt;span class="py"&gt;binlog-format&lt;/span&gt; &lt;span class="p"&gt;=&lt;/span&gt; &lt;span class="s"&gt;ROW&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Configuration details:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;code&gt;server-id&lt;/code&gt;: Must be unique within the replication cluster.&lt;/li&gt;
&lt;li&gt;
&lt;code&gt;log-bin&lt;/code&gt;: Enables MySQL binary logging.&lt;/li&gt;
&lt;li&gt;
&lt;code&gt;binlog-format=ROW&lt;/code&gt;: Required for CDC to capture row-level changes accurately.&lt;/li&gt;
&lt;/ul&gt;

&lt;h2&gt;
  
  
  Build a Simple HTTP Receiver
&lt;/h2&gt;

&lt;p&gt;Next, let's create a lightweight HTTP service to receive data from SeaTunnel.&lt;/p&gt;

&lt;p&gt;Create a file named &lt;code&gt;server.go&lt;/code&gt;:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight go"&gt;&lt;code&gt;&lt;span class="k"&gt;package&lt;/span&gt; &lt;span class="n"&gt;main&lt;/span&gt;

&lt;span class="k"&gt;import&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;
    &lt;span class="s"&gt;"fmt"&lt;/span&gt;
    &lt;span class="s"&gt;"io"&lt;/span&gt;
    &lt;span class="s"&gt;"log"&lt;/span&gt;
    &lt;span class="s"&gt;"net/http"&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt;

&lt;span class="c"&gt;// Handle all HTTP requests and print the received payload&lt;/span&gt;
&lt;span class="k"&gt;func&lt;/span&gt; &lt;span class="n"&gt;handler&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;w&lt;/span&gt; &lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;ResponseWriter&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;r&lt;/span&gt; &lt;span class="o"&gt;*&lt;/span&gt;&lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Request&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
    &lt;span class="k"&gt;defer&lt;/span&gt; &lt;span class="n"&gt;r&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Body&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Close&lt;/span&gt;&lt;span class="p"&gt;()&lt;/span&gt;

    &lt;span class="n"&gt;body&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;err&lt;/span&gt; &lt;span class="o"&gt;:=&lt;/span&gt; &lt;span class="n"&gt;io&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;ReadAll&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;r&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Body&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="n"&gt;err&lt;/span&gt; &lt;span class="o"&gt;!=&lt;/span&gt; &lt;span class="no"&gt;nil&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;msg&lt;/span&gt; &lt;span class="o"&gt;:=&lt;/span&gt; &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Sprintf&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Failed to read request body: %v"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;err&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
        &lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Error&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;w&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;msg&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;StatusInternalServerError&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
        &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Println&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;msg&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
        &lt;span class="k"&gt;return&lt;/span&gt;
    &lt;span class="p"&gt;}&lt;/span&gt;

    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="nb"&gt;len&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;body&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="m"&gt;0&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Printf&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Request Body: %s&lt;/span&gt;&lt;span class="se"&gt;\n&lt;/span&gt;&lt;span class="s"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="kt"&gt;string&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;body&lt;/span&gt;&lt;span class="p"&gt;))&lt;/span&gt;
        &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Printf&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Payload Size: %d bytes&lt;/span&gt;&lt;span class="se"&gt;\n&lt;/span&gt;&lt;span class="s"&gt;"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="nb"&gt;len&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;body&lt;/span&gt;&lt;span class="p"&gt;))&lt;/span&gt;
    &lt;span class="p"&gt;}&lt;/span&gt; &lt;span class="k"&gt;else&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Println&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Empty request body"&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="p"&gt;}&lt;/span&gt;

    &lt;span class="n"&gt;w&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;WriteHeader&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;StatusOK&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Fprintf&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;w&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s"&gt;"Data received successfully. Payload size: %d bytes."&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="nb"&gt;len&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="n"&gt;body&lt;/span&gt;&lt;span class="p"&gt;))&lt;/span&gt;
&lt;span class="p"&gt;}&lt;/span&gt;

&lt;span class="k"&gt;func&lt;/span&gt; &lt;span class="n"&gt;main&lt;/span&gt;&lt;span class="p"&gt;()&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
    &lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;HandleFunc&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"/"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;handler&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;

    &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Println&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"HTTP server started on port 9090..."&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="n"&gt;fmt&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Println&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;`Test with: curl -X POST -d '{"test":123}' http://localhost:9090/`&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;

    &lt;span class="k"&gt;if&lt;/span&gt; &lt;span class="n"&gt;err&lt;/span&gt; &lt;span class="o"&gt;:=&lt;/span&gt; &lt;span class="n"&gt;http&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;ListenAndServe&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;":9090"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="no"&gt;nil&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt; &lt;span class="n"&gt;err&lt;/span&gt; &lt;span class="o"&gt;!=&lt;/span&gt; &lt;span class="no"&gt;nil&lt;/span&gt; &lt;span class="p"&gt;{&lt;/span&gt;
        &lt;span class="n"&gt;log&lt;/span&gt;&lt;span class="o"&gt;.&lt;/span&gt;&lt;span class="n"&gt;Fatal&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s"&gt;"Failed to start server: "&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="n"&gt;err&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
    &lt;span class="p"&gt;}&lt;/span&gt;
&lt;span class="p"&gt;}&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Start the service:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;go run server.go
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Once you see &lt;strong&gt;"HTTP server started on port 9090..."&lt;/strong&gt;, your API endpoint is ready to receive data.&lt;/p&gt;

&lt;h2&gt;
  
  
  Prepare Test Data
&lt;/h2&gt;

&lt;p&gt;Create a sample table in MySQL and insert two initial records:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;TABLE&lt;/span&gt; &lt;span class="nv"&gt;`post`&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;
  &lt;span class="nv"&gt;`id`&lt;/span&gt; &lt;span class="nb"&gt;int&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;11&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;NOT&lt;/span&gt; &lt;span class="k"&gt;NULL&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="nv"&gt;`content`&lt;/span&gt; &lt;span class="nb"&gt;varchar&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;50&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;DEFAULT&lt;/span&gt; &lt;span class="k"&gt;NULL&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="nv"&gt;`author`&lt;/span&gt; &lt;span class="nb"&gt;varchar&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;50&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;DEFAULT&lt;/span&gt; &lt;span class="k"&gt;NULL&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="k"&gt;PRIMARY&lt;/span&gt; &lt;span class="k"&gt;KEY&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="nv"&gt;`id`&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="n"&gt;ENGINE&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;InnoDB&lt;/span&gt; &lt;span class="k"&gt;DEFAULT&lt;/span&gt; &lt;span class="n"&gt;CHARSET&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="nv"&gt;`post`&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'Getting Started with MySQL'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'Alice'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;
&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="nv"&gt;`post`&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;2&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'Understanding HTTP'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'Bob'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Create the SeaTunnel Job
&lt;/h2&gt;

&lt;p&gt;Create a file named &lt;code&gt;mysqlcdc_http.conf&lt;/code&gt; under the &lt;code&gt;job/&lt;/code&gt; directory.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight hocon"&gt;&lt;code&gt;&lt;span class="nl"&gt;env&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;parallelism&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;job.mode&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"STREAMING"&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;checkpoint.interval&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="mi"&gt;10000&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="nl"&gt;source&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;MySQL-CDC&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;username&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;password&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;table-names&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="w"&gt;
      &lt;/span&gt;&lt;span class="s2"&gt;"your_database.post"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="k"&gt;url&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"jdbc:mysql://your-mysql-host:3306/your_database"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;schema-changes.enabled&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;true&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="nl"&gt;sink&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;Http&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="k"&gt;url&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"http://your-http-server:9090/"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;headers&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
      &lt;/span&gt;&lt;span class="nl"&gt;Accept&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"application/json"&lt;/span&gt;&lt;span class="w"&gt;
      &lt;/span&gt;&lt;span class="nl"&gt;Content-Type&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"application/json;charset=utf-8"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Before running the job, replace the following with your own environment:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;MySQL hostname&lt;/li&gt;
&lt;li&gt;Database name&lt;/li&gt;
&lt;li&gt;Username and password&lt;/li&gt;
&lt;li&gt;HTTP endpoint URL&lt;/li&gt;
&lt;/ul&gt;

&lt;h2&gt;
  
  
  Start the Streaming Job
&lt;/h2&gt;

&lt;p&gt;From the SeaTunnel installation directory, run:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;bin/seatunnel.sh &lt;span class="nt"&gt;--config&lt;/span&gt; job/mysqlcdc_http.conf &lt;span class="nt"&gt;-m&lt;/span&gt; &lt;span class="nb"&gt;local&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;SeaTunnel will start capturing MySQL changes and stream them to your HTTP service in real time.&lt;/p&gt;

&lt;h2&gt;
  
  
  Verify Real-Time Synchronization
&lt;/h2&gt;

&lt;p&gt;After the job starts successfully, you'll immediately see the two existing records being pushed to your HTTP service:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Starting HTTP server on port 9090...

Request Body:
{"id":1,"content":"Getting Started with MySQL","author":"Alice"}
Payload Size: 62 bytes

Request Body:
{"id":2,"content":"Understanding HTTP","author":"Bob"}
Payload Size: 56 bytes
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;SeaTunnel first performs an initial snapshot, so existing records are synchronized automatically.&lt;/p&gt;

&lt;p&gt;Now insert a new record into MySQL:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="nv"&gt;`post`&lt;/span&gt;
&lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;3&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'Real-Time Sync with SeaTunnel'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'Charlie'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;The HTTP service immediately receives the new data:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;Request Body:
{"id":3,"content":"Real-Time Sync with SeaTunnel","author":"Charlie"}

Payload Size: 69 bytes
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This confirms that incremental changes are being captured and delivered to your HTTP endpoint in real time.&lt;/p&gt;

&lt;h2&gt;
  
  
  Conclusion
&lt;/h2&gt;

&lt;p&gt;With Apache SeaTunnel, building a &lt;strong&gt;real-time MySQL-to-HTTP data pipeline&lt;/strong&gt; is straightforward.&lt;/p&gt;

&lt;p&gt;Instead of granting downstream systems direct database access, you can stream data securely through standard HTTP APIs. This architecture is especially useful for:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Integrating with third-party SaaS platforms&lt;/li&gt;
&lt;li&gt;Connecting internal business systems across departments&lt;/li&gt;
&lt;li&gt;Feeding custom web services or microservices&lt;/li&gt;
&lt;li&gt;Building event-driven applications&lt;/li&gt;
&lt;li&gt;Decoupling data producers from downstream consumers&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Whether you're integrating legacy systems, modern web services, or external business platforms, SeaTunnel's CDC and HTTP connectors provide a flexible, scalable, and production-ready solution for real-time data delivery.&lt;/p&gt;

</description>
      <category>mysql</category>
      <category>apacheseatunnel</category>
      <category>http</category>
      <category>datascience</category>
    </item>
    <item>
      <title>Why Fivetran and Airbyte Still Fall Short for Enterprise Data Ingestion</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Thu, 09 Jul 2026 09:38:06 +0000</pubDate>
      <link>https://dev.to/seatunnel/why-fivetran-and-airbyte-still-fall-short-for-enterprise-data-ingestion-nog</link>
      <guid>https://dev.to/seatunnel/why-fivetran-and-airbyte-still-fall-short-for-enterprise-data-ingestion-nog</guid>
      <description>&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F5agicimyvp3g5xi7n7fy.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F5agicimyvp3g5xi7n7fy.jpg" width="800" height="371"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;In Reddit's &lt;strong&gt;r/dataengineering&lt;/strong&gt; community, discussions about choosing a data ingestion platform never seem to end. Whenever a team is building a modern data stack (MDS) for production, one question inevitably comes up:&lt;/p&gt;

&lt;blockquote&gt;
&lt;p&gt;Should we choose Fivetran? Airbyte? Or is there a more capable open-source alternative?&lt;/p&gt;
&lt;/blockquote&gt;

&lt;p&gt;After observing countless discussions from data engineers, one thing has become increasingly clear:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;No ELT platform can cover every enterprise data ingestion scenario.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;However, if you look beyond connector counts and evaluate real-world production environments, the problem runs much deeper. Today's data engineering teams are not simply struggling with missing connectors—they're dealing with architectural limitations, security and compliance requirements, and the reliability challenges of distributed data movement.&lt;/p&gt;

&lt;p&gt;The real challenge isn't building another connector.&lt;/p&gt;

&lt;p&gt;It's building a data ingestion platform that works reliably at enterprise scale.&lt;/p&gt;

&lt;h2&gt;
  
  
  Traditional ELT-Based Data Ingestion Is Reaching Its Limits. SeaTunnel Takes a Different Approach.
&lt;/h2&gt;

&lt;p&gt;Across Reddit, experienced data engineers repeatedly describe the same production issues. Whether using commercial SaaS products or open-source ELT tools, the same three pain points continue to surface.&lt;/p&gt;

&lt;h3&gt;
  
  
  1. Turning Streaming Workloads into Micro-Batches Comes at a High Cost
&lt;/h3&gt;

&lt;p&gt;Many organizations require near real-time synchronization from operational databases such as PostgreSQL or third-party SaaS applications. A 15-minute SLA—or even lower—is becoming the standard rather than the exception.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The challenge&lt;/strong&gt;：&lt;/p&gt;

&lt;p&gt;Traditional ELT platforms are fundamentally built around scheduled micro-batch execution.&lt;/p&gt;

&lt;p&gt;To satisfy a 15-minute SLA, they continuously poll databases or repeatedly call APIs. As synchronization frequency increases, organizations often encounter:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Higher CPU utilization on source databases&lt;/li&gt;
&lt;li&gt;Expensive API rate-limit penalties (for example, Salesforce)&lt;/li&gt;
&lt;li&gt;Long-running historical synchronization jobs&lt;/li&gt;
&lt;li&gt;Endless retry loops during large backfills&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  SeaTunnel's Batch-Stream Integrated Architecture
&lt;/h3&gt;

&lt;p&gt;Instead of forcing streaming scenarios into micro-batches, SeaTunnel provides a unified &lt;strong&gt;Batch-Stream Integration API&lt;/strong&gt; at the ingestion layer.&lt;/p&gt;

&lt;p&gt;For production databases such as PostgreSQL, a single ingestion job can seamlessly combine historical loading and CDC streaming.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Automatic Full + Incremental Synchronization&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;When a job starts:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;SeaTunnel first performs a high-speed batch snapshot of historical data.&lt;/li&gt;
&lt;li&gt;Once the snapshot completes, it automatically records the latest log position.&lt;/li&gt;
&lt;li&gt;The pipeline immediately switches to CDC mode by continuously consuming PostgreSQL WAL or MySQL Binlog changes.&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;No manual intervention.&lt;/p&gt;

&lt;p&gt;No duplicated data.&lt;/p&gt;

&lt;p&gt;No synchronization gap.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Eliminating Source Database Pressure&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Once incremental synchronization begins, SeaTunnel reads directly from database logs instead of repeatedly scanning tables.&lt;/p&gt;

&lt;p&gt;The result is:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Millisecond-level latency instead of 15-minute polling cycles&lt;/li&gt;
&lt;li&gt;Minimal CPU overhead on production databases&lt;/li&gt;
&lt;li&gt;No excessive API requests&lt;/li&gt;
&lt;li&gt;No unnecessary polling traffic&lt;/li&gt;
&lt;/ul&gt;

&lt;h2&gt;
  
  
  2. PII Should Never Travel Through Your Pipeline Unprotected
&lt;/h2&gt;

&lt;p&gt;Fivetran follows a strict ELT philosophy:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Load first. Transform later.&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;That means personally identifiable information (PII) is first copied into the warehouse before tools like dbt perform masking or transformations.&lt;/p&gt;

&lt;p&gt;For highly regulated industries—including finance, healthcare, and government—this creates an immediate compliance problem.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The challenge&lt;/strong&gt;：&lt;/p&gt;

&lt;p&gt;Sensitive information reaches the warehouse before it is anonymized.&lt;/p&gt;

&lt;p&gt;Many organizations are forced to build custom preprocessing services using Python proxies or middleware simply to satisfy internal compliance requirements.&lt;/p&gt;

&lt;p&gt;As a result, what should have been a simple ELT deployment becomes increasingly complicated.&lt;/p&gt;

&lt;h3&gt;
  
  
  SeaTunnel Protects Sensitive Data Before It Lands
&lt;/h3&gt;

&lt;p&gt;As an open-source data ingestion platform, SeaTunnel removes this architectural constraint by introducing lightweight streaming transformations directly into the ingestion pipeline.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;In-Flight Data Masking&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Before records are written into Snowflake, Iceberg, or any downstream storage, SeaTunnel performs transformations entirely in memory.&lt;/p&gt;

&lt;p&gt;Using built-in operators such as:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Replace&lt;/li&gt;
&lt;li&gt;Filter&lt;/li&gt;
&lt;li&gt;FieldMapper&lt;/li&gt;
&lt;li&gt;Custom UDF plugins&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Engineers can:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Hash email addresses with SHA-256&lt;/li&gt;
&lt;li&gt;Remove sensitive columns&lt;/li&gt;
&lt;li&gt;Replace confidential values&lt;/li&gt;
&lt;li&gt;Apply enterprise-specific masking logic&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Sensitive information is transformed &lt;strong&gt;before it ever reaches the destination system&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Compliance is enforced directly inside the ingestion layer rather than being delegated to downstream processing.&lt;/p&gt;

&lt;p&gt;That significantly reduces engineering complexity while simplifying security audits.&lt;/p&gt;

&lt;h2&gt;
  
  
  3. The Long-Tail Connector Trap in Open Source
&lt;/h2&gt;

&lt;p&gt;Airbyte has grown rapidly thanks to its community-driven connector ecosystem.&lt;/p&gt;

&lt;p&gt;Connector quantity, however, doesn't necessarily translate into production readiness.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The challenge&lt;/strong&gt;：&lt;/p&gt;

&lt;p&gt;Many long-tail connectors are implemented as lightweight Python wrappers.&lt;/p&gt;

&lt;p&gt;They work well for small datasets but often struggle when processing billions of records during historical synchronization.&lt;/p&gt;

&lt;p&gt;Without distributed partitioning or parallel extraction capabilities, common production issues include:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Task failures&lt;/li&gt;
&lt;li&gt;Pipeline hangs&lt;/li&gt;
&lt;li&gt;Memory exhaustion&lt;/li&gt;
&lt;li&gt;Constant manual restarts&lt;/li&gt;
&lt;/ul&gt;

&lt;h3&gt;
  
  
  SeaTunnel Solves This with the Zeta Engine
&lt;/h3&gt;

&lt;p&gt;Rather than stitching together scripting frameworks, SeaTunnel built its own distributed execution engine specifically for data synchronization.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Distributed Data Splitting&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Large datasets are automatically partitioned into thousands of parallel splits.&lt;/p&gt;

&lt;p&gt;Each split is processed independently across distributed workers, maximizing available bandwidth—even for legacy storage systems.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Checkpointing and Exactly-Once Recovery&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Built upon the principles of the Chandy-Lamport distributed snapshot algorithm, the Zeta Engine provides native checkpointing and fault recovery.&lt;/p&gt;

&lt;p&gt;If an API rate limit, network interruption, or node failure occurs, SeaTunnel automatically resumes from the last consistent checkpoint.&lt;/p&gt;

&lt;p&gt;The result is &lt;strong&gt;Exactly-Once&lt;/strong&gt; data delivery without requiring operators to restart production jobs in the middle of the night.&lt;/p&gt;

&lt;h2&gt;
  
  
  Reimagining Data Ingestion: How Apache SeaTunnel Builds a Unified Data Movement Layer
&lt;/h2&gt;

&lt;p&gt;Within the open-source community, SeaTunnel is increasingly recognized as a unified data movement platform.&lt;/p&gt;

&lt;p&gt;With approximately &lt;strong&gt;200+ connectors&lt;/strong&gt; covering databases, messaging systems, files, data lakes, data warehouses, search engines, and vector databases, it provides a single platform for enterprise-scale data movement.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzkapfkushhp067o7lwgn.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzkapfkushhp067o7lwgn.jpg" width="800" height="267"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;Its architecture is fundamentally different from traditional ELT tools.&lt;/p&gt;

&lt;p&gt;Instead of optimizing individual connectors, SeaTunnel redesigns the entire data ingestion layer.&lt;/p&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F7idxo19102oseb1dnxkm.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F7idxo19102oseb1dnxkm.jpg" width="799" height="592"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  1. Unified Abstraction Through SeaTunnelRow
&lt;/h3&gt;

&lt;p&gt;Connecting hundreds of heterogeneous systems traditionally creates an M×N integration problem.&lt;/p&gt;

&lt;p&gt;SeaTunnel solves this by introducing a unified internal data model called &lt;strong&gt;SeaTunnelRow&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Whether the source contains:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;MySQL VARCHAR fields&lt;/li&gt;
&lt;li&gt;Elasticsearch Objects&lt;/li&gt;
&lt;li&gt;Milvus FloatVectors&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;all data is converted into a standardized internal representation before processing.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;The advantage&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Developers only need to build a Source connector once.&lt;/p&gt;

&lt;p&gt;That connector can immediately work with every existing Sink supported by SeaTunnel.&lt;/p&gt;

&lt;p&gt;This architecture enables rapid support for both legacy enterprise systems and emerging AI infrastructure without rebuilding the entire connector ecosystem.&lt;/p&gt;

&lt;h3&gt;
  
  
  2. A Distributed Engine Built Specifically for Data Movement
&lt;/h3&gt;

&lt;p&gt;Historically, organizations relied on Spark or Flink to achieve high-performance synchronization.&lt;/p&gt;

&lt;p&gt;While powerful, those engines also introduce significant operational complexity through Kubernetes, YARN, and large distributed clusters.&lt;/p&gt;

&lt;p&gt;SeaTunnel approaches the problem differently.&lt;/p&gt;

&lt;p&gt;Its distributed execution engine —— &lt;strong&gt;Zeta Engine&lt;/strong&gt; is purpose-built for data synchronization.&lt;/p&gt;

&lt;p&gt;Key capabilities include:&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Distributed Data Splitting&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Large datasets are automatically divided into thousands of independent splits executed by distributed TaskGroups, maximizing throughput across databases, files, and cloud storage.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Dynamic Thread Sharing&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Traditional synchronization frameworks often dedicate one thread per task.&lt;/p&gt;

&lt;p&gt;At enterprise scale—with thousands of tables—that model quickly wastes CPU resources due to excessive context switching.&lt;/p&gt;

&lt;p&gt;SeaTunnel introduces &lt;strong&gt;Dynamic Thread Sharing&lt;/strong&gt;, allowing many synchronization tasks to reuse a shared thread pool and dramatically improve resource utilization.&lt;/p&gt;

&lt;p&gt;&lt;strong&gt;Native Two-Phase Commit and Distributed Checkpointing&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Enterprise-grade data movement requires financial-level consistency.&lt;/p&gt;

&lt;p&gt;The Zeta Engine natively implements distributed checkpointing together with Two-Phase Commit (2PC), enabling &lt;strong&gt;Exactly-Once&lt;/strong&gt; guarantees without depending on heavyweight external processing engines.&lt;/p&gt;

&lt;h2&gt;
  
  
  From Another Connector Tool to Open Data Infrastructure
&lt;/h2&gt;

&lt;p&gt;The conversations happening every day on Reddit reveal an important trend.&lt;/p&gt;

&lt;p&gt;The next generation of data integration is no longer about:&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Better dashboards&lt;/li&gt;
&lt;li&gt;More polished UIs&lt;/li&gt;
&lt;li&gt;Or adding a few more SaaS connectors&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;The real challenge lies inside the enterprise.&lt;/p&gt;

&lt;p&gt;Organizations need to move data across legacy systems, modern cloud platforms, streaming infrastructure, data lakes, warehouses, and increasingly, AI-native vector databases—while simultaneously meeting strict compliance requirements and real-time SLAs.&lt;/p&gt;

&lt;p&gt;Fivetran and Airbyte have undoubtedly transformed modern ELT and made data integration significantly easier.&lt;/p&gt;

&lt;p&gt;But as enterprise architectures become increasingly distributed and real-time, the industry is beginning to demand something beyond traditional ELT.&lt;/p&gt;

&lt;p&gt;It needs an open, distributed, batch-and-stream integrated data ingestion layer that is designed for production from the ground up.&lt;/p&gt;

&lt;p&gt;With its purpose-built distributed Zeta Engine, native batch-stream integration, approximately &lt;strong&gt;200+ connectors&lt;/strong&gt;, and support for databases, data lakes, warehouses, messaging systems, files, search platforms, and AI vector databases, &lt;strong&gt;Apache SeaTunnel&lt;/strong&gt; is evolving from a synchronization tool into a foundational layer for modern data infrastructure.&lt;/p&gt;

&lt;h2&gt;
  
  
  References
&lt;/h2&gt;

&lt;ul&gt;
&lt;li&gt;
&lt;strong&gt;Apache SeaTunnel GitHub Repository&lt;/strong&gt;: &lt;a href="https://github.com/apache/seatunnel" rel="noopener noreferrer"&gt;https://github.com/apache/seatunnel&lt;/a&gt;
&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Apache SeaTunnel Official Website&lt;/strong&gt;: &lt;a href="https://seatunnel.apache.org/" rel="noopener noreferrer"&gt;https://seatunnel.apache.org/&lt;/a&gt;
&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;Download Apache SeaTunnel&lt;/strong&gt;: &lt;a href="https://seatunnel.apache.org/download" rel="noopener noreferrer"&gt;https://seatunnel.apache.org/download&lt;/a&gt;
&lt;/li&gt;
&lt;li&gt;
&lt;strong&gt;WhaleOps Official Website&lt;/strong&gt;: &lt;a href="https://www.whaleops.io/" rel="noopener noreferrer"&gt;https://www.whaleops.io/&lt;/a&gt;
&lt;/li&gt;
&lt;/ul&gt;

</description>
      <category>fivetran</category>
      <category>airbyte</category>
      <category>apacheseatunnel</category>
      <category>datascience</category>
    </item>
    <item>
      <title>🔑 No more primary key conflicts! Merge data from multiple MySQL databases seamlessly with Apache SeaTunnel. ⚡
#ApacheSeaTunnel #MySQL #CDC #DataEngineering #OpenSource</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Thu, 02 Jul 2026 03:29:29 +0000</pubDate>
      <link>https://dev.to/seatunnel/no-more-primary-key-conflicts-merge-data-from-multiple-mysql-databases-seamlessly-with-apache-2iga</link>
      <guid>https://dev.to/seatunnel/no-more-primary-key-conflicts-merge-data-from-multiple-mysql-databases-seamlessly-with-apache-2iga</guid>
      <description>&lt;div class="ltag__link--embedded"&gt;
  &lt;div class="crayons-story "&gt;
  &lt;a href="https://dev.to/seatunnel/how-apache-seatunnel-eliminates-primary-key-conflicts-when-consolidating-data-from-multiple-tables-1mda" class="crayons-story__hidden-navigation-link"&gt;How Apache SeaTunnel Eliminates Primary Key Conflicts When Consolidating Data from Multiple Tables&lt;/a&gt;


  &lt;div class="crayons-story__body crayons-story__body-full_post"&gt;
    &lt;div class="crayons-story__top"&gt;
      &lt;div class="crayons-story__meta"&gt;
        &lt;div class="crayons-story__author-pic"&gt;

          &lt;a href="/seatunnel" class="crayons-avatar  crayons-avatar--l  "&gt;
            &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" alt="seatunnel profile" class="crayons-avatar__image"&gt;
          &lt;/a&gt;
        &lt;/div&gt;
        &lt;div&gt;
          &lt;div&gt;
            &lt;a href="/seatunnel" class="crayons-story__secondary fw-medium m:hidden"&gt;
              Apache SeaTunnel
            &lt;/a&gt;
            &lt;div class="profile-preview-card relative mb-4 s:mb-0 fw-medium hidden m:inline-block"&gt;
              
                Apache SeaTunnel
                
              
              &lt;div id="story-author-preview-content-4046684" class="profile-preview-card__content crayons-dropdown branded-7 p-4 pt-0"&gt;
                &lt;div class="gap-4 grid"&gt;
                  &lt;div class="-mt-4"&gt;
                    &lt;a href="/seatunnel" class="flex"&gt;
                      &lt;span class="crayons-avatar crayons-avatar--xl mr-2 shrink-0"&gt;
                        &lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Fuser%2Fprofile_image%2F844122%2Fc6155eb3-df58-448b-8d88-36865c4f1d84.jpg" class="crayons-avatar__image" alt=""&gt;
                      &lt;/span&gt;
                      &lt;span class="crayons-link crayons-subtitle-2 mt-5"&gt;Apache SeaTunnel&lt;/span&gt;
                    &lt;/a&gt;
                  &lt;/div&gt;
                  &lt;div class="print-hidden"&gt;
                    
                      Follow
                    
                  &lt;/div&gt;
                  &lt;div class="author-preview-metadata-container"&gt;&lt;/div&gt;
                &lt;/div&gt;
              &lt;/div&gt;
            &lt;/div&gt;

          &lt;/div&gt;
          &lt;a href="https://dev.to/seatunnel/how-apache-seatunnel-eliminates-primary-key-conflicts-when-consolidating-data-from-multiple-tables-1mda" class="crayons-story__tertiary fs-xs"&gt;&lt;time&gt;Jul 2&lt;/time&gt;&lt;span class="time-ago-indicator-initial-placeholder"&gt;&lt;/span&gt;&lt;/a&gt;
        &lt;/div&gt;
      &lt;/div&gt;

    &lt;/div&gt;

    &lt;div class="crayons-story__indention"&gt;
      &lt;h2 class="crayons-story__title crayons-story__title-full_post"&gt;
        &lt;a href="https://dev.to/seatunnel/how-apache-seatunnel-eliminates-primary-key-conflicts-when-consolidating-data-from-multiple-tables-1mda" id="article-link-4046684"&gt;
          How Apache SeaTunnel Eliminates Primary Key Conflicts When Consolidating Data from Multiple Tables
        &lt;/a&gt;
      &lt;/h2&gt;
        &lt;div class="crayons-story__tags"&gt;
        &lt;/div&gt;
      &lt;div class="crayons-story__bottom"&gt;
        &lt;div class="crayons-story__details"&gt;
            &lt;a href="https://dev.to/seatunnel/how-apache-seatunnel-eliminates-primary-key-conflicts-when-consolidating-data-from-multiple-tables-1mda#comments" class="crayons-btn crayons-btn--s crayons-btn--ghost crayons-btn--icon-left flex items-center"&gt;
              

              &lt;span class="hidden s:inline"&gt;Add&amp;nbsp;Comment&lt;/span&gt;
            &lt;/a&gt;
        &lt;/div&gt;
        &lt;div class="crayons-story__save"&gt;
          &lt;small class="crayons-story__tertiary fs-xs mr-2"&gt;
            4 min read
          &lt;/small&gt;
            
              &lt;span class="bm-initial crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
              &lt;span class="bm-success crayons-icon c-btn__icon"&gt;
                

              &lt;/span&gt;
            
        &lt;/div&gt;
      &lt;/div&gt;
    &lt;/div&gt;
  &lt;/div&gt;
&lt;/div&gt;

&lt;/div&gt;


</description>
    </item>
    <item>
      <title>How Apache SeaTunnel Eliminates Primary Key Conflicts When Consolidating Data from Multiple Tables</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Thu, 02 Jul 2026 03:27:22 +0000</pubDate>
      <link>https://dev.to/seatunnel/how-apache-seatunnel-eliminates-primary-key-conflicts-when-consolidating-data-from-multiple-tables-1mda</link>
      <guid>https://dev.to/seatunnel/how-apache-seatunnel-eliminates-primary-key-conflicts-when-consolidating-data-from-multiple-tables-1mda</guid>
      <description>&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzjk0osif5kavauw6z86p.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fzjk0osif5kavauw6z86p.jpg" width="800" height="96"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;p&gt;⭐ Star Apache SeaTunnel on GitHub:&lt;br&gt;
&lt;a href="https://github.com/apache/seatunnel" rel="noopener noreferrer"&gt;https://github.com/apache/seatunnel&lt;/a&gt;&lt;/p&gt;
&lt;h2&gt;
  
  
  Overview
&lt;/h2&gt;

&lt;p&gt;In many real-world scenarios, multiple databases may contain tables with similar data. For example, data from different business systems may be stored in tables with the same schema but located in separate databases.&lt;/p&gt;

&lt;p&gt;When these tables need to be consolidated into a single table for reporting and analytics, a common challenge arises: because the source tables share the same primary key design, directly merging the data results in duplicate primary keys.&lt;/p&gt;

&lt;p&gt;Apache SeaTunnel provides an elegant solution to this challenge. In this article, we'll walk through how Apache SeaTunnel resolves this issue in a simple and effective way.&lt;/p&gt;
&lt;h2&gt;
  
  
  Solution Design
&lt;/h2&gt;

&lt;p&gt;This solution is designed to synchronize the &lt;code&gt;test&lt;/code&gt; table from two independent MySQL databases (&lt;code&gt;source1&lt;/code&gt; and &lt;code&gt;source2&lt;/code&gt;) into the &lt;code&gt;test&lt;/code&gt; table of a third database (&lt;code&gt;source3&lt;/code&gt;) in real time and with high accuracy.&lt;/p&gt;

&lt;p&gt;To support real-time synchronization and capture data changes, the solution uses Change Data Capture (CDC) based on the MySQL Binary Log (Binlog).&lt;/p&gt;

&lt;p&gt;To prevent duplicate auto-increment primary key IDs from different source tables, the destination table introduces an additional &lt;code&gt;sources&lt;/code&gt; column. This column, together with the original &lt;code&gt;id&lt;/code&gt;, forms a composite primary key, ensuring both data uniqueness and traceability.&lt;/p&gt;
&lt;h2&gt;
  
  
  Prerequisites
&lt;/h2&gt;

&lt;ol&gt;
&lt;li&gt;&lt;p&gt;MySQL 5.7&lt;/p&gt;&lt;/li&gt;
&lt;li&gt;&lt;p&gt;Apache SeaTunnel 2.3.12&lt;/p&gt;&lt;/li&gt;
&lt;/ol&gt;
&lt;h2&gt;
  
  
  Verify the MySQL Configuration
&lt;/h2&gt;

&lt;p&gt;First, verify that MySQL Binary Logging (Binlog) is enabled.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="n"&gt;mysql&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="k"&gt;show&lt;/span&gt; &lt;span class="n"&gt;variables&lt;/span&gt; &lt;span class="k"&gt;where&lt;/span&gt; &lt;span class="n"&gt;variable_name&lt;/span&gt; &lt;span class="k"&gt;in&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="s1"&gt;'log_bin'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'binlog_format'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'binlog_row_image'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'gtid_mode'&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="s1"&gt;'enforce_gtid_consistency'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;

&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;--------------------------+-------+&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;Variable_name&lt;/span&gt;            &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;Value&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;--------------------------+-------+&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;binlog_format&lt;/span&gt;            &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="k"&gt;ROW&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;binlog_row_image&lt;/span&gt;         &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="k"&gt;FULL&lt;/span&gt;  &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;enforce_gtid_consistency&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="k"&gt;OFF&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;gtid_mode&lt;/span&gt;                &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="k"&gt;OFF&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;log_bin&lt;/span&gt;                  &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="k"&gt;ON&lt;/span&gt;    &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;--------------------------+-------+&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;If the value of &lt;code&gt;log_bin&lt;/code&gt; is not &lt;code&gt;ON&lt;/code&gt;, update the MySQL configuration file (&lt;code&gt;mysql.cnf&lt;/code&gt;) as follows:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight properties"&gt;&lt;code&gt;&lt;span class="py"&gt;log-bin&lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="s"&gt;mysql-bin&lt;/span&gt;
&lt;span class="py"&gt;server-id&lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="s"&gt;1&lt;/span&gt;
&lt;span class="py"&gt;binlog_format&lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="s"&gt;ROW&lt;/span&gt;
&lt;span class="py"&gt;binlog_checksum&lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="s"&gt;NONE&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Restart the MySQL service after updating the configuration.&lt;/p&gt;

&lt;h2&gt;
  
  
  Configure Apache SeaTunnel
&lt;/h2&gt;

&lt;h3&gt;
  
  
  1. Install the Required Connectors
&lt;/h3&gt;

&lt;p&gt;Add the following connectors to the &lt;code&gt;${SEATUNNEL_HOME}/config/plugin_config&lt;/code&gt; file:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;connector-cdc-mysql
connector-jdbc
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Then install the connectors.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;sh bin/install-plugin.sh
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  2. Install the MySQL JDBC Driver
&lt;/h3&gt;

&lt;p&gt;This example uses &lt;code&gt;mysql-connector-java-8.0.28.jar&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;Copy the JAR file to the &lt;code&gt;${SEATUNNEL_HOME}/lib/&lt;/code&gt; directory.&lt;/p&gt;

&lt;h2&gt;
  
  
  Prepare the Test Data
&lt;/h2&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;DATABASE&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;DATABASE&lt;/span&gt; &lt;span class="n"&gt;source2&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;DATABASE&lt;/span&gt; &lt;span class="n"&gt;source3&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="n"&gt;USE&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;TABLE&lt;/span&gt; &lt;span class="nv"&gt;`test`&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;
  &lt;span class="nv"&gt;`id`&lt;/span&gt; &lt;span class="nb"&gt;INT&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;11&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;NOT&lt;/span&gt; &lt;span class="k"&gt;NULL&lt;/span&gt; &lt;span class="n"&gt;AUTO_INCREMENT&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="nv"&gt;`name`&lt;/span&gt; &lt;span class="nb"&gt;VARCHAR&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;50&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="k"&gt;PRIMARY&lt;/span&gt; &lt;span class="k"&gt;KEY&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="nv"&gt;`id`&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;USING&lt;/span&gt; &lt;span class="n"&gt;BTREE&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="n"&gt;ENGINE&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;InnoDB&lt;/span&gt; &lt;span class="n"&gt;AUTO_INCREMENT&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s1"&gt;'张三'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;
&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;2&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s1"&gt;'李四'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;

&lt;span class="n"&gt;USE&lt;/span&gt; &lt;span class="n"&gt;source2&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;TABLE&lt;/span&gt; &lt;span class="nv"&gt;`test`&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;
  &lt;span class="nv"&gt;`id`&lt;/span&gt; &lt;span class="nb"&gt;INT&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;11&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;NOT&lt;/span&gt; &lt;span class="k"&gt;NULL&lt;/span&gt; &lt;span class="n"&gt;AUTO_INCREMENT&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="nv"&gt;`name`&lt;/span&gt; &lt;span class="nb"&gt;VARCHAR&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;50&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="k"&gt;PRIMARY&lt;/span&gt; &lt;span class="k"&gt;KEY&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="nv"&gt;`id`&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;USING&lt;/span&gt; &lt;span class="n"&gt;BTREE&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="n"&gt;ENGINE&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;InnoDB&lt;/span&gt; &lt;span class="n"&gt;AUTO_INCREMENT&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s1"&gt;'王五'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;
&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;2&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s1"&gt;'赵六'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;

&lt;span class="n"&gt;USE&lt;/span&gt; &lt;span class="n"&gt;source3&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;CREATE&lt;/span&gt; &lt;span class="k"&gt;TABLE&lt;/span&gt; &lt;span class="nv"&gt;`test`&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;
  &lt;span class="nv"&gt;`id`&lt;/span&gt; &lt;span class="nb"&gt;INT&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;11&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;NOT&lt;/span&gt; &lt;span class="k"&gt;NULL&lt;/span&gt; &lt;span class="n"&gt;AUTO_INCREMENT&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="nv"&gt;`name`&lt;/span&gt; &lt;span class="nb"&gt;VARCHAR&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;50&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="nv"&gt;`sources`&lt;/span&gt; &lt;span class="nb"&gt;VARCHAR&lt;/span&gt;&lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;50&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;
  &lt;span class="k"&gt;PRIMARY&lt;/span&gt; &lt;span class="k"&gt;KEY&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="nv"&gt;`id`&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt; &lt;span class="nv"&gt;`sources`&lt;/span&gt;&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="k"&gt;USING&lt;/span&gt; &lt;span class="n"&gt;BTREE&lt;/span&gt;
&lt;span class="p"&gt;)&lt;/span&gt; &lt;span class="n"&gt;ENGINE&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;InnoDB&lt;/span&gt; &lt;span class="n"&gt;AUTO_INCREMENT&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt; &lt;span class="nb"&gt;CHARACTER&lt;/span&gt; &lt;span class="k"&gt;SET&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="n"&gt;utf8mb4&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Prepare the SeaTunnel Job
&lt;/h2&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight hocon"&gt;&lt;code&gt;&lt;span class="nl"&gt;env&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;parallelism&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="mi"&gt;2&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;job.mode&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"STREAMING"&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="nl"&gt;checkpoint.interval&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="mi"&gt;20000&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="nl"&gt;source&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;

  &lt;/span&gt;&lt;span class="nl"&gt;MySQL-CDC&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;plugin_output&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"source_data1"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="k"&gt;url&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"jdbc:mysql://10.0.12.100:3306/source1"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;username&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;password&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;table-names&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"source1.test"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;startup.mode&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="l"&gt;initial&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

  &lt;/span&gt;&lt;span class="nl"&gt;MySQL-CDC&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;plugin_output&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"source_data2"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="k"&gt;url&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"jdbc:mysql://10.0.12.100:3306/source2"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;username&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;password&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;table-names&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"source2.test"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;startup.mode&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="l"&gt;initial&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="nl"&gt;transform&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;

  &lt;/span&gt;&lt;span class="nl"&gt;Sql&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;plugin_input&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"source_data1"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;plugin_output&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"result1"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;query&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"SELECT *, 'source1' AS sources FROM source_data1"&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

  &lt;/span&gt;&lt;span class="nl"&gt;Sql&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;plugin_input&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"source_data2"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;plugin_output&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"result2"&lt;/span&gt;&lt;span class="w"&gt;
    &lt;/span&gt;&lt;span class="nl"&gt;query&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"SELECT *, 'source2' AS sources FROM source_data2"&lt;/span&gt;&lt;span class="w"&gt;
  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="nl"&gt;sink&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;

  &lt;/span&gt;&lt;span class="nl"&gt;Jdbc&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;{&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;plugin_input&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"result1"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s2"&gt;"result2"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="k"&gt;url&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"jdbc:mysql://10.0.12.100:3306/source3"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;driver&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"com.mysql.cj.jdbc.Driver"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;username&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;password&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"root"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;database&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"source3"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;table&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="s2"&gt;"test"&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;generate_sink_sql&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="kc"&gt;true&lt;/span&gt;&lt;span class="w"&gt;

    &lt;/span&gt;&lt;span class="nl"&gt;primary_keys&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;=&lt;/span&gt;&lt;span class="w"&gt; &lt;/span&gt;&lt;span class="p"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"id"&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s2"&gt;"sources"&lt;/span&gt;&lt;span class="p"&gt;]&lt;/span&gt;&lt;span class="w"&gt;

  &lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;

&lt;/span&gt;&lt;span class="p"&gt;}&lt;/span&gt;&lt;span class="w"&gt;
&lt;/span&gt;&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;This example uses a single configuration to synchronize data from two databases. Alternatively, you can configure two separate synchronization jobs.&lt;/p&gt;

&lt;p&gt;Two source connectors are configured: &lt;code&gt;source_data1&lt;/code&gt; and &lt;code&gt;source_data2&lt;/code&gt;.&lt;/p&gt;

&lt;p&gt;Two SQL transform components are also configured to process the corresponding source connectors.&lt;/p&gt;

&lt;p&gt;Each SQL transform assigns a fixed value to the &lt;code&gt;sources&lt;/code&gt; field.&lt;/p&gt;

&lt;p&gt;The sink consumes the outputs from both SQL transforms and writes the data to the &lt;code&gt;test&lt;/code&gt; table. The composite primary key is configured to support update operations.&lt;/p&gt;

&lt;p&gt;The destination table must be created manually. Otherwise, the following error will be reported:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;BLOB/TEXT column 'sources' used in key specification without a key length
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Test
&lt;/h2&gt;

&lt;h3&gt;
  
  
  1. Start the Job
&lt;/h3&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;bin/seatunnel.sh &lt;span class="nt"&gt;--config&lt;/span&gt; job/mysql_mysql.conf &lt;span class="nt"&gt;-m&lt;/span&gt; &lt;span class="nb"&gt;local&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  2. Verify the Data
&lt;/h3&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="n"&gt;mysql&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="k"&gt;select&lt;/span&gt; &lt;span class="o"&gt;*&lt;/span&gt; &lt;span class="k"&gt;from&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;----+--------+---------+&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;name&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;sources&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;----+--------+---------+&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;张三&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;王五&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;李四&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;赵六&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;----+--------+---------+&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  3. Modify the Source Data
&lt;/h3&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="n"&gt;USE&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;INSERT&lt;/span&gt; &lt;span class="k"&gt;INTO&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt; &lt;span class="k"&gt;VALUES&lt;/span&gt; &lt;span class="p"&gt;(&lt;/span&gt;&lt;span class="mi"&gt;3&lt;/span&gt;&lt;span class="p"&gt;,&lt;/span&gt;&lt;span class="s1"&gt;'钱七'&lt;/span&gt;&lt;span class="p"&gt;);&lt;/span&gt;

&lt;span class="k"&gt;UPDATE&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt;
&lt;span class="k"&gt;SET&lt;/span&gt; &lt;span class="n"&gt;name&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="s1"&gt;'张三1'&lt;/span&gt;
&lt;span class="k"&gt;WHERE&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="n"&gt;USE&lt;/span&gt; &lt;span class="n"&gt;source2&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="k"&gt;DELETE&lt;/span&gt; &lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt;
&lt;span class="k"&gt;WHERE&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  4. Verify the Data Again
&lt;/h3&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight sql"&gt;&lt;code&gt;&lt;span class="n"&gt;mysql&lt;/span&gt;&lt;span class="o"&gt;&amp;gt;&lt;/span&gt; &lt;span class="k"&gt;SELECT&lt;/span&gt; &lt;span class="o"&gt;*&lt;/span&gt; &lt;span class="k"&gt;FROM&lt;/span&gt; &lt;span class="n"&gt;test&lt;/span&gt;&lt;span class="p"&gt;;&lt;/span&gt;

&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;----+---------+---------+&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;id&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;name&lt;/span&gt;    &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;sources&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;----+---------+---------+&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;张三&lt;/span&gt;&lt;span class="mi"&gt;1&lt;/span&gt;   &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;李四&lt;/span&gt;    &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;赵六&lt;/span&gt;    &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source2&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;|&lt;/span&gt;  &lt;span class="mi"&gt;3&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="err"&gt;钱七&lt;/span&gt;    &lt;span class="o"&gt;|&lt;/span&gt; &lt;span class="n"&gt;source1&lt;/span&gt; &lt;span class="o"&gt;|&lt;/span&gt;
&lt;span class="o"&gt;+&lt;/span&gt;&lt;span class="c1"&gt;----+---------+---------+&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Summary
&lt;/h2&gt;

&lt;p&gt;The steps above demonstrate the complete process of synchronizing data from tables in two databases into a single table in another database.&lt;/p&gt;

&lt;p&gt;I have seen other articles mentioning that the destination table can be created automatically, but I have not been able to reproduce this behavior during my testing.&lt;/p&gt;

&lt;p&gt;If you have successfully implemented automatic destination table creation, feel free to share your experience in the comments.&lt;/p&gt;

</description>
    </item>
    <item>
      <title>Deploy Apache SeaTunnel 2.3.11 with Docker: A Complete Guide to Syncing Kafka Data to Hive and Elasticsearch</title>
      <dc:creator>Apache SeaTunnel</dc:creator>
      <pubDate>Thu, 02 Jul 2026 03:06:23 +0000</pubDate>
      <link>https://dev.to/seatunnel/deploy-apache-seatunnel-2311-with-docker-a-complete-guide-to-syncing-kafka-data-to-hive-and-1kgi</link>
      <guid>https://dev.to/seatunnel/deploy-apache-seatunnel-2311-with-docker-a-complete-guide-to-syncing-kafka-data-to-hive-and-1kgi</guid>
      <description>&lt;p&gt;This guide walks you through the complete process of deploying Apache SeaTunnel 2.3.11 with Docker. It covers everything from environment setup and dependency installation to configuring Kafka virtual tables, data sources, and building end-to-end data synchronization pipelines from Kafka to Hive and Elasticsearch.&lt;/p&gt;

&lt;h2&gt;
  
  
  Prerequisites
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Project Directory Structure
&lt;/h3&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;seatunnel-docker/
├── docker-compose.yml              # Main Docker Compose configuration
├── hive/                           # Hive configuration
│   ├── hive-site.xml
│   └── lib/                        # Required dependency JARs
│       └── postgresql-42.5.1.jar
├── init-sql/                       # Database initialization scripts
│   └── seatunnel_server_mysql.sql
├── seatunnel/                      # SeaTunnel server configuration
│   ├── Dockerfile
│   └── apache-seatunnel-2.3.11/    # Extracted SeaTunnel binary package
│       └── lib/                    # Required dependency JARs
│           ├── hive-exec-3.1.3.jar
│           ├── hive-metastore-3.1.3.jar
│           ├── libfb303-0.9.3.jar
│           ├── mysql-connector-java-8.0.28.jar
│           └── seatunnel-hadoop3-3.1.4-uber.jar
└── seatunnel-web/                  # SeaTunnel Web configuration
    ├── Dockerfile
    └── apache-seatunnel-web-1.0.3-bin/  # Extracted SeaTunnel Web package
        └── libs/                   # Required dependency JARs
            └── mysql-connector-java-8.0.28.jar
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Download Apache SeaTunnel
&lt;/h2&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="c"&gt;# Apache SeaTunnel 2.3.11&lt;/span&gt;
https://dlcdn.apache.org/seatunnel/2.3.11/apache-seatunnel-2.3.11-bin.tar.gz

&lt;span class="c"&gt;# Build SeaTunnel Web 1.0.3 from source&lt;/span&gt;
git clone https://github.com/apache/seatunnel-web.git
&lt;span class="nb"&gt;cd &lt;/span&gt;seatunnel-web
sh build.sh code
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Download Required Dependencies
&lt;/h2&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="c"&gt;# Required by the Hive Metastore container (PostgreSQL is used as the Hive metastore database)&lt;/span&gt;
https://jdbc.postgresql.org/download/postgresql-42.5.1.jar

&lt;span class="c"&gt;# Additional dependencies required for Hive synchronization&lt;/span&gt;
&lt;span class="c"&gt;# (In practice, only the first three JARs are required)&lt;/span&gt;
https://repo1.maven.org/maven2/org/apache/hive/hive-exec/3.1.3/hive-exec-3.1.3.jar
https://repo1.maven.org/maven2/org/apache/hive/hive-metastore/3.1.3/hive-metastore-3.1.3.jar
https://repo.maven.apache.org/maven2/org/apache/thrift/libfb303/0.9.3/libfb303-0.9.3.jar
https://repo1.maven.org/maven2/org/apache/thrift/libthrift/0.12.0/libthrift-0.12.0.jar
https://repo1.maven.org/maven2/org/apache/hive/hive-common/3.1.3/hive-common-3.1.3.jar
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Create the Project Directory
&lt;/h2&gt;

&lt;p&gt;Place all downloaded packages and configuration files into the &lt;strong&gt;seatunnel-docker&lt;/strong&gt; directory.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;mkdir &lt;/span&gt;seatunnel-docker
&lt;span class="nb"&gt;cd &lt;/span&gt;seatunnel-docker
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Deploy with Docker
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Configure &lt;code&gt;docker-compose.yml&lt;/code&gt;
&lt;/h3&gt;



&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;version&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s1"&gt;'&lt;/span&gt;&lt;span class="s"&gt;3.9'&lt;/span&gt;

&lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;driver&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;bridge&lt;/span&gt;
    &lt;span class="na"&gt;ipam&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;config&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="na"&gt;subnet&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.0/24&lt;/span&gt;

&lt;span class="na"&gt;services&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="c1"&gt;# ===== Hive Services =====&lt;/span&gt;
  &lt;span class="na"&gt;hive-metastore-db&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;postgres:15&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive-metastore-db&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive-metastore-db&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;POSTGRES_DB&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;metastore_db&lt;/span&gt;
      &lt;span class="na"&gt;POSTGRES_USER&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive&lt;/span&gt;
      &lt;span class="na"&gt;POSTGRES_PASSWORD&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive123456&lt;/span&gt;
    &lt;span class="na"&gt;ports&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;5432:5432"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-metastore-db-data:/var/lib/postgresql/data&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.2&lt;/span&gt;
    &lt;span class="na"&gt;healthcheck&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;  &lt;span class="c1"&gt;# Health check&lt;/span&gt;
      &lt;span class="na"&gt;test&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="pi"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;CMD-SHELL"&lt;/span&gt;&lt;span class="pi"&gt;,&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;pg_isready&lt;/span&gt;&lt;span class="nv"&gt; &lt;/span&gt;&lt;span class="s"&gt;-U&lt;/span&gt;&lt;span class="nv"&gt; &lt;/span&gt;&lt;span class="s"&gt;hive&lt;/span&gt;&lt;span class="nv"&gt; &lt;/span&gt;&lt;span class="s"&gt;-d&lt;/span&gt;&lt;span class="nv"&gt; &lt;/span&gt;&lt;span class="s"&gt;metastore_db"&lt;/span&gt;&lt;span class="pi"&gt;]&lt;/span&gt;
      &lt;span class="na"&gt;interval&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;5s&lt;/span&gt;
      &lt;span class="na"&gt;timeout&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;5s&lt;/span&gt;
      &lt;span class="na"&gt;retries&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;10&lt;/span&gt;
      &lt;span class="na"&gt;start_period&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;10s&lt;/span&gt;

  &lt;span class="na"&gt;hive-metastore&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;apache/hive:4.0.0&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive-metastore&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive-metastore&lt;/span&gt;
    &lt;span class="na"&gt;depends_on&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;hive-metastore-db&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;condition&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;service_healthy&lt;/span&gt;  &lt;span class="c1"&gt;# Wait until the database is healthy&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;SERVICE_NAME&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;metastore&lt;/span&gt;
      &lt;span class="na"&gt;DB_DRIVER&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;postgres&lt;/span&gt;
      &lt;span class="na"&gt;SERVICE_OPTS&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="pi"&gt;&amp;gt;-&lt;/span&gt;
        &lt;span class="s"&gt;-Djavax.jdo.option.ConnectionDriverName=org.postgresql.Driver&lt;/span&gt;
        &lt;span class="s"&gt;-Djavax.jdo.option.ConnectionURL=jdbc:postgresql://hive-metastore-db:5432/metastore_db&lt;/span&gt;
        &lt;span class="s"&gt;-Djavax.jdo.option.ConnectionUserName=hive&lt;/span&gt;
        &lt;span class="s"&gt;-Djavax.jdo.option.ConnectionPassword=hive123456&lt;/span&gt;
    &lt;span class="na"&gt;ports&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;9083:9083"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive/lib/postgresql-42.5.1.jar:/opt/hive/lib/postgresql-42.5.1.jar&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive/hive-site.xml:/opt/hive/conf/hive-site.xml&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.3&lt;/span&gt;

  &lt;span class="na"&gt;hive-server2&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;apache/hive:4.0.0&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive-server2&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hive-server2&lt;/span&gt;
    &lt;span class="na"&gt;depends_on&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;hive-metastore&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;HIVE_SERVER2_THRIFT_PORT&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;10000&lt;/span&gt;
      &lt;span class="na"&gt;SERVICE_NAME&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;hiveserver2&lt;/span&gt;
      &lt;span class="na"&gt;IS_RESUME&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;true"&lt;/span&gt;
      &lt;span class="na"&gt;SERVICE_OPTS&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;-Dhive.metastore.uris=thrift://hive-metastore:9083"&lt;/span&gt;
    &lt;span class="na"&gt;ports&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;10000:10000"&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;10002:10002"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.4&lt;/span&gt;

  &lt;span class="c1"&gt;# ===== MySQL =====&lt;/span&gt;
  &lt;span class="na"&gt;mysql-seatunnel&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;mysql:8.0.42&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;mysql-seatunnel&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;mysql-seatunnel&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;MYSQL_ROOT_PASSWORD&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;root123456&lt;/span&gt;
      &lt;span class="na"&gt;MYSQL_DATABASE&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel&lt;/span&gt;
      &lt;span class="na"&gt;MYSQL_ROOT_HOST&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s1"&gt;'&lt;/span&gt;&lt;span class="s"&gt;%'&lt;/span&gt;
    &lt;span class="na"&gt;ports&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;3806:3306"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./mysql_data:/var/lib/mysql&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./init-sql:/docker-entrypoint-initdb.d&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.5&lt;/span&gt;
    &lt;span class="na"&gt;command&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;--default-authentication-plugin=mysql_native_password&lt;/span&gt;
    &lt;span class="na"&gt;healthcheck&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;test&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="pi"&gt;[&lt;/span&gt;&lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;CMD"&lt;/span&gt;&lt;span class="pi"&gt;,&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;mysqladmin"&lt;/span&gt;&lt;span class="pi"&gt;,&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;ping"&lt;/span&gt;&lt;span class="pi"&gt;,&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;-h"&lt;/span&gt;&lt;span class="pi"&gt;,&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;localhost"&lt;/span&gt;&lt;span class="pi"&gt;]&lt;/span&gt;
      &lt;span class="na"&gt;interval&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;10s&lt;/span&gt;
      &lt;span class="na"&gt;timeout&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;5s&lt;/span&gt;
      &lt;span class="na"&gt;retries&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;5&lt;/span&gt;

  &lt;span class="c1"&gt;# ===== SeaTunnel Services =====&lt;/span&gt;
  &lt;span class="na"&gt;seatunnel-master&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;build&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;context&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;./seatunnel&lt;/span&gt;
      &lt;span class="na"&gt;dockerfile&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;Dockerfile&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel:2.3.11&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master&lt;/span&gt;
    &lt;span class="na"&gt;extra_hosts&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore:172.16.0.3"&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore-db:172.16.0.2"&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;SEATUNNEL_HOME=/opt/seatunnel&lt;/span&gt;
    &lt;span class="na"&gt;command&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="pi"&gt;&amp;gt;&lt;/span&gt;
      &lt;span class="s"&gt;sh -c "&lt;/span&gt;
      &lt;span class="s"&gt;cd /opt/seatunnel &amp;amp;&amp;amp;&lt;/span&gt;
      &lt;span class="s"&gt;exec bin/seatunnel-cluster.sh -r master&lt;/span&gt;
      &lt;span class="s"&gt;"&lt;/span&gt;
    &lt;span class="na"&gt;ports&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;5801:5801"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./seatunnel/apache-seatunnel-2.3.11/:/opt/seatunnel/&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./logs/master:/opt/seatunnel/logs&lt;/span&gt;
      &lt;span class="c1"&gt;# Mount the Hive warehouse directory to ensure data is persisted on the host&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.10&lt;/span&gt;

  &lt;span class="na"&gt;seatunnel-worker1&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel:2.3.11&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker1&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker1&lt;/span&gt;
    &lt;span class="na"&gt;extra_hosts&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore:172.16.0.3"&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore-db:172.16.0.2"&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;SEATUNNEL_HOME=/opt/seatunnel&lt;/span&gt;
    &lt;span class="na"&gt;command&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="pi"&gt;&amp;gt;&lt;/span&gt;
      &lt;span class="s"&gt;sh -c "&lt;/span&gt;
      &lt;span class="s"&gt;cd /opt/seatunnel &amp;amp;&amp;amp;&lt;/span&gt;
      &lt;span class="s"&gt;exec bin/seatunnel-cluster.sh -r worker&lt;/span&gt;
      &lt;span class="s"&gt;"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./seatunnel/apache-seatunnel-2.3.11/:/opt/seatunnel/&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./logs/worker1:/opt/seatunnel/logs&lt;/span&gt;
      &lt;span class="c1"&gt;# Mount the Hive warehouse directory to ensure data is persisted on the host&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
    &lt;span class="na"&gt;depends_on&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.11&lt;/span&gt;

  &lt;span class="na"&gt;seatunnel-worker2&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel:2.3.11&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker2&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker2&lt;/span&gt;
    &lt;span class="na"&gt;extra_hosts&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore:172.16.0.3"&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore-db:172.16.0.2"&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;SEATUNNEL_HOME=/opt/seatunnel&lt;/span&gt;
    &lt;span class="na"&gt;command&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="pi"&gt;&amp;gt;&lt;/span&gt;
      &lt;span class="s"&gt;sh -c "&lt;/span&gt;
      &lt;span class="s"&gt;cd /opt/seatunnel &amp;amp;&amp;amp;&lt;/span&gt;
      &lt;span class="s"&gt;exec bin/seatunnel-cluster.sh -r worker&lt;/span&gt;
      &lt;span class="s"&gt;"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./seatunnel/apache-seatunnel-2.3.11/:/opt/seatunnel/&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./logs/worker2:/opt/seatunnel/logs&lt;/span&gt;
      &lt;span class="c1"&gt;# Mount the Hive warehouse directory to ensure data is persisted on the host&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
    &lt;span class="na"&gt;depends_on&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.12&lt;/span&gt;

  &lt;span class="na"&gt;seatunnel-web&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;build&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;context&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;./seatunnel-web&lt;/span&gt;
      &lt;span class="na"&gt;dockerfile&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;Dockerfile&lt;/span&gt;
    &lt;span class="na"&gt;image&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-web:1.0.3&lt;/span&gt;
    &lt;span class="na"&gt;container_name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-web&lt;/span&gt;
    &lt;span class="na"&gt;hostname&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel-web&lt;/span&gt;
    &lt;span class="na"&gt;extra_hosts&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore:172.16.0.3"&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore-db:172.16.0.2"&lt;/span&gt;
    &lt;span class="na"&gt;environment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;SEATUNNEL_HOME=/opt/seatunnel&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;SEATUNNEL_WEB_HOME=/opt/seatunnel-web&lt;/span&gt;
    &lt;span class="na"&gt;ports&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;8801:8801"&lt;/span&gt;
    &lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./seatunnel/apache-seatunnel-2.3.11/:/opt/seatunnel/&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./seatunnel-web/apache-seatunnel-web-1.0.3-bin/:/opt/seatunnel-web/&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./logs/web:/opt/seatunnel-web/logs&lt;/span&gt;
      &lt;span class="c1"&gt;# Mount the Hive warehouse directory to keep the runtime environment consistent&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
    &lt;span class="na"&gt;depends_on&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master&lt;/span&gt;
    &lt;span class="na"&gt;networks&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;seatunnel-network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;ipv4_address&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;172.16.0.13&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  SeaTunnel Configuration
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Dockerfile
&lt;/h3&gt;

&lt;p&gt;Create the following &lt;code&gt;Dockerfile&lt;/code&gt; for the SeaTunnel service:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight docker"&gt;&lt;code&gt;&lt;span class="k"&gt;FROM&lt;/span&gt;&lt;span class="s"&gt; eclipse-temurin:8-jdk-ubi9-minimal&lt;/span&gt;

&lt;span class="k"&gt;WORKDIR&lt;/span&gt;&lt;span class="s"&gt; /opt/seatunnel/&lt;/span&gt;

&lt;span class="c"&gt;# Environment variables&lt;/span&gt;
&lt;span class="k"&gt;ENV&lt;/span&gt;&lt;span class="s"&gt; SEATUNNEL_HOME=/opt/seatunnel&lt;/span&gt;
&lt;span class="k"&gt;ENV&lt;/span&gt;&lt;span class="s"&gt; PATH=$PATH:$SEATUNNEL_HOME/bin&lt;/span&gt;

&lt;span class="c"&gt;# Expose the cluster communication port&lt;/span&gt;
&lt;span class="k"&gt;EXPOSE&lt;/span&gt;&lt;span class="s"&gt; 5801&lt;/span&gt;

&lt;span class="c"&gt;# Startup command&lt;/span&gt;
&lt;span class="k"&gt;CMD&lt;/span&gt;&lt;span class="s"&gt; ["sh", "bin/seatunnel-cluster.sh", "-r", "master"]&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Configure &lt;code&gt;hazelcast-client.yaml&lt;/code&gt;
&lt;/h3&gt;

&lt;p&gt;Edit:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;seatunnel/apache-seatunnel-2.3.11/config/hazelcast-client.yaml
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Configure the SeaTunnel client to connect to the cluster:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;hazelcast-client&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="na"&gt;cluster-name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel&lt;/span&gt;
  &lt;span class="na"&gt;properties&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.logging.type&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;log4j2&lt;/span&gt;
  &lt;span class="na"&gt;connection-strategy&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;connection-retry&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;cluster-connect-timeout-millis&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;3000&lt;/span&gt;
  &lt;span class="na"&gt;network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;cluster-members&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master:5801&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Configure &lt;code&gt;hazelcast-master.yaml&lt;/code&gt;
&lt;/h3&gt;

&lt;p&gt;Edit:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;seatunnel/apache-seatunnel-2.3.11/config/hazelcast-master.yaml
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Configure the master node:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;hazelcast&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="na"&gt;cluster-name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel&lt;/span&gt;
  &lt;span class="na"&gt;network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;rest-api&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;enabled&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;false&lt;/span&gt;
      &lt;span class="na"&gt;endpoint-groups&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;CLUSTER_WRITE&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
          &lt;span class="na"&gt;enabled&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;true&lt;/span&gt;
        &lt;span class="na"&gt;DATA&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
          &lt;span class="na"&gt;enabled&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;true&lt;/span&gt;
    &lt;span class="na"&gt;join&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;tcp-ip&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;enabled&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;true&lt;/span&gt;
        &lt;span class="na"&gt;member-list&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
          &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master:5801&lt;/span&gt;
          &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker1:5802&lt;/span&gt;
          &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker2:5802&lt;/span&gt;
    &lt;span class="na"&gt;port&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;auto-increment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;false&lt;/span&gt;
      &lt;span class="na"&gt;port&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;5801&lt;/span&gt;
  &lt;span class="na"&gt;properties&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.invocation.max.retry.count&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;20&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.tcp.join.port.try.count&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;30&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.logging.type&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;log4j2&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.operation.generic.thread.count&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;50&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.failuredetector.type&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;phi-accrual&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.interval.seconds&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;2&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.max.no.heartbeat.seconds&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;180&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.phiaccrual.failuredetector.threshold&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;10&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.phiaccrual.failuredetector.sample.size&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;200&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.phiaccrual.failuredetector.min.std.dev.millis&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;100&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Configure &lt;code&gt;hazelcast-worker.yaml&lt;/code&gt;
&lt;/h3&gt;

&lt;p&gt;Edit:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;seatunnel/apache-seatunnel-2.3.11/config/hazelcast-worker.yaml
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Configure each worker node:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;hazelcast&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="na"&gt;cluster-name&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;seatunnel&lt;/span&gt;
  &lt;span class="na"&gt;network&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;join&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;tcp-ip&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
        &lt;span class="na"&gt;enabled&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;true&lt;/span&gt;
        &lt;span class="na"&gt;member-list&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
          &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-master:5801&lt;/span&gt;
          &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker1:5802&lt;/span&gt;
          &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;seatunnel-worker2:5802&lt;/span&gt;
    &lt;span class="na"&gt;port&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
      &lt;span class="na"&gt;auto-increment&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="kc"&gt;false&lt;/span&gt;
      &lt;span class="na"&gt;port&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;5802&lt;/span&gt;
  &lt;span class="na"&gt;properties&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.invocation.max.retry.count&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;20&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.tcp.join.port.try.count&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;30&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.logging.type&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;log4j2&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.operation.generic.thread.count&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;50&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.failuredetector.type&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="s"&gt;phi-accrual&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.interval.seconds&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;2&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.max.no.heartbeat.seconds&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;180&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.phiaccrual.failuredetector.threshold&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;10&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.phiaccrual.failuredetector.sample.size&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;200&lt;/span&gt;
    &lt;span class="na"&gt;hazelcast.heartbeat.phiaccrual.failuredetector.min.std.dev.millis&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt; &lt;span class="m"&gt;100&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Install Connector Plugins
&lt;/h3&gt;

&lt;p&gt;If no options appear in the &lt;strong&gt;Source&lt;/strong&gt; component when creating a synchronization job, the required connector plugins have not been installed.&lt;/p&gt;

&lt;p&gt;Run the following command to install all supported connector plugins:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;cd &lt;/span&gt;seatunnel/apache-seatunnel-2.3.11/
sh bin/install-plugin.sh
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Hive Configuration
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Configure &lt;code&gt;hive-site.xml&lt;/code&gt;
&lt;/h3&gt;

&lt;p&gt;Edit:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;hive/hive-site.xml
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Update the configuration as follows:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight xml"&gt;&lt;code&gt;&lt;span class="cp"&gt;&amp;lt;?xml version="1.0" encoding="UTF-8"?&amp;gt;&lt;/span&gt;
&lt;span class="nt"&gt;&amp;lt;configuration&amp;gt;&lt;/span&gt;
    &lt;span class="nt"&gt;&amp;lt;property&amp;gt;&lt;/span&gt;
        &lt;span class="nt"&gt;&amp;lt;name&amp;gt;&lt;/span&gt;hive.metastore.uris&lt;span class="nt"&gt;&amp;lt;/name&amp;gt;&lt;/span&gt;
        &lt;span class="nt"&gt;&amp;lt;value&amp;gt;&lt;/span&gt;thrift://hive-metastore:9083&lt;span class="nt"&gt;&amp;lt;/value&amp;gt;&lt;/span&gt;
    &lt;span class="nt"&gt;&amp;lt;/property&amp;gt;&lt;/span&gt;

    &lt;span class="nt"&gt;&amp;lt;property&amp;gt;&lt;/span&gt;
        &lt;span class="nt"&gt;&amp;lt;name&amp;gt;&lt;/span&gt;hive.metastore.warehouse.dir&lt;span class="nt"&gt;&amp;lt;/name&amp;gt;&lt;/span&gt;
        &lt;span class="nt"&gt;&amp;lt;value&amp;gt;&lt;/span&gt;/opt/hive/data/warehouse&lt;span class="nt"&gt;&amp;lt;/value&amp;gt;&lt;/span&gt;
    &lt;span class="nt"&gt;&amp;lt;/property&amp;gt;&lt;/span&gt;

    &lt;span class="nt"&gt;&amp;lt;property&amp;gt;&lt;/span&gt;
        &lt;span class="nt"&gt;&amp;lt;name&amp;gt;&lt;/span&gt;metastore.metastore.event.db.notification.api.auth&lt;span class="nt"&gt;&amp;lt;/name&amp;gt;&lt;/span&gt;
        &lt;span class="nt"&gt;&amp;lt;value&amp;gt;&lt;/span&gt;false&lt;span class="nt"&gt;&amp;lt;/value&amp;gt;&lt;/span&gt;
    &lt;span class="nt"&gt;&amp;lt;/property&amp;gt;&lt;/span&gt;
&lt;span class="nt"&gt;&amp;lt;/configuration&amp;gt;&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Add Required Dependencies
&lt;/h3&gt;

&lt;p&gt;Place the following JDBC driver in the &lt;code&gt;hive/lib&lt;/code&gt; directory.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;postgresql-42.5.1.jar
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  MySQL Configuration
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Initialize the Database
&lt;/h3&gt;

&lt;p&gt;Copy the initialization SQL script from the SeaTunnel Web package into the &lt;code&gt;init-sql&lt;/code&gt; directory.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="nb"&gt;cd &lt;/span&gt;seatunnel-docker

&lt;span class="nb"&gt;cp &lt;/span&gt;seatunnel-web/apache-seatunnel-web-1.0.3-bin/script/seatunnel_server_mysql.sql &lt;span class="se"&gt;\&lt;/span&gt;
init-sql/seatunnel_server_mysql.sql
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Start the Docker Environment
&lt;/h2&gt;

&lt;p&gt;Build and start all services:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="c"&gt;# Build and start all services&lt;/span&gt;
docker compose up &lt;span class="nt"&gt;-d&lt;/span&gt; &lt;span class="nt"&gt;--build&lt;/span&gt;

&lt;span class="c"&gt;# Open the SeaTunnel Web UI&lt;/span&gt;
&lt;span class="c"&gt;# Default credentials:&lt;/span&gt;
&lt;span class="c"&gt;# Username: admin&lt;/span&gt;
&lt;span class="c"&gt;# Password: admin&lt;/span&gt;
open http://localhost:8801
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Running Example
&lt;/h2&gt;

&lt;p&gt;After all services have started successfully, log in to the SeaTunnel Web UI and complete the following configuration steps.&lt;/p&gt;

&lt;h2&gt;
  
  
  Configure the Display Language
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Login Page
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F49rwkx9a3ymslwcm99gz.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2F49rwkx9a3ymslwcm99gz.jpg" width="800" height="1154"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Open Settings
&lt;/h3&gt;

&lt;p&gt;&lt;a href="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fgceclj4btjq9kecqlyo5.jpg" class="article-body-image-wrapper"&gt;&lt;img src="https://media2.dev.to/dynamic/image/width=800%2Cheight=%2Cfit=scale-down%2Cgravity=auto%2Cformat=auto/https%3A%2F%2Fdev-to-uploads.s3.us-east-2.amazonaws.com%2Fuploads%2Farticles%2Fgceclj4btjq9kecqlyo5.jpg" width="290" height="318"&gt;&lt;/a&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Change the Language
&lt;/h3&gt;

&lt;p&gt;Select your preferred language from the language settings.&lt;/p&gt;

&lt;h2&gt;
  
  
  Configure Data Sources
&lt;/h2&gt;

&lt;p&gt;Before creating synchronization jobs, configure the required data sources.&lt;/p&gt;

&lt;h3&gt;
  
  
  Configure a Kafka Data Source
&lt;/h3&gt;

&lt;p&gt;Create a Kafka connection by providing the cluster address and connection parameters.&lt;/p&gt;

&lt;p&gt;&lt;em&gt;(Insert screenshot)&lt;/em&gt;&lt;/p&gt;

&lt;h3&gt;
  
  
  Configure an Elasticsearch Data Source
&lt;/h3&gt;

&lt;p&gt;Configure your Elasticsearch cluster information, including the endpoint and authentication credentials if required.&lt;/p&gt;

&lt;h3&gt;
  
  
  Configure a Hive Metastore Local Data Source
&lt;/h3&gt;

&lt;p&gt;You can configure the Hive Metastore endpoint using:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;thrift://hive-metastore:9083
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Configure Virtual Tables
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Create a Virtual Table
&lt;/h3&gt;

&lt;p&gt;Follow these steps to create a virtual table:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;Navigate to &lt;strong&gt;Virtual Tables&lt;/strong&gt;.&lt;/li&gt;
&lt;li&gt;Click &lt;strong&gt;Create&lt;/strong&gt;.&lt;/li&gt;
&lt;li&gt;Select an existing data source.&lt;/li&gt;
&lt;li&gt;Configure the virtual table properties.&lt;/li&gt;
&lt;li&gt;Click &lt;strong&gt;Next&lt;/strong&gt; to define field mappings.&lt;/li&gt;
&lt;li&gt;Review the configuration.&lt;/li&gt;
&lt;li&gt;Save the virtual table.&lt;/li&gt;
&lt;/ol&gt;

&lt;h2&gt;
  
  
  Create Synchronization Jobs
&lt;/h2&gt;

&lt;p&gt;Once the data sources and virtual tables are ready, you can build synchronization pipelines using the visual designer.&lt;/p&gt;

&lt;h3&gt;
  
  
  Kafka → Hive Synchronization
&lt;/h3&gt;

&lt;h3&gt;
  
  
  Configure the Job Components
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;Source&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Configure the Kafka source by selecting the previously created Kafka data source.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Field Mapper&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Open the &lt;strong&gt;Model&lt;/strong&gt; view to define field mappings between the source and destination schemas.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Sink&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Configure Hive as the destination and specify the target database and table.&lt;/p&gt;

&lt;h3&gt;
  
  
  Kafka → Elasticsearch Synchronization
&lt;/h3&gt;

&lt;h3&gt;
  
  
  Configure the Job Components
&lt;/h3&gt;

&lt;ul&gt;
&lt;li&gt;Source&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Configure the Kafka source.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Field Mapper&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Configure field mappings in the &lt;strong&gt;Model&lt;/strong&gt; view.&lt;/p&gt;

&lt;ul&gt;
&lt;li&gt;Sink&lt;/li&gt;
&lt;/ul&gt;

&lt;p&gt;Configure Elasticsearch as the destination.&lt;/p&gt;

&lt;p&gt;Specify the target index and any required connection parameters.&lt;/p&gt;

&lt;h3&gt;
  
  
  General Workflow for Creating Synchronization Jobs
&lt;/h3&gt;

&lt;p&gt;To create a synchronization job in SeaTunnel Web:&lt;/p&gt;

&lt;ol&gt;
&lt;li&gt;Navigate to &lt;strong&gt;Jobs&lt;/strong&gt; → &lt;strong&gt;Synchronization Job Definitions&lt;/strong&gt;.&lt;/li&gt;
&lt;li&gt;Click &lt;strong&gt;Create&lt;/strong&gt;.&lt;/li&gt;
&lt;li&gt;Drag or select the &lt;strong&gt;Source&lt;/strong&gt;, &lt;strong&gt;Field Mapper&lt;/strong&gt;, and &lt;strong&gt;Sink&lt;/strong&gt; components to build the pipeline.&lt;/li&gt;
&lt;li&gt;Double-click the &lt;strong&gt;Source&lt;/strong&gt; component and select the configured Kafka data source.&lt;/li&gt;
&lt;li&gt;Double-click &lt;strong&gt;Field Mapper&lt;/strong&gt;, then open the &lt;strong&gt;Model&lt;/strong&gt; view to configure field mappings.&lt;/li&gt;
&lt;li&gt;Double-click the &lt;strong&gt;Sink&lt;/strong&gt; component and configure Hive or Elasticsearch as the destination.&lt;/li&gt;
&lt;li&gt;Save the job.&lt;/li&gt;
&lt;li&gt;Start the synchronization job.&lt;/li&gt;
&lt;/ol&gt;

&lt;blockquote&gt;
&lt;p&gt;&lt;strong&gt;Important&lt;/strong&gt;&lt;/p&gt;

&lt;p&gt;Before saving the job, make sure to configure the &lt;strong&gt;Job Mode&lt;/strong&gt;.&lt;/p&gt;

&lt;p&gt;Otherwise, the job cannot be saved and the following error will be displayed:&lt;/p&gt;


&lt;pre class="highlight plaintext"&gt;&lt;code&gt;job env can't be empty, please change config
&lt;/code&gt;&lt;/pre&gt;

&lt;/blockquote&gt;

&lt;h2&gt;
  
  
  Hive Operations
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Create a Table
&lt;/h3&gt;

&lt;p&gt;Use one of the following commands to create a table in Hive.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;&lt;span class="c"&gt;# Access the HiveServer2 container&lt;/span&gt;
docker &lt;span class="nb"&gt;exec&lt;/span&gt; &lt;span class="nt"&gt;-it&lt;/span&gt; hive-server2 beeline &lt;span class="nt"&gt;-u&lt;/span&gt; jdbc:hive2://localhost:10000 &lt;span class="nt"&gt;-e&lt;/span&gt; &lt;span class="s2"&gt;"
CREATE TABLE IF NOT EXISTS default.test_user_data3 (
user_id STRING,
type STRING,
content STRING
)
ROW FORMAT DELIMITED
FIELDS TERMINATED BY '&lt;/span&gt;&lt;span class="se"&gt;\t&lt;/span&gt;&lt;span class="s2"&gt;'
STORED AS TEXTFILE;
"&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Alternatively, you can create the table in Parquet format, which is recommended for better storage efficiency and query performance.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;docker &lt;span class="nb"&gt;exec&lt;/span&gt; &lt;span class="nt"&gt;-it&lt;/span&gt; hive-server2 beeline &lt;span class="nt"&gt;-u&lt;/span&gt; jdbc:hive2://localhost:10000 &lt;span class="nt"&gt;-e&lt;/span&gt; &lt;span class="s2"&gt;"
CREATE TABLE IF NOT EXISTS default.test_user_data3 (
user_id STRING,
type STRING,
content STRING
)
STORED AS PARQUET;
"&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  View the Table Schema
&lt;/h3&gt;

&lt;p&gt;Run the following command to verify that the table has been created successfully.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;docker &lt;span class="nb"&gt;exec&lt;/span&gt; &lt;span class="nt"&gt;-it&lt;/span&gt; hive-server2 beeline &lt;span class="nt"&gt;-u&lt;/span&gt; jdbc:hive2://localhost:10000 &lt;span class="nt"&gt;-e&lt;/span&gt; &lt;span class="s2"&gt;"
SHOW TABLES IN default;
DESCRIBE default.test_user_data3;
"&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Query Table Data
&lt;/h3&gt;

&lt;p&gt;Run the following command to query the synchronized data stored in Hive.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;docker &lt;span class="nb"&gt;exec&lt;/span&gt; &lt;span class="nt"&gt;-it&lt;/span&gt; hive-server2 beeline &lt;span class="nt"&gt;-u&lt;/span&gt; jdbc:hive2://localhost:10000 &lt;span class="nt"&gt;-e&lt;/span&gt; &lt;span class="s2"&gt;"
SELECT * FROM default.test_user_data3 LIMIT 10;
"&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h2&gt;
  
  
  Troubleshooting
&lt;/h2&gt;

&lt;h3&gt;
  
  
  Hive Metastore URI Parsing Error
&lt;/h3&gt;

&lt;p&gt;If the following exception is reported:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;seatunnel seatunnel-web ERROR &lt;span class="o"&gt;[&lt;/span&gt;qtp2135089262-20] &lt;span class="o"&gt;[&lt;/span&gt;MetaStoreUtils.logAndThrowMetaException&lt;span class="o"&gt;()&lt;/span&gt;:166] - Got exception: java.net.URISyntaxException Illegal character &lt;span class="k"&gt;in &lt;/span&gt;&lt;span class="nb"&gt;hostname &lt;/span&gt;at index 44: thrift://hive-metastore.seatunnel-docker_seatunnel-network:9083
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Add static hostname mappings to the corresponding services in &lt;code&gt;docker-compose.yml&lt;/code&gt;.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;extra_hosts&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
   &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore:172.16.0.3"&lt;/span&gt;
   &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s2"&gt;"&lt;/span&gt;&lt;span class="s"&gt;hive-metastore-db:172.16.0.2"&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Hive Synchronization Fails with &lt;code&gt;java.lang.NoClassDefFoundError&lt;/code&gt;
&lt;/h3&gt;

&lt;p&gt;If a Hive synchronization job fails with &lt;code&gt;java.lang.NoClassDefFoundError&lt;/code&gt;, ensure that the required dependency JARs are available in:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;seatunnel/apache-seatunnel-2.3.11/lib
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Required dependencies:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;hive-exec-3.1.3.jar
hive-metastore-3.1.3.jar
libfb303-0.9.3.jar
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Hive Synchronization Job Completes Successfully but No Data Is Written
&lt;/h3&gt;

&lt;p&gt;If the synchronization job completes successfully but no data is written to Hive, verify that the Hive warehouse directory is mounted correctly in &lt;code&gt;docker-compose.yml&lt;/code&gt;.&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight yaml"&gt;&lt;code&gt;&lt;span class="na"&gt;volumes&lt;/span&gt;&lt;span class="pi"&gt;:&lt;/span&gt;
  &lt;span class="c1"&gt;# Mount the Hive warehouse directory to ensure data is persisted on the host&lt;/span&gt;
  &lt;span class="pi"&gt;-&lt;/span&gt; &lt;span class="s"&gt;./hive-warehouse:/opt/hive/data/warehouse&lt;/span&gt;
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;h3&gt;
  
  
  Check Which Worker Executes the Job
&lt;/h3&gt;

&lt;p&gt;You can identify which worker node is executing the synchronization job by reviewing the master log:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight plaintext"&gt;&lt;code&gt;./logs/master/seatunnel-engine-master.log
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



&lt;p&gt;Example:&lt;br&gt;
&lt;/p&gt;

&lt;div class="highlight js-code-highlight"&gt;
&lt;pre class="highlight shell"&gt;&lt;code&gt;Task &lt;span class="o"&gt;[&lt;/span&gt;TaskGroupLocation&lt;span class="o"&gt;{&lt;/span&gt;&lt;span class="nv"&gt;jobId&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;1080750681855361026, &lt;span class="nv"&gt;pipelineId&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;1, &lt;span class="nv"&gt;taskGroupId&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;2&lt;span class="o"&gt;}]&lt;/span&gt; will be executed on worker &lt;span class="o"&gt;[[&lt;/span&gt;seatunnel-worker2]:5801], slotID &lt;span class="o"&gt;[&lt;/span&gt;2], resourceProfile &lt;span class="o"&gt;[&lt;/span&gt;ResourceProfile&lt;span class="o"&gt;{&lt;/span&gt;&lt;span class="nv"&gt;cpu&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;CPU&lt;span class="o"&gt;{&lt;/span&gt;&lt;span class="nv"&gt;core&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;0&lt;span class="o"&gt;}&lt;/span&gt;, &lt;span class="nv"&gt;heapMemory&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;Memory&lt;span class="o"&gt;{&lt;/span&gt;&lt;span class="nv"&gt;bytes&lt;/span&gt;&lt;span class="o"&gt;=&lt;/span&gt;0&lt;span class="o"&gt;}}]&lt;/span&gt;, sequence &lt;span class="o"&gt;[&lt;/span&gt;db6b679c-67cc-43b8-b64a-acaa85c2a4c0], assigned &lt;span class="o"&gt;[&lt;/span&gt;1080750681855361026]
&lt;/code&gt;&lt;/pre&gt;

&lt;/div&gt;



</description>
      <category>apacheseatunnel</category>
      <category>elasticsearch</category>
      <category>docker</category>
      <category>kafka</category>
    </item>
  </channel>
</rss>
