xxl-job 整合 Seatunnel 实现定时任务

流处理

shell 复制代码
#!/bin/bash
SEATUNNEL_CMD="$SEATUNNEL_HOME/bin/seatunnel.sh"
SEATUNNEL_HOST=localhost
SEATUNNEL_PORT=5801

# 定义任务停止时执行的清理操作
exit_func() {
    # 在这里放入你希望在任务停止时执行的操作,比如释放资源、记录日志等
	$SEATUNNEL_CMD -can "$JOB_ID"
    exit;
}

# 捕获 SIGINT (Ctrl+C) 和 SIGTERM (手动终止) 信号
trap exit_func SIGINT SIGTERM SIGHUP SIGQUIT SIGKILL

# 将配置内容写入变量
config_content=$(cat <<EOL
env {
    "job.mode"=STREAMING
    "job.name"="SeaTunnel_Job"
    "savemode.execute.location"=CLUSTER
}
source {
    MySQL-CDC {
        "snapshot.split.size"="8096"
        "snapshot.fetch.size"="1024"
        "incremental.parallelism"="1"
        "connect.timeout.ms"="30000"
        "connect.max-retries"="3"
        "connection.pool.size"="20"
        "chunk-key.even-distribution.factor.lower-bound"="0.05"
        "chunk-key.even-distribution.factor.upper-bound"="100.0"
        "sample-sharding.threshold"="1000"
        "inverse-sampling.rate"="1000"
        "startup.mode"=INITIAL
        "exactly_once"="false"
        "stop.mode"=NEVER
        parallelism="1"
        "result_table_name"=Table15381274549824
        catalog {
            factory=Mysql
        }
        database-names=[
            "test_source"
        ]
        table-names=[
            "test_source.user"
        ]
        format=DEFAULT
        password="123456"
        username=root
        base-url="jdbc:mysql://127.0.0.1:3306/test_cdc"
        server-time-zone=UTC
    }
}
transform {
}
sink {
    Jdbc {
        "schema_save_mode"="CREATE_SCHEMA_WHEN_NOT_EXIST"
        "data_save_mode"="APPEND_DATA"
        "create_index"="true"
        "connection_check_timeout_sec"="30"
        "batch_size"="1000"
        "is_exactly_once"="false"
        "max_commit_attempts"="3"
        "transaction_timeout_sec"="-1"
        "max_retries"="0"
        "auto_commit"="true"
        "support_upsert_by_query_primary_key_exist"="false"
        "multi_table_sink_replica"="1"
        "source_table_name"=Table15381274549824
        "generate_sink_sql"=true
        database="test_jdbc"
        table=user
        driver="com.mysql.cj.jdbc.Driver"
        url="jdbc:mysql://127.0.0.1:3306/test_jdbc"
        password="123456"
        user=root
    }
}
EOL
)

echo "开始执行任务"
echo "--------    配置信息    --------------"
echo "$config_content"
echo "--------    end    --------------"

# 将配置内容写入标准输入并传递给 SeaTunnel
SUBJOB_OUTPUT=$(echo "$config_content" | $SEATUNNEL_CMD --config /dev/stdin --async 2>&1)
JOB_ID=$(echo "$SUBJOB_OUTPUT" | grep "job name" | awk -F'job id: ' '{print $2}' | awk -F',' '{print $1}')

echo "任务Id: $JOB_ID"

# 监控任务状态
while true; do
    STATUS_OUTPUT=$(curl -s http://$SEATUNNEL_HOST:$SEATUNNEL_PORT/hazelcast/rest/maps/job-info/$JOB_ID)
    echo $(date "+%Y-%m-%d %H:%M:%S.%3N") "写入数量 : "$(echo "$STATUS_OUTPUT" | awk -F'"SinkWriteCount":"' '{print $2}' | awk -F '","' '{print $1}')", 读取数量 :"$(echo "$STATUS_OUTPUT" | awk -F'"SourceReceivedCount":"' '{print $2}' | awk -F '","' '{print $1}')
    
	TASK_STATE=$(echo "$STATUS_OUTPUT" | awk -F'"jobStatus":"' '{print $2}' | awk -F '","' '{print $1}')

    if [[ "$TASK_STATE" == "FINISHED" ]]; then
        echo "任务完成, 状态: $TASK_STATE"
        exit 0
    fi
    
    if [[ "$TASK_STATE" != "RUNNING" ]]; then
        echo "任务已结束,状态:$TASK_STATE"
        exit 1
    else
        echo "任务运行中 ... 状态: $TASK_STATE"
        sleep 300
    fi
done

批处理

shell 复制代码
#!/bin/bash

SEATUNNEL_CMD="$SEATUNNEL_HOME/bin/seatunnel.sh"

# 定义任务停止时执行的清理操作
exit_func() {
    # 在这里放入你希望在任务停止时执行的操作,比如释放资源、记录日志等
	$SEATUNNEL_CMD -can "$JOB_ID"
    exit;
}

# 捕获 SIGINT (Ctrl+C) 和 SIGTERM (手动终止) 信号
trap exit_func SIGINT SIGTERM SIGHUP SIGQUIT SIGKILL


# 将配置内容写入变量
config_content=$(cat <<EOL
env {
  # You can set SeaTunnel environment configuration here
  parallelism = 2
  job.mode = "BATCH"
  checkpoint.interval = 10000
}

source {
  # This is a example source plugin **only for test and demonstrate the feature source plugin**
  FakeSource {
    parallelism = 2
    result_table_name = "fake"
    row.num = 16
    schema = {
      fields {
        name = "string"
        age = "int"
      }
    }
  }

  # If you would like to get more information about how to configure SeaTunnel and see full list of source plugins,
  # please go to https://seatunnel.apache.org/docs/connector-v2/source
}

sink {
  Console {
  }

  # If you would like to get more information about how to configure SeaTunnel and see full list of sink plugins,
  # please go to https://seatunnel.apache.org/docs/connector-v2/sink
}
EOL
)

echo "开始执行任务"
# 将配置内容写入标准输入并传递给 SeaTunnel
SUBJOB_OUTPUT=$(echo "$config_content" | $SEATUNNEL_CMD --config /dev/stdin --async 2>&1)
JOB_ID=$(echo "$SUBJOB_OUTPUT" | grep "job name" | awk -F'job id: ' '{print $2}' | awk -F',' '{print $1}')

echo "任务Id: $JOB_ID"

# 监控任务状态
while true; do
    # 查询任务状态
    STATUS_OUTPUT=$($SEATUNNEL_CMD -j "$JOB_ID" 2>&1)
    TASK_STATE=$(echo "$STATUS_OUTPUT" | grep "$JOB_ID" | awk -F'"jobStatus":"' '{print $2}' | awk -F '","' '{print $1}')

    if [[ "$TASK_STATE" == "FINISHED" ]]; then
        echo "任务完成, 状态: $TASK_STATE"
        exit 0
    fi
    # 检查任务是否已完成
    if [[ "$TASK_STATE" != "RUNNING" ]]; then
        echo "任务已结束,状态:$TASK_STATE"
        exit 1
    else
        echo "任务运行中 ... 状态: $TASK_STATE"
        # 等待 5 秒后再次查询
        sleep 5
    fi
done
相关推荐
浪小满1 分钟前
linux下使用脚本实现对进程的内存占用自动化监测
linux·运维·自动化·内存占用情况监测
东软吴彦祖15 分钟前
包安装利用 LNMP 实现 phpMyAdmin 的负载均衡并利用Redis实现会话保持nginx
linux·redis·mysql·nginx·缓存·负载均衡
卷卷的小趴菜学编程37 分钟前
c++之List容器的模拟实现
服务器·c语言·开发语言·数据结构·c++·算法·list
艾杰Hydra40 分钟前
LInux配置PXE 服务器
linux·运维·服务器
多恩Stone43 分钟前
【ubuntu 连接显示器无法显示】可以通过 ssh 连接 ubuntu 服务器正常使用,但服务器连接显示器没有输出
服务器·ubuntu·计算机外设
慵懒的猫mi1 小时前
deepin分享-Linux & Windows 双系统时间不一致解决方案
linux·运维·windows·mysql·deepin
阿无@_@1 小时前
2、ceph的安装——方式二ceph-deploy
linux·ceph·centos
牙牙7051 小时前
ansible一键安装nginx二进制版本
服务器·nginx·ansible
小高不明2 小时前
仿 RabbitMQ 的消息队列2(实战项目)
java·数据库·spring boot·spring·rabbitmq·mvc
DZSpace2 小时前
使用 Helm 安装 Redis 集群
数据库·redis·缓存