flink-cep实践

java 复制代码
package com.techwolf.hubble;

import com.alibaba.fastjson.JSONObject;
import com.techwolf.hubble.constant.Config;
import com.techwolf.hubble.model.TestEvent;
import org.apache.flink.api.common.eventtime.TimestampAssigner;
import org.apache.flink.api.common.eventtime.TimestampAssignerSupplier;
import org.apache.flink.api.common.eventtime.WatermarkStrategy;
import org.apache.flink.api.common.functions.MapFunction;
import org.apache.flink.cep.CEP;
import org.apache.flink.cep.PatternFlatSelectFunction;
import org.apache.flink.cep.PatternFlatTimeoutFunction;
import org.apache.flink.cep.PatternStream;
import org.apache.flink.cep.pattern.Pattern;
import org.apache.flink.cep.pattern.conditions.SimpleCondition;
import org.apache.flink.streaming.api.TimeCharacteristic;
import org.apache.flink.streaming.api.datastream.DataStream;
import org.apache.flink.streaming.api.datastream.SingleOutputStreamOperator;
import org.apache.flink.streaming.api.environment.StreamExecutionEnvironment;
import org.apache.flink.streaming.api.functions.sink.PrintSinkFunction;
import org.apache.flink.streaming.api.windowing.time.Time;
import org.apache.flink.util.Collector;
import org.apache.flink.util.OutputTag;

import java.util.List;
import java.util.Map;


/**
 * Hello world!
 *
 */
public class App {

    public static void main(String[] args) throws Exception{
        //初始化环境
        StreamExecutionEnvironment env=StreamExecutionEnvironment.getExecutionEnvironment();

        env.setStreamTimeCharacteristic(TimeCharacteristic.EventTime);
        //定义时间戳提取器作为输入流分配时间戳和水位线
        WatermarkStrategy<TestEvent> watermarkStrategy=WatermarkStrategy.<TestEvent>
                forMonotonousTimestamps().withTimestampAssigner(new EventTimeAssignerSupplier());

        DataStream<TestEvent> inputDataSteam=env.fromElements(
                new TestEvent("1","A",System.currentTimeMillis()-100*1000,"1"),
                new TestEvent("1","A",System.currentTimeMillis()-85*1000,"2"),
                new TestEvent("1","A",System.currentTimeMillis()-80*1000,"3"),
                new TestEvent("1","A",System.currentTimeMillis()-75*1000,"4"),
                new TestEvent("1","A",System.currentTimeMillis()-60*1000,"5"),
                new TestEvent("1","A",System.currentTimeMillis()-55*1000,"6"),
                new TestEvent("1","A",System.currentTimeMillis()-40*1000,"7"),
                new TestEvent("1","A",System.currentTimeMillis()-35*1000,"8"),
                new TestEvent("1","A",System.currentTimeMillis()-20*1000,"9"),
                new TestEvent("1","A",System.currentTimeMillis()-10*1000,"10"),
                new TestEvent("1","B",System.currentTimeMillis()-5*1000,"11")
        ).assignTimestampsAndWatermarks(watermarkStrategy);

        Pattern<TestEvent,TestEvent> pattern=Pattern.<TestEvent>begin("begin")
                .where(new SimpleCondition<TestEvent>() {
                    @Override
                    public boolean filter(TestEvent testEvent) throws Exception {
                        return testEvent.getAction().equals("A");
                    }
                }).
                followedBy("end")
                .where(new SimpleCondition<TestEvent>() {
                    @Override
                    public boolean filter(TestEvent testEvent) throws Exception {
                        return testEvent.getAction().equals("B");
                    }
                }).within(Time.seconds(10));


        PatternStream<TestEvent> patternStream=CEP.pattern(inputDataSteam.keyBy(TestEvent::getId),pattern);
        OutputTag<TestEvent> timeOutTag=new OutputTag<TestEvent>("timeOutTag"){};

        //处理匹配结果
        SingleOutputStreamOperator<TestEvent> twentySingleOutputStream=patternStream
                .flatSelect(timeOutTag,new EventTimeOut(),new FlatSelect())
                .uid("match_twenty_minutes_pattern");
        DataStream<String> result=twentySingleOutputStream.getSideOutput(timeOutTag).map(new MapFunction<TestEvent, String>() {
            @Override
            public String map(TestEvent testEvent) throws Exception {
                return JSONObject.toJSONString(testEvent);
            }
        });
        result.print();
        env.execute(Config.JOB_NAME);
    }

    public static class EventTimeOut implements PatternFlatTimeoutFunction<TestEvent,TestEvent> {
        private static final long serialVersionUID = -2471077777598713906L;
        @Override
        public void timeout(Map<String, List<TestEvent>> map, long l, Collector<TestEvent> collector) throws Exception {
            if (null != map.get("begin")) {
                for (TestEvent event : map.get("begin")) {
                    collector.collect(event);
                }
            }
        }
    }

    public static class FlatSelect implements PatternFlatSelectFunction<TestEvent,TestEvent> {
        private static final long serialVersionUID = 1753544074226581611L;
        @Override
        public void flatSelect(Map<String, List<TestEvent>> map, Collector<TestEvent> collector) throws Exception {
            if (null != map.get("begin")) {
                for (TestEvent event : map.get("begin")) {
                    collector.collect(event);
                }
            }
        }
    }

    public static class EventTimeAssignerSupplier implements TimestampAssignerSupplier<TestEvent> {
        private static final long serialVersionUID = -9040340771307752904L;

        @Override
        public TimestampAssigner<TestEvent> createTimestampAssigner(Context context) {
            return new EventTimeAssigner();
        }
    }

    public static class EventTimeAssigner implements TimestampAssigner<TestEvent> {
        @Override
        public long extractTimestamp(TestEvent event, long l) {
            return event.getEventTime();
        }
    }
}
相关推荐
浪子小院2 分钟前
ModelEngine 智能体全流程开发实战:从 0 到 1 搭建多协作办公助手
大数据·人工智能
AEIC学术交流中心43 分钟前
【快速EI检索 | ACM出版】2026年大数据与智能制造国际学术会议(BDIM 2026)
大数据·制造
wending-Y1 小时前
记录一次排查Flink一直重启的问题
大数据·flink
Hello.Reader1 小时前
Flink 对接 Azure Blob Storage / ADLS Gen2:wasb:// 与 abfs://(读写、Checkpoint、插件与认证)
flink·flask·azure
UI设计兰亭妙微1 小时前
医疗大数据平台电子病例界面设计
大数据·界面设计
初恋叫萱萱1 小时前
模型瘦身实战:用 `cann-model-compression-toolkit` 实现高效 INT8 量化
大数据
互联网科技看点2 小时前
孕期科学补铁,保障母婴健康-仁合益康蛋白琥珀酸铁口服溶液成为产妇优选方案
大数据
Dxy12393102162 小时前
深度解析 Elasticsearch:从倒排索引到 DSL 查询的实战突围
大数据·elasticsearch·搜索引擎
Hello.Reader2 小时前
Flink 文件系统通用配置默认文件系统与连接数限制实战
vue.js·flink·npm
YongCheng_Liang2 小时前
零基础学大数据:大数据基础与前置技术夯实
大数据·big data