flink-cep实践

java 复制代码
package com.techwolf.hubble;

import com.alibaba.fastjson.JSONObject;
import com.techwolf.hubble.constant.Config;
import com.techwolf.hubble.model.TestEvent;
import org.apache.flink.api.common.eventtime.TimestampAssigner;
import org.apache.flink.api.common.eventtime.TimestampAssignerSupplier;
import org.apache.flink.api.common.eventtime.WatermarkStrategy;
import org.apache.flink.api.common.functions.MapFunction;
import org.apache.flink.cep.CEP;
import org.apache.flink.cep.PatternFlatSelectFunction;
import org.apache.flink.cep.PatternFlatTimeoutFunction;
import org.apache.flink.cep.PatternStream;
import org.apache.flink.cep.pattern.Pattern;
import org.apache.flink.cep.pattern.conditions.SimpleCondition;
import org.apache.flink.streaming.api.TimeCharacteristic;
import org.apache.flink.streaming.api.datastream.DataStream;
import org.apache.flink.streaming.api.datastream.SingleOutputStreamOperator;
import org.apache.flink.streaming.api.environment.StreamExecutionEnvironment;
import org.apache.flink.streaming.api.functions.sink.PrintSinkFunction;
import org.apache.flink.streaming.api.windowing.time.Time;
import org.apache.flink.util.Collector;
import org.apache.flink.util.OutputTag;

import java.util.List;
import java.util.Map;


/**
 * Hello world!
 *
 */
public class App {

    public static void main(String[] args) throws Exception{
        //初始化环境
        StreamExecutionEnvironment env=StreamExecutionEnvironment.getExecutionEnvironment();

        env.setStreamTimeCharacteristic(TimeCharacteristic.EventTime);
        //定义时间戳提取器作为输入流分配时间戳和水位线
        WatermarkStrategy<TestEvent> watermarkStrategy=WatermarkStrategy.<TestEvent>
                forMonotonousTimestamps().withTimestampAssigner(new EventTimeAssignerSupplier());

        DataStream<TestEvent> inputDataSteam=env.fromElements(
                new TestEvent("1","A",System.currentTimeMillis()-100*1000,"1"),
                new TestEvent("1","A",System.currentTimeMillis()-85*1000,"2"),
                new TestEvent("1","A",System.currentTimeMillis()-80*1000,"3"),
                new TestEvent("1","A",System.currentTimeMillis()-75*1000,"4"),
                new TestEvent("1","A",System.currentTimeMillis()-60*1000,"5"),
                new TestEvent("1","A",System.currentTimeMillis()-55*1000,"6"),
                new TestEvent("1","A",System.currentTimeMillis()-40*1000,"7"),
                new TestEvent("1","A",System.currentTimeMillis()-35*1000,"8"),
                new TestEvent("1","A",System.currentTimeMillis()-20*1000,"9"),
                new TestEvent("1","A",System.currentTimeMillis()-10*1000,"10"),
                new TestEvent("1","B",System.currentTimeMillis()-5*1000,"11")
        ).assignTimestampsAndWatermarks(watermarkStrategy);

        Pattern<TestEvent,TestEvent> pattern=Pattern.<TestEvent>begin("begin")
                .where(new SimpleCondition<TestEvent>() {
                    @Override
                    public boolean filter(TestEvent testEvent) throws Exception {
                        return testEvent.getAction().equals("A");
                    }
                }).
                followedBy("end")
                .where(new SimpleCondition<TestEvent>() {
                    @Override
                    public boolean filter(TestEvent testEvent) throws Exception {
                        return testEvent.getAction().equals("B");
                    }
                }).within(Time.seconds(10));


        PatternStream<TestEvent> patternStream=CEP.pattern(inputDataSteam.keyBy(TestEvent::getId),pattern);
        OutputTag<TestEvent> timeOutTag=new OutputTag<TestEvent>("timeOutTag"){};

        //处理匹配结果
        SingleOutputStreamOperator<TestEvent> twentySingleOutputStream=patternStream
                .flatSelect(timeOutTag,new EventTimeOut(),new FlatSelect())
                .uid("match_twenty_minutes_pattern");
        DataStream<String> result=twentySingleOutputStream.getSideOutput(timeOutTag).map(new MapFunction<TestEvent, String>() {
            @Override
            public String map(TestEvent testEvent) throws Exception {
                return JSONObject.toJSONString(testEvent);
            }
        });
        result.print();
        env.execute(Config.JOB_NAME);
    }

    public static class EventTimeOut implements PatternFlatTimeoutFunction<TestEvent,TestEvent> {
        private static final long serialVersionUID = -2471077777598713906L;
        @Override
        public void timeout(Map<String, List<TestEvent>> map, long l, Collector<TestEvent> collector) throws Exception {
            if (null != map.get("begin")) {
                for (TestEvent event : map.get("begin")) {
                    collector.collect(event);
                }
            }
        }
    }

    public static class FlatSelect implements PatternFlatSelectFunction<TestEvent,TestEvent> {
        private static final long serialVersionUID = 1753544074226581611L;
        @Override
        public void flatSelect(Map<String, List<TestEvent>> map, Collector<TestEvent> collector) throws Exception {
            if (null != map.get("begin")) {
                for (TestEvent event : map.get("begin")) {
                    collector.collect(event);
                }
            }
        }
    }

    public static class EventTimeAssignerSupplier implements TimestampAssignerSupplier<TestEvent> {
        private static final long serialVersionUID = -9040340771307752904L;

        @Override
        public TimestampAssigner<TestEvent> createTimestampAssigner(Context context) {
            return new EventTimeAssigner();
        }
    }

    public static class EventTimeAssigner implements TimestampAssigner<TestEvent> {
        @Override
        public long extractTimestamp(TestEvent event, long l) {
            return event.getEventTime();
        }
    }
}
相关推荐
时序数据说10 小时前
时序数据库为什么选IoTDB?
大数据·数据库·物联网·开源·时序数据库·iotdb
Hello.Reader11 小时前
Elasticsearch JS 客户端子客户端(Child Client)实践指南
大数据·javascript·elasticsearch
阑梦清川13 小时前
派聪明RAG知识库----关于elasticsearch报错,重置密码的解决方案
大数据·elasticsearch·jenkins
ID_1800790547314 小时前
淘宝拍立淘按图搜索API接口功能详细说明
大数据·python·json·图搜索算法
我要学习别拦我~15 小时前
读《精益数据分析》:媒体内容平台全链路梳理
大数据·数据分析·媒体
六哥探店实录17 小时前
外卖:重构餐饮的线上服务密码
大数据·生活·美食
计算机毕设-小月哥19 小时前
【限时分享:Hadoop+Spark+Vue技术栈电信客服数据分析系统完整实现方案
大数据·vue.js·hadoop·python·信息可视化·spark·计算机毕业设计
tonydf19 小时前
ELK开启安全策略
大数据·后端·安全
阿里云大数据AI技术19 小时前
从“字”到“画”:基于Elasticsearch Serverless 的多模态商品搜索实践
大数据·人工智能·搜索引擎
TDengine (老段)20 小时前
TDengine IDMP 基本功能(3.数据三化处理)
大数据·数据库·物联网·ai·语言模型·时序数据库·tdengine