分流
所谓的分流就是将一条数据去拆分成多条数据流, 就是基于一个DataStream , 来得到多个平等的DataStream
简单的代码实现 使用侧输出流
package com.dcit.chapter08;
import com.dcit.chacpter01.ClickSource;
import com.dcit.chacpter01.Event;
import org.apache.flink.api.common.eventtime.SerializableTimestampAssigner;
import org.apache.flink.api.common.eventtime.WatermarkStrategy;
import org.apache.flink.api.java.tuple.Tuple3;
import org.apache.flink.streaming.api.datastream.SingleOutputStreamOperator;
import org.apache.flink.streaming.api.environment.StreamExecutionEnvironment;
import org.apache.flink.streaming.api.functions.ProcessFunction;
import org.apache.flink.util.Collector;
import org.apache.flink.util.OutputTag;
import java.time.Duration;
public class SplitStreamTest {
public static void main(String[] args) throws Exception {
StreamExecutionEnvironment env = StreamExecutionEnvironment.getExecutionEnvironment();
env.setParallelism(1);
SingleOutputStreamOperator<Event> stream = env.addSource(new ClickSource())
.assignTimestampsAndWatermarks(WatermarkStrategy.<Event>forBoundedOutOfOrderness(Duration.ZERO)
.withTimestampAssigner(new SerializableTimestampAssigner<Event>() {
@Override
public long extractTimestamp(Event event, long l) {
return event.timestamp;
}
}));
// 定义标签输出流
OutputTag<Tuple3<String, String, Long>> bobTag = new OutputTag<Tuple3<String, String, Long>>("Bob"){};
OutputTag<Tuple3<String, String, Long>> maryTag = new OutputTag<Tuple3<String, String, Long>>("Mary"){};
SingleOutputStreamOperator<Event> splitStream = stream.process(new ProcessFunction<Event, Event>() {
@Override
public void processElement(Event value, Context ctx, Collector<Event> out) throws Exception {
if (value.user.equals("Mary") ) {
// 进行筛选Mary的数据放入标签为maryTag的侧输出流中
ctx.output(maryTag, Tuple3.of(value.user, value.url, value.timestamp));
} else if (value.user.equals("Bob")) {
// 进行筛选Bob的数据放入标签为bobTag的侧输出流中
ctx.output(bobTag, Tuple3.of(value.user, value.url, value.timestamp));
} else {
out.collect(value);
}
}
});
// 获取侧输出流的数据
splitStream.getSideOutput(bobTag).print("bobTag");
splitStream.getSideOutput(maryTag).print("maryTag");
splitStream.print();
env.execute();
}
}