Flume总结（1）

一、日志采集：从网络端口接收数据，下沉到logger

文件netcat-logger.conf:

 # Name the components on this agent

 #给那三个组件取个名字

 a1.sources = r1

 a1.sinks = k1

 a1.channels = c1

 # Describe/configure the source

 #类型, 从网络端口接收数据,在本机启动, 所以localhost, type=spoolDir采集目录源,目录里有就采

 a1.sources.r1.type = netcat

 a1.sources.r1.bind = localhost

 a1.sources.r1.port = 44444

 # Describe the sink

 a1.sinks.k1.type = logger

 # Use a channel which buffers events in memory

 #下沉的时候是一批一批的, 下沉的时候是一个个eventChannel参数解释：

 #capacity：默认该通道中最大的可以存储的event数量

 #trasactionCapacity：每次最大可以从source中拿到或者送到sink中的event数量

 a1.channels.c1.type = memory

 a1.channels.c1.capacity = 1000

 a1.channels.c1.transactionCapacity = 100

 # Bind the source and sink to the channel

 a1.sources.r1.channels = c1

 a1.sinks.k1.channel = c1

启动命令：
#告诉flum启动一个agent,指定配置参数, --name:agent的名字,
flume-ng agent --conf conf --conf-file conf/netcat-logger.conf --name a1 -Dflume.root.logger=INFO,console

传入数据：

[root@mini03 ~]# telnet localhost 44444

Trying ::1...

telnet: connect to address ::1: Connection refused

Trying 127.0.0.1...

Connected to localhost.

Escape character is '^]'.

hello world!^H^H^H^H^H^H^H^H^H^H^H^H^H^H

OK

tianjun2012!

OK

控制台看到的数据

2017-05-08 13:41:35,766 (SinkRunner-PollingRunner-DefaultSinkProcessor) [INFO - org.apache.flume.sink.LoggerSink.process(LoggerSink.java:94)] Event: { headers:{} body: 68 65 6C 6C 6F 20 77 6F 72 6C 64 21 08 08 08 08 hello world!.... }

2017-05-08 13:41:40,153 (SinkRunner-PollingRunner-DefaultSinkProcessor) [INFO - org.apache.flume.sink.LoggerSink.process(LoggerSink.java:94)] Event: { headers:{} body: 74 69 61 6E 6A 75 6E 32 30 31 32 21 0D tianjun2012!. }

二、监视文件夹

启动命令：
bin/flume-ng agent -c ./conf -f ./conf/spooldir-logger.conf -n a1 -Dflume.root.logger=INFO,console

测试：往/home/hadoop/flumespool放文件（mv ././xxxFile /home/hadoop/flumeSpool），但是不要在里面生成文件

# Name the components on this agent

a1.sources = r1

a1.sinks = k1

a1.channels = c1

# Describe/configure the source

#监听目录,spoolDir指定目录, fileHeader要不要给文件夹前坠名

a1.sources.r1.type = spooldir

a1.sources.r1.spoolDir = /home/hadoop/flumespool

a1.sources.r1.fileHeader = true

# Describe the sink

a1.sinks.k1.type = logger

# Use a channel which buffers events in memory

a1.channels.c1.type = memory

a1.channels.c1.capacity = 1000

a1.channels.c1.transactionCapacity = 100

# Bind the source and sink to the channel

a1.sources.r1.channels = c1

a1.sinks.k1.channel = c1

三、用tail命令获取数据，下沉到hdfs

 # Name the components on this agent

 a1.sources = r1

 a1.sinks = k1

 a1.channels = c1

 # Describe/configure the source

 a1.sources.r1.type = exec

 a1.sources.r1.command = tail -F /home/hadoop/log/test.log

 a1.sources.r1.channels = c1

 # Describe the sink

 a1.sinks.k1.type = hdfs

 a1.sinks.k1.channel = c1

 a1.sinks.k1.hdfs.path = hdfs://mini01:9000/flume/events/%y-%m-%d/%H%M/

 a1.sinks.k1.hdfs.filePrefix = events-

 a1.sinks.k1.hdfs.round = true

 a1.sinks.k1.hdfs.roundValue = 10

 a1.sinks.k1.hdfs.roundUnit = minute

 a1.sinks.k1.hdfs.rollInterval = 3

 a1.sinks.k1.hdfs.rollSize = 20

 a1.sinks.k1.hdfs.rollCount = 5

 a1.sinks.k1.hdfs.batchSize = 1

 a1.sinks.k1.hdfs.useLocalTimeStamp = true

 #生成的文件类型，默认是Sequencefile，可用DataStream，则为普通文本

 a1.sinks.k1.hdfs.fileType = DataStream

 # Use a channel which buffers events in memory

 a1.channels.c1.type = memory

 a1.channels.c1.capacity = 1000

 a1.channels.c1.transactionCapacity = 100

 # Bind the source and sink to the channel

 a1.sources.r1.channels = c1

 a1.sinks.k1.channel = c1

启动命令：
flume-ng agent -c conf -f conf/tail-hdfs.conf -n a1

模拟写入日志：

 [root@mini03 log]# i=1;

 while(( $i<=500000 ));

  do echo $i >> /home/hadoop/log/test.log;

  sleep 0.5;

 let 'i++';done

查看hdfs上的文件内容

 [root@mini01 ~]# hdfs dfs -cat /flume/events/17-05-08/1530/*

 1

 2

 3

 4

 5

 6

 7

 8

 9

 10

 11

 12

 13

 14

 15

 16

 17

 18

 19

注意，本例中，为了快速看到效果，这个值都设置比较小，真实情况需要调整

a1.sinks.k1.hdfs.roundValue = 10

a1.sinks.k1.hdfs.rollInterval = 3

a1.sinks.k1.hdfs.rollSize = 20

a1.sinks.k1.hdfs.rollCount = 5

22 a1.sinks.k1.hdfs.batchSize = 1

下面给一个真实环境中的配置：

agent1.sources = spooldirSource
agent1.channels = fileChannel
agent1.sinks = hdfsSink

agent1.sources.spooldirSource.type=spooldir
agent1.sources.spooldirSource.spoolDir=/home/hadoop/log
agent1.sources.spooldirSource.channels=fileChannel

agent1.sinks.hdfsSink.type=hdfs
agent1.sinks.hdfsSink.hdfs.path=hdfs://mini01:9000/weblog/flume-input/%y-%m-%d
agent1.sinks.hdfsSink.hdfs.filePrefix=flume-
agent1.sinks.sink1.hdfs.round = true
# Number of seconds to wait before rolling current file (0 = never roll based on time interval)
agent1.sinks.hdfsSink.hdfs.rollInterval = 3600
# File size to trigger roll, in bytes (0: never roll based on file size)
agent1.sinks.hdfsSink.hdfs.rollSize = 128000000
agent1.sinks.hdfsSink.hdfs.rollCount = 0
agent1.sinks.hdfsSink.hdfs.batchSize = 1000

#Rounded down to the highest multiple of this (in the unit configured using hdfs.roundUnit), less than current time.
agent1.sinks.hdfsSink.hdfs.roundValue = 1
agent1.sinks.hdfsSink.hdfs.roundUnit = minute
agent1.sinks.hdfsSink.hdfs.useLocalTimeStamp = true
agent1.sinks.hdfsSink.channel=fileChannel
agent1.sinks.hdfsSink.hdfs.fileType = DataStream

agent1.channels.fileChannel.type = file
agent1.channels.fileChannel.checkpointDir=/tmp/flume/flume-bineckpoint
agent1.channels.fileChannel.dataDirs=/tmp/flume/dataDir

bin/flume-ng agent --conf ./conf/ -f conf/spooldir-hdfs.conf -Dflume.root.logger=DEBUG,console -n agent1 > log.log 2>&1 &

秒客网

Flume总结（1）

相关文章