This commit is contained in:
dami
2026-07-02 13:54:31 +08:00
parent 8ea16fcd85
commit 0f1b95cdfc
2 changed files with 410 additions and 145 deletions
+256 -64
View File
@@ -1,55 +1,106 @@
--[[
webstats_common.lua - WebStats 统计系统核心模块
==================================================
本模块提供 WebStats 系统的核心功能,包括:
- 数据库连接管理
- 日志数据处理与存储
- 定时任务调度
- 爬虫/客户端识别
- IP/URL 统计
架构说明:
1. 单例模式:通过 getInstance() 获取全局唯一实例
2. 共享内存:使用 ngx.shared.mw_total 作为日志缓存队列
3. SQLite 数据库:每个站点独立一个数据库文件
4. 定时任务:每 0.5 秒执行一次日志持久化
主要数据流向:
webstats_log.lua (日志采集) -> ngx.shared.mw_total (缓存队列) -> cron() (定时处理) -> SQLite (持久化)
依赖模块:
- cjson: JSON 编解码
- lsqlite3: SQLite 数据库操作
- webstats_config: 全局配置
- webstats_sites: 站点配置
]]
local setmetatable = setmetatable
local _M = { _VERSION = '1.0' }
local mt = { __index = _M }
-- 依赖模块
local json = require "cjson"
local sqlite3 = require "lsqlite3"
local config = require "webstats_config"
local sites = require "webstats_sites"
-- 调试模式开关
local debug_mode = true
-- 共享内存队列键名
local total_key = "log_kv_total"
-- 未匹配到站点时使用的默认名称
local unset_server_name = "unset"
-- 日志 ID 最大值,超过后重置
local max_log_id = 99999999999999
-- Nginx 共享内存字典实例
local cache = ngx.shared.mw_total
local today = ngx.re.gsub(ngx.today(),'-','')
local request_header = ngx.req.get_headers()
local method = ngx.req.get_method()
-- 当前日期(日)
local day = os.date("%d")
local number_day = tonumber(day)
local day_column = "day"..number_day
local flow_column = "flow"..number_day
local spider_column = "spider_flow"..number_day
local auto_config = nil
-- 数据库列名:day + 日期(如 day01)
local day_column = "day" .. number_day
-- 数据库列名:flow + 日期(如 flow01)
local flow_column = "flow" .. number_day
-- 日志文件存储目录
local log_dir = "{$SERVER_APP}/logs"
--[[
创建模块实例
@return table 模块实例对象
]]
function _M.new(self)
local self = {
total_key = total_key,
params = nil,
site_config = nil,
config = nil,
total_key = total_key, -- 共享内存队列键名
params = nil, -- 请求参数
site_config = nil, -- 站点配置
config = nil, -- 全局配置
}
return setmetatable(self, mt)
end
--[[
获取单例实例(懒加载)
首次调用时创建实例并启动定时任务
@return table 全局唯一的模块实例
]]
function _M.getInstance(self)
if self.instance == nil then
self.instance = self:new()
self:cron()
self:cron() -- 启动定时任务
end
assert(self.instance ~= nil)
return self.instance
end
--[[
初始化 SQLite 数据库连接
@param input_sn string 站点名称
@return table|nil SQLite 数据库对象,失败返回 nil
SQLite 优化配置:
- synchronous = 0: 关闭同步写入,提升性能(可能丢失数据)
- cache_size = 8000: 设置页缓存大小为 8000 页
- page_size = 32768: 设置页大小为 32KB
- journal_mode = wal: 使用 WAL 模式,支持并发读写
- journal_size_limit = 20GB: WAL 文件大小限制
]]
function _M.initDB(self, input_sn)
local path = log_dir .. '/' .. input_sn .. "/logs.db"
local db, err = sqlite3.open(path)
@@ -66,29 +117,54 @@ function _M.initDB(self, input_sn)
return db
end
--[[
获取共享内存队列键名
@return string 队列键名
]]
function _M.getTotalKey(self)
return self.total_key
end
--[[
JSON 编码
@param msg table 待编码的数据
@return string JSON 字符串
]]
function _M.to_json(self, msg)
return json.encode(msg)
end
function _M.setConfData( self, config, site_config )
--[[
设置配置数据
@param config table 全局配置
@param site_config table 站点配置
]]
function _M.setConfData(self, config, site_config)
self.config = config
self.site_config = site_config
end
function _M.setParams( self, params )
--[[
设置请求参数
@param params table 请求参数
]]
function _M.setParams(self, params)
self.params = params
end
--[[
设置输入站点名称并获取合并后的配置
将站点配置与全局配置合并,站点配置优先级更高
@param input_sn string 站点名称
@return table 合并后的配置
]]
function _M.setInputSn(self, input_sn)
local global_config = config["global"]
if config[input_sn] == nil then
auto_config = global_config
else
auto_config = config[input_sn]
-- 合并全局配置到站点配置(站点配置优先)
for k, v in pairs(global_config) do
if auto_config[k] == nil then
auto_config[k] = v
@@ -98,96 +174,153 @@ function _M.setInputSn(self, input_sn)
return auto_config
end
--[[
获取当前请求的域名
@return string 域名,失败返回 "unknown"
]]
function _M.get_domain(self)
local domain = ngx.req.get_headers()['host']
-- domain = ngx.re.gsub(domain, "_", ".")
if domain == nil then
domain = "unknown"
end
return domain
end
--[[
字符串分割函数
@param str string|table 待分割的字符串或已分割的表
@param reps string 分隔符
@return table 分割后的数组
注意:如果输入已经是 table 类型,直接返回原表(用于处理反向代理传递的数据)
]]
function _M.split(self, str, reps)
local arr = {}
-- 修复反向代理代过来的数据
if "table" == type(str) then
return str
end
string.gsub(str,'[^'..reps..']+',function(w) table.insert(arr,w) end)
local arr = {}
local pattern = "[^" .. reps .. "]+"
for w in string.gmatch(str, pattern) do
arr[#arr + 1] = w
end
return arr
end
--[[
获取数组长度
@param arr table 数组
@return number 数组长度,空或 nil 返回 0
]]
function _M.arrlen(self, arr)
if not arr then return 0 end
local count = 0
for _,v in ipairs(arr) do
count = count + 1
end
return count
return #arr
end
--[[
验证 IPv4 地址格式
@param client_ip string IP 地址
@return boolean 是否为有效的 IPv4 地址
]]
function _M.is_ipaddr(self, client_ip)
local cipn = self:split(client_ip,'.')
if self:arrlen(cipn) < 4 then return false end
for _,v in ipairs({1,2,3,4})
do
local ipv = tonumber(cipn[v])
if ipv == nil then return false end
if ipv > 255 or ipv < 0 then return false end
if not client_ip then return false end
local parts = self:split(client_ip, '.')
if #parts ~= 4 then return false end
for i = 1, 4 do
local ipv = tonumber(parts[i])
if not ipv or ipv < 0 or ipv > 255 then return false end
end
return true
end
--[[
根据域名/服务器名获取站点名称(带缓存)
匹配规则:
1. 精确匹配站点名称
2. 精确匹配站点域名列表
3. 通配符匹配(支持 * 通配符)
@param input_sn string 输入的域名或服务器名
@return string 匹配到的站点名称,未匹配返回 "unset"
缓存策略:匹配结果缓存 24 小时(86400 秒)
]]
function _M.get_sn(self, input_sn)
-- 优先从缓存获取
local dst_name = cache:get(input_sn)
if dst_name then return dst_name end
-- self:D(json.encode(sites))
for _,v in ipairs(sites)
do
-- 遍历所有站点进行匹配
for _, v in ipairs(sites) do
-- 精确匹配站点名称
if input_sn == v["name"] then
cache:set(input_sn, v['name'], 86400)
return v["name"]
end
-- self:D("get_sn:"..json.encode(v))
for _,dst_domain in ipairs(v['domains'])
do
-- 遍历站点的域名列表
for _, dst_domain in ipairs(v['domains']) do
-- 精确匹配域名
if input_sn == dst_domain then
cache:set(input_sn, v['name'], 86400)
return v['name']
elseif string.find(dst_domain, "*") then
local new_domain = string.gsub(dst_domain, '*', '.*')
if string.find(input_sn, new_domain) then
dst_domain = v['name']
cache:set(input_sn, dst_domain, 86400)
-- 通配符匹配(如 *.example.com)
elseif string.find(dst_domain, "*", 1, true) then
local pattern = "^" .. string.gsub(dst_domain, "*", ".*") .. "$"
if ngx.re.match(input_sn, pattern, "ijo") then
cache:set(input_sn, v['name'], 86400)
return v['name']
end
end
end
end
-- 未匹配到任何站点
cache:set(input_sn, unset_server_name, 86400)
return unset_server_name
end
--[[
获取存储键(按小时分组)
格式:YYYYMMDDHH(如 2026070213)
@return string 存储键
]]
function _M.get_store_key(self)
return os.date("%Y%m%d%H", ngx.time())
end
--[[
根据指定时间获取存储键
@param htime number 时间戳
@return string 存储键
]]
function _M.get_store_key_with_time(self, htime)
return os.date("%Y%m%d%H", htime)
end
--[[
获取响应体大小
@return number 响应体字节数
]]
function _M.get_length(self)
local clen = ngx.var.body_bytes_sent
local clen = ngx.var.body_bytes_sent
if clen == nil then clen = 0 end
return tonumber(clen)
end
--[[
获取日志唯一 ID(自动递增)
使用共享内存实现跨 Worker 进程的 ID 递增
@param input_sn string 站点名称
@return number 日志 ID
溢出处理:ID 达到 max_log_id 后重置为 1
]]
function _M.get_last_id(self, input_sn)
local last_insert_id_key = input_sn .. "_last_id"
-- 原子递增,初始值为 0
local new_id, err = cache:incr(last_insert_id_key, 1, 0)
-- 溢出检查
if new_id >= max_log_id then
cache:set(last_insert_id_key, 1)
new_id = cache:get(last_insert_id_key)
@@ -195,12 +328,18 @@ function _M.get_last_id(self, input_sn)
return new_id
end
--[[
获取 HTTP 请求原始数据(请求头 + 请求体)
仅对非 GET 请求记录请求体
@return string JSON 格式的请求数据
]]
function _M.get_http_origin(self)
local data = ""
local headers = ngx.req.get_headers()
if not headers then return data end
local req_method = ngx.req.get_method()
-- 仅记录非 GET 请求的请求体
if req_method ~= 'GET' then
data = ngx.var.request_body
if not data then
@@ -216,48 +355,101 @@ function _M.get_http_origin(self)
return json.encode(headers)
end
--[[
定时任务前置准备(初始化统计记录)
为当前小时和下一小时预创建统计记录,避免 UPDATE 时记录不存在
处理流程:
1. 获取锁防止并发执行
2. 遍历所有站点
3. 为每个站点初始化数据库连接
4. 在事务中预创建统计记录(request_stat, client_stat, spider_stat)
5. 提交事务并关闭数据库
@return boolean 是否成功
]]
function _M.cronPre(self)
self:lock_working('cron_init_stat')
-- 获取当前小时和下一小时的存储键
local time_key = self:get_store_key()
local time_key_next = self:get_store_key_with_time(ngx.time()+3600)
local time_key_next = self:get_store_key_with_time(ngx.time() + 3600)
-- 需要预创建的统计表
local wc_stat = {'request_stat', 'client_stat', 'spider_stat'}
for site_k, site_v in ipairs(sites) do
-- 遍历所有站点
for _, site_v in ipairs(sites) do
local input_sn = site_v["name"]
local db = self:initDB(input_sn)
local wc_stat = {
'request_stat',
'client_stat',
'spider_stat'
}
local v1 = true
local v2 = true
for _,ws_v in pairs(wc_stat) do
v1 = self:_update_stat_pre(db, ws_v, time_key)
v2 = self:_update_stat_pre(db, ws_v, time_key_next)
-- 安全地初始化数据库连接
local ok, db = pcall(function() return self:initDB(input_sn) end)
if not ok or not db then
self:D("cronPre initDB failed for " .. input_sn .. ": " .. tostring(db))
self:unlock_working('cron_init_stat')
return false
end
if db and db:isopen() then
db:execute([[COMMIT]])
db:close()
-- 开启事务
db:exec([[BEGIN TRANSACTION]])
local success = true
-- 预创建当前小时和下一小时的统计记录
for _, ws_v in ipairs(wc_stat) do
if not self:_update_stat_pre(db, ws_v, time_key) then
success = false
break
end
if not self:_update_stat_pre(db, ws_v, time_key_next) then
success = false
break
end
end
if not v1 or not v2 then
-- 提交或回滚事务
if success then
pcall(function() db:execute([[COMMIT]]) end)
else
pcall(function() db:exec([[ROLLBACK]]) end)
end
-- 安全关闭数据库
pcall(function() db:close() end)
if not success then
self:unlock_working('cron_init_stat')
return false
end
end
self:unlock_working('cron_init_stat')
return true
end
--[[
定时任务调度器
启动一个每 0.5 秒执行一次的定时器,负责:
1. 从共享内存队列中读取日志数据
2. 将日志写入 SQLite 数据库
3. 统计请求、客户端、爬虫数据
4. 更新 IP 和 URI 统计
5. 清理过期日志
执行流程:
1. 检查队列是否有数据
2. 获取锁防止并发执行
3. 初始化所有站点的数据库连接
4. 批量处理队列中的日志数据
5. 更新统计信息
6. 清理过期日志
7. 提交事务并关闭连接
]]
function _M.cron(self)
--[[
定时任务核心处理函数
@param premature boolean 是否为定时器提前触发
]]
local timer_every_get_data = function (premature)
local llen = ngx.shared.mw_total:llen(total_key)
+154 -81
View File
@@ -1,5 +1,28 @@
--[[
webstats_log.lua - WebStats 日志采集模块
=========================================
本模块在 Nginx 的 log_by_lua_block 阶段执行,负责:
- 采集请求日志数据
- 过滤不需要统计的请求
- 识别爬虫和客户端类型
- 计算 PV/UV/IP 等指标
- 将日志数据发送到共享内存队列
执行时机:每次请求完成后(log_by_lua_block 阶段)
主要数据流向:
请求完成 -> log_by_lua_block -> 数据采集 -> 过滤处理 -> 爬虫/客户端识别 -> 统计计算 -> 入队
依赖模块:
- cjson: JSON 编解码
- webstats_common: 核心工具模块
- webstats_config: 全局配置
- webstats_sites: 站点配置
]]
log_by_lua_block {
-- 添加 Lua 模块搜索路径
local cpath = "{$SERVER_APP}/lua/"
if not package.cpath:find(cpath) then
package.cpath = cpath .. "?.so;" .. package.cpath
@@ -8,84 +31,64 @@ log_by_lua_block {
package.path = cpath .. "?.lua;" .. package.path
end
local ver = '0.2.4'
-- 调试模式开关
local debug_mode = true
-- 引入核心模块(单例模式)
local __C = require "webstats_common"
local C = __C:getInstance()
-- cache start ---
local cache = ngx.shared.mw_total
local function cache_set(server_name, id, key, val)
local line_kv = "log_kv_" .. server_name .. '_' .. id .. "_" .. key
cache:set(line_kv, val)
end
local function cache_clear(server_name, id, key)
local line_kv = "log_kv_"..server_name..'_'..id.."_"..key
cache:delete(line_kv)
end
local function cache_get(server_name, id, key)
local line_kv = "log_kv_"..server_name..'_'..id.."_"..key
local value = cache:get(line_kv)
return value
end
-- cache end ---
-- domain config is import
local db = nil
-- 依赖模块
local json = require "cjson"
local sqlite3 = require "lsqlite3"
local config = require "webstats_config"
local sites = require "webstats_sites"
-- string.gsub(C:get_sn(ngx.var.server_name),'_','.')
-- 获取站点名称(通过域名匹配)
local server_name = C:get_sn(ngx.var.server_name)
-- 设置配置数据
C:setConfData(config, sites)
-- 获取合并后的站点配置
local auto_config = C:setInputSn(server_name)
-- 获取请求头部和方法
local request_header = ngx.req.get_headers()
local method = ngx.req.get_method()
local excluded = false
local day = os.date("%d")
local number_day = tonumber(day)
local day_column = "day"..number_day
local flow_column = "flow"..number_day
local spider_column = "spider_flow"..number_day
--- default common var end ---
--[[
排除函数模块
=============
以下函数用于判断请求是否应该被排除(不统计),包括:
- IP 排除:全局和站点级别的排除 IP
- 状态码排除:指定状态码的请求不统计
- 扩展名排除:指定扩展名的文件不统计
- URL 排除:指定 URL 路径不统计
]]
local function init_var()
return true
end
--------------------- exclude_func start --------------------------
--[[
加载全局排除 IP 列表到缓存
将配置中的全局排除 IP 存入共享内存,避免每次请求都读取配置
]]
local function load_global_exclude_ip()
local load_key = "global_exclude_ip_load"
-- update global exclude ip
local global_exclude_ip = auto_config["exclude_ip"]
if global_exclude_ip then
for i, _ip in pairs(global_exclude_ip)
do
-- global
-- D("set global exclude ip: ".._ip)
if not cache:get("global_exclude_ip_".._ip) then
cache:set("global_exclude_ip_".._ip, true)
for _, _ip in pairs(global_exclude_ip) do
if not cache:get("global_exclude_ip_" .. _ip) then
cache:set("global_exclude_ip_" .. _ip, true)
end
end
end
-- set tag
cache:set(load_key, true)
end
--[[
加载站点级排除 IP 列表到缓存
@param input_server_name string 站点名称
@return boolean 是否成功
]]
local function load_exclude_ip(input_server_name)
local load_key = input_server_name .. "_exclude_ip_load"
local site_config = config[input_server_name]
@@ -94,24 +97,25 @@ log_by_lua_block {
site_exclude_ip = site_config["exclude_ip"]
end
-- update server_name exclude ip
if site_exclude_ip then
for i, _ip in pairs(site_exclude_ip)
do
cache:set(input_server_name .. "_exclude_ip_".._ip, true)
for _, _ip in pairs(site_exclude_ip) do
cache:set(input_server_name .. "_exclude_ip_" .. _ip, true)
end
end
-- set tag
cache:set(load_key, true)
return true
end
--[[
根据状态码判断是否排除
检查当前请求的状态码是否在排除列表中
@return boolean 是否排除
]]
local function filter_status()
if not auto_config['exclude_status'] then return false end
local the_status = tostring(ngx.status)
for _,v in ipairs(auto_config['exclude_status'])
do
for _, v in ipairs(auto_config['exclude_status']) do
if the_status == v then
return true
end
@@ -119,6 +123,11 @@ log_by_lua_block {
return false
end
--[[
根据文件扩展名判断是否排除
检查请求的 URI 是否以排除的扩展名结尾
@return boolean 是否排除
]]
local function exclude_extension()
local uri = ngx.var.uri
if not uri then return false end
@@ -135,6 +144,11 @@ log_by_lua_block {
return false
end
--[[
根据 URL 判断是否排除
支持精确匹配和正则匹配两种模式
@return boolean 是否排除
]]
local function exclude_url()
local request_uri = ngx.var.request_uri
if not request_uri then return false end
@@ -142,15 +156,18 @@ log_by_lua_block {
local url_conf = auto_config['exclude_url']
if not url_conf then return false end
-- 去掉开头的 '/'
local the_uri = string.sub(request_uri, 2)
for _, conf in ipairs(url_conf) do
local mode = conf["mode"]
local url = conf["url"]
if mode == "regular" then
-- 正则匹配模式
if ngx.re.find(the_uri, url, "ijo") then
return true
end
else
-- 精确匹配模式
if the_uri == url then
return true
end
@@ -159,6 +176,13 @@ log_by_lua_block {
return false
end
--[[
根据 IP 判断是否排除
先检查站点级排除 IP,再检查全局排除 IP
@param input_server_name string 站点名称
@param ip string 客户端 IP
@return boolean 是否排除
]]
local function exclude_ip(input_server_name, ip)
local site_config = config[input_server_name]
if site_config then
@@ -175,8 +199,11 @@ log_by_lua_block {
return cache:get("global_exclude_ip_" .. ip) ~= nil
end
--------------------- exclude_func end ---------------------------
--[[
状态码过滤表
需要记录详细信息的状态码(4xx 和 5xx 错误)
]]
local status_codes_to_log = {
["400"] = true, ["401"] = true, ["402"] = true, ["403"] = true, ["404"] = true,
["405"] = true, ["406"] = true, ["407"] = true, ["408"] = true, ["409"] = true,
@@ -186,18 +213,44 @@ log_by_lua_block {
["449"] = true, ["451"] = true, ["499"] = true, ["500"] = true, ["501"] = true,
["502"] = true, ["503"] = true, ["504"] = true, ["505"] = true, ["506"] = true,
["507"] = true, ["509"] = true, ["510"] = true
}
}
local http_methods = {["get"] = true, ["post"] = true, ["put"] = true, ["patch"] = true, ["delete"] = true}
--[[
HTTP 方法过滤表
需要统计的 HTTP 方法
]]
local http_methods = {["get"] = true, ["post"] = true, ["put"] = true, ["patch"] = true, ["delete"] = true}
local cache = ngx.shared.mw_total
local total_key = "log_kv_total"
-- 共享内存字典实例
local cache = ngx.shared.mw_total
-- 日志队列键名
local total_key = "log_kv_total"
local function cache_logs(input_sn)
--[[
日志采集函数
采集请求数据,识别爬虫/客户端,计算统计指标,然后入队
执行流程:
1. 获取日志 ID 和客户端 IP
2. 判断是否需要排除(状态码/扩展名/URL/IP)
3. 构建 IP 列表(含代理链)
4. 采集请求数据(URI、状态码、响应大小等)
5. 构建日志记录
6. 识别爬虫/客户端类型
7. 计算 PV/UV/IP 指标
8. 将数据入队到共享内存
@param input_sn string 站点名称
]]
local function cache_logs(input_sn)
-- 获取日志唯一 ID
local new_id = C:get_last_id(input_sn)
-- 获取客户端真实 IP(支持 CDN 转发)
local ip = C:get_client_ip()
-- 判断是否排除(状态码/扩展名/URL/IP)
local excluded = filter_status() or exclude_extension() or exclude_url() or exclude_ip(input_sn, ip)
-- 构建 IP 列表(包含代理链)
local ip_list = request_header["x-forwarded-for"]
if ip and not ip_list then
ip_list = ip
@@ -212,21 +265,23 @@ local function cache_logs(input_sn)
ip_list = ip_list .. "," .. remote_addr
end
local request_time = C:get_request_time()
local client_port = ngx.var.remote_port
local uri = tostring(ngx.var.uri)
local status_code = ngx.status
local protocol = ngx.var.server_protocol
local request_uri = ngx.var.request_uri
local body_length = C:get_length()
local domain = C:get_domain()
local referer = ngx.var.http_referer
local user_agent = request_header['user-agent']
local now_time = ngx.time()
-- 采集请求数据
local request_time = C:get_request_time() -- 请求耗时(毫秒)
local client_port = ngx.var.remote_port -- 客户端端口
local uri = tostring(ngx.var.uri) -- 请求路径
local status_code = ngx.status -- 状态码
local protocol = ngx.var.server_protocol -- 协议(HTTP/1.1 等)
local request_uri = ngx.var.request_uri -- 完整请求路径(含参数)
local body_length = C:get_length() -- 响应体大小
local domain = C:get_domain() -- 域名
local referer = ngx.var.http_referer -- 来源页
local user_agent = request_header['user-agent'] -- 用户代理
local now_time = ngx.time() -- 当前时间戳
-- 构建日志记录
local kv = {
id = new_id,
time_key = os.date("%Y%m%d%H", now_time),
time_key = os.date("%Y%m%d%H", now_time), -- 小时级存储键
time = now_time,
ip = ip,
domain = domain,
@@ -240,19 +295,22 @@ local function cache_logs(input_sn)
referer = referer,
user_agent = user_agent,
protocol = protocol,
is_spider = 0,
is_spider = 0, -- 是否爬虫
request_time = request_time,
excluded = excluded,
request_headers = '',
ip_list = ip_list,
excluded = excluded, -- 是否排除
request_headers = '', -- 请求头(异常时记录)
ip_list = ip_list, -- IP 列表(含代理链)
client_port = client_port
}
-- 初始化统计字段
local request_stat_fields = {req = 1, length = body_length}
local spider_stat_fields = {}
local client_stat_fields = {}
-- 非排除请求的额外处理
if not excluded then
-- 记录异常请求的原始数据(500/POST/403)
if status_code == 500 or (method == "POST" and config['global']["record_post_args"] == true) or (status_code == 403 and config['global']["record_get_403_args"] == true) then
local ok, data = pcall(function() return C:get_http_origin() end)
if ok and data then
@@ -260,17 +318,21 @@ local function cache_logs(input_sn)
end
end
-- 统计状态码
if status_codes_to_log[tostring(status_code)] then
request_stat_fields["status_" .. status_code] = 1
end
-- 统计 HTTP 方法
local lower_method = string.lower(method)
if http_methods[lower_method] then
request_stat_fields["http_" .. lower_method] = 1
end
-- 识别爬虫/客户端
local is_spider, request_spider, spider_index = C:match_spider(user_agent)
if not is_spider then
-- 非爬虫:识别客户端类型,计算 PV/UV/IP
client_stat_fields = C:match_client_arr(user_agent)
local pvc, uvc = C:statistics_request(ip, is_spider, body_length)
local ipc = C:statistics_ipc(input_sn, ip)
@@ -279,12 +341,14 @@ local function cache_logs(input_sn)
if uvc > 0 then request_stat_fields["uv"] = 1 end
if pvc > 0 then request_stat_fields["pv"] = 1 end
else
-- 爬虫:记录爬虫类型
kv["is_spider"] = spider_index
spider_stat_fields[request_spider] = 1
request_stat_fields["spider"] = 1
end
end
-- 构建最终数据结构
local data = {
server_name = input_sn,
stat_fields = {
@@ -295,27 +359,36 @@ local function cache_logs(input_sn)
log_kv = kv,
}
-- 入队到共享内存队列
cache:rpush(total_key, json.encode(data))
end
--[[
应用入口函数
加载排除 IP 列表,然后采集日志
]]
local function run_app()
init_var()
load_global_exclude_ip()
load_exclude_ip(server_name)
cache_logs(server_name)
end
--[[
应用入口包装函数(带错误捕获)
在调试模式下,使用 pcall 捕获异常并记录日志
]]
local function run_app_ok()
if not debug_mode then return run_app() end
local presult, err = pcall( function() run_app() end)
local presult, err = pcall(function() run_app() end)
if not presult then
C:D("debug error on :"..tostring(err))
C:D("debug error on :" .. tostring(err))
return true
end
end
-- 执行应用
return run_app_ok()
}