Zig 可观测性:结构化日志、OpenTelemetry 与指标采集

可观测性是现代分布式系统的核心能力。本文系统讲解 Zig 中 std.log 接口的自定义扩展、JSON 结构化日志输出、日志轮转策略、OpenTelemetry trace 与 metric 的 SDK 集成、Prometheus Pull 模型端点实现,以及 gauge/counter/histogram 指标类型的工程实践,帮助开发者构建完整的 Zig 服务可观测体系。

引言

可观测性(Observability)不是可选项,而是生产系统的必备基础设施。一个完整的可观测体系包含三大支柱:日志(Logs)、指标(Metrics)和追踪(Traces)。Zig 作为系统级语言,在可观测性方面有着天然优势——直接控制内存和 I/O 意味着可以零开销地嵌入遥测逻辑,没有运行时垃圾回收带来的停顿干扰。

本文覆盖 Zig 标准库日志接口的扩展、JSON 结构化日志、文件轮转、基于 OpenTelemetry 的分布式追踪、Prometheus 指标暴露端点,以及 gauge/counter/histogram 三种指标类型的实现。

前置:/zig-http-server/(HTTP 端点服务)、/zig-json-serialization/(JSON 序列化)、/zig-concurrency-atomics/(并发指标计数)。


目录


1. std.log 接口与自定义实现

1.1 默认日志行为

Zig 标准库的 std.log 是编译期分发的日志门面:

const std = @import("std");

pub fn main() void {
    std.log.debug("调试信息: x={d}", .{42});      // Debug 和 Trace 默认被过滤
    std.log.info("服务启动: port={d}", .{8080});   // 默认输出到 stderr
    std.log.warn("连接超时: addr={s}", .{"10.0.0.1"});
    std.log.err("数据库连接失败: {}", .{error.ConnectionRefused});
}

默认实现将日志输出到 stderr,格式为简单文本。通过编译参数可以控制日志级别:

# 仅输出 warn 及以上
zig build -Dlog-level=warn

1.2 自定义日志处理器

通过覆盖 std.log.defaultLog 可以实现自定义输出目标(文件、网络、结构化格式):

const std = @import("std");

// 全局日志文件(需要同步访问)
var log_file: ?std.fs.File = null;
var log_mutex: std.Thread.Mutex = .{};

pub fn customLog(
    comptime level: std.log.Level,
    comptime scope: @Type(.enum_literal),
    comptime format: []const u8,
    args: anytype,
) void {
    const level_str = switch (level) {
        .err => "ERROR",
        .warn => "WARN",
        .info => "INFO",
        .debug => "DEBUG",
    };

    const now = std.time.timestamp();

    log_mutex.lock();
    defer log_mutex.unlock();

    const writer = if (log_file) |f| f.writer() else std.io.getStdErr().writer();

    // 格式: [timestamp] [LEVEL] [scope] message
    writer.print("[{d}] [{s}] [{s}] ", .{ now, level_str, @tagName(scope) }) catch return;
    writer.print(format ++ "\n", args) catch return;
}

// 在根作用域覆盖默认日志
pub const log_level: std.log.Level = .info;
pub const log = std.log.scoped(.app);

1.3 多作用域分级

Zig 支持为不同模块配置不同日志级别:

// 设置全局默认级别为 info,但网络模块输出 debug
pub const std_options = struct {
    pub const log_level: std.log.Level = .info;
    pub const log_scope_levels: []const std.log.ScopeLevel = &.{
        .{ .scope = .network, .level = .debug },
        .{ .scope = .db, .level = .warn },
    };
};

// 各模块使用独立 scope
const net_log = std.log.scoped(.network);
const db_log = std.log.scoped(.db);

pub fn initNetwork() void {
    net_log.debug("创建 socket...", .{});      // 输出(network debug)
}

pub fn initDb() void {
    db_log.debug("打开连接池", .{});            // 不输出(db 只输出 warn+)
}

2. JSON 结构化日志

结构化日志(Structured Logging)将日志输出为 JSON 格式,方便日志平台(ELK、Loki、Grafana)解析和索引。

2.1 JSON 格式日志实现

const std = @import("std");

pub const JsonLogger = struct {
    allocator: std.mem.Allocator,
    file: ?std.fs.File,
    mutex: std.Thread.Mutex,

    pub fn log(
        self: *JsonLogger,
        comptime level: std.log.Level,
        message: []const u8,
        fields: anytype,
    ) !void {
        var map = std.StringHashMap(std.json.Value).init(self.allocator);
        defer {
            var it = map.iterator();
            while (it.next()) |entry| {
                self.allocator.free(entry.key_ptr.*);
            }
            map.deinit();
        }

        try map.put(try self.allocator.dupe(u8, "timestamp"), .{ .integer = std.time.timestamp() });
        try map.put(try self.allocator.dupe(u8, "level"), .{ .string = @tagName(level) });
        try map.put(try self.allocator.dupe(u8, "message"), .{ .string = message });

        // 动态添加额外字段
        const fields_info = @typeInfo(@TypeOf(fields)).Struct;
        inline for (fields_info.fields) |field| {
            const key = try std.fmt.allocPrint(self.allocator, "{s}", .{field.name});
            const val = @field(fields, field.name);
            const jv = switch (@typeInfo(@TypeOf(val))) {
                .int, .comptime_int => std.json.Value{ .integer = val },
                .float, .comptime_float => std.json.Value{ .float = val },
                .bool => std.json.Value{ .bool = val },
                else => std.json.Value{ .string = val },
            };
            try map.put(key, jv);
        }

        var buf = std.ArrayList(u8).init(self.allocator);
        defer buf.deinit();

        try std.json.stringify(map, .{}, buf.writer());

        self.mutex.lock();
        defer self.mutex.unlock();

        const writer = if (self.file) |f| f.writer() else std.io.getStdErr().writer();
        try writer.print("{s}\n", .{buf.items});
    }
};

// 使用
var logger = JsonLogger{
    .allocator = allocator,
    .file = try std.fs.cwd().createFile("app.log", .{}),
    .mutex = .{},
};
try logger.log(.info, "用户登录", .{ .user_id = 42, .ip = "192.168.1.1" });
// 输出: {"timestamp":1696000000,"level":"info","message":"用户登录","user_id":42,"ip":"192.168.1.1"}

2.2 zap 日志库简介

社区也提供了 zlog 等纯 Zig 日志库,但 std.log + std.json 的标准库组合已足够满足大多数结构化日志需求,且无外部依赖。


3. 日志轮转与文件管理

生产环境的日志文件不能无限增长。轮转(Log Rotation)策略包括按大小切割、按时间归档和保留策略。

3.1 按大小轮转

pub const RotatingLog = struct {
    allocator: std.mem.Allocator,
    base_path: []const u8,
    max_size: usize,       // 单文件最大字节
    max_files: usize,      // 保留文件数
    current_file: std.fs.File,
    current_size: usize,
    mutex: std.Thread.Mutex,

    pub fn init(allocator: std.mem.Allocator, path: []const u8, max_size: usize, max_files: usize) !RotatingLog {
        const file = try std.fs.cwd().createFile(path, .{ .truncate = false });
        const stat = try file.stat();
        return RotatingLog{
            .allocator = allocator,
            .base_path = try allocator.dupe(u8, path),
            .max_size = max_size,
            .max_files = max_files,
            .current_file = file,
            .current_size = @intCast(stat.size),
            .mutex = .{},
        };
    }

    pub fn write(self: *RotatingLog, data: []const u8) !void {
        self.mutex.lock();
        defer self.mutex.unlock();

        if (self.current_size + data.len > self.max_size) {
            try self.rotate();
        }

        try self.current_file.writeAll(data);
        self.current_size += data.len;
    }

    fn rotate(self: *RotatingLog) !void {
        self.current_file.close();

        // 删除最老的文件
        const oldest = try std.fmt.allocPrint(self.allocator, "{s}.{d}", .{ self.base_path, self.max_files - 1 });
        defer self.allocator.free(oldest);
        std.fs.cwd().deleteFile(oldest) catch {};

        // 依次移动文件 app.log.2 → app.log.3, app.log.1 → app.log.2
        var i: usize = self.max_files - 1;
        while (i > 0) : (i -= 1) {
            const from = try std.fmt.allocPrint(self.allocator, "{s}.{d}", .{ self.base_path, i - 1 });
            defer self.allocator.free(from);
            const to = try std.fmt.allocPrint(self.allocator, "{s}.{d}", .{ self.base_path, i });
            defer self.allocator.free(to);
            std.fs.cwd().rename(from, to) catch {};
        }

        // 当前文件移到 .0
        std.fs.cwd().rename(self.base_path, try std.fmt.allocPrint(self.allocator, "{s}.0", .{self.base_path})) catch {};

        self.current_file = try std.fs.cwd().createFile(self.base_path, .{});
        self.current_size = 0;
    }

    pub fn deinit(self: *RotatingLog) void {
        self.current_file.close();
        self.allocator.free(self.base_path);
    }
};

4. OpenTelemetry Trace 分布式追踪

分布式追踪记录请求在微服务间的流转路径。OpenTelemetry(OTel)是 CNCF 可观测性标准,通过 trace/span 建模调用链。

4.1 OTel Trace 概念

  • Trace:一次完整请求的调用链。
  • Span:调用链中的单个操作单元(如「查询数据库」「调用下游 API」)。
  • Context:跨服务传递的追踪上下文(trace_id + span_id)。

4.2 Zig 中的 Trace 实现

const std = @import("std");

pub const Span = struct {
    trace_id: [16]u8,
    span_id: [8]u8,
    name: []const u8,
    start_time: i64,
    end_time: ?i64 = null,
    parent_id: ?[8]u8 = null,
    attributes: std.StringHashMap([]const u8),

    pub fn init(allocator: std.mem.Allocator, name: []const u8, trace_id: [16]u8, parent_id: ?[8]u8) !Span {
        var span_id: [8]u8 = undefined;
        std.crypto.random.bytes(&span_id);
        return Span{
            .trace_id = trace_id,
            .span_id = span_id,
            .name = name,
            .start_time = std.time.milliTimestamp(),
            .parent_id = parent_id,
            .attributes = std.StringHashMap([]const u8).init(allocator),
        };
    }

    pub fn setAttribute(self: *Span, key: []const u8, value: []const u8) !void {
        try self.attributes.put(key, value);
    }

    pub fn end(self: *Span) void {
        self.end_time = std.time.milliTimestamp();
    }

    pub fn exportJson(self: *Span, allocator: std.mem.Allocator) ![]u8 {
        var buf = std.ArrayList(u8).init(allocator);
        defer buf.deinit();
        try std.json.stringify(self, .{}, buf.writer());
        return buf.toOwnedSlice();
    }
};

4.3 Trace Context 传播

HTTP 请求间通过标准头传播 trace context:

// 从传入请求中提取 traceparent
pub fn extractTraceparent(headers: std.http.Headers) ?[16]u8 {
    const tp = headers.get("traceparent") orelse return null;
    // traceparent 格式: 00-<trace-id>-<span-id>-<flags>
    // 例: 00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01
    if (tp.len < 55) return null;
    var trace_id: [16]u8 = undefined;
    _ = std.fmt.hexToBytes(&trace_id, tp[3..35]) catch return null;
    return trace_id;
}

// 向下游服务注入 traceparent
pub fn injectTraceparent(writer: anytype, trace_id: [16]u8, span_id: [8]u8) !void {
    try writer.writeAll("traceparent: 00-");
    try writer.writeAll(std.fmt.fmtSliceHexLower(&trace_id));
    try writer.writeAll("-");
    try writer.writeAll(std.fmt.fmtSliceHexLower(&span_id));
    try writer.writeAll("-01\r\n");
}

4.4 OTLP 导出

生产环境使用 OTLP(OpenTelemetry Protocol)通过 gRPC 或 HTTP 将 trace 导出到 collector(如 Jaeger、Tempo)。Zig 可以通过 HTTP POST /v1/traces 发送 protobuf/JSON 格式的 span batch:

// 简化的 OTLP/JSON 导出
const otlp_endpoint = "http://localhost:4318/v1/traces";

实际集成建议通过一个轻量 C 绑定或直接构造 HTTP 请求发送 JSON payload。


5. Prometheus 指标与 Pull 端点

5.1 Prometheus 数据模型

Prometheus 采用 Pull 模型:服务暴露 /metrics HTTP 端点,Prometheus Server 定期抓取。指标格式为纯文本行协议:

# HELP http_requests_total Total HTTP requests
# TYPE http_requests_total counter
http_requests_total{method="GET",status="200"} 1027
http_requests_total{method="POST",status="500"} 12

5.2 指标端点实现

const std = @import("std");

pub const PrometheusRegistry = struct {
    allocator: std.mem.Allocator,
    mutex: std.Thread.Mutex,
    metrics: std.ArrayList(Metric),

    pub const Metric = struct {
        name: []const u8,
        help: []const u8,
        mtype: enum { counter, gauge, histogram },
        lines: std.ArrayList(u8), // 序列化的行数据
    };

    pub fn registerCounter(self: *PrometheusRegistry, name: []const u8, help: []const u8) !*Metric {
        self.mutex.lock();
        defer self.mutex.unlock();

        const m = try self.allocator.create(Metric);
        m.* = .{
            .name = try self.allocator.dupe(u8, name),
            .help = try self.allocator.dupe(u8, help),
            .mtype = .counter,
            .lines = std.ArrayList(u8).init(self.allocator),
        };
        try self.metrics.append(m.*);
        return m;
    }

    pub fn render(self: *PrometheusRegistry, writer: anytype) !void {
        self.mutex.lock();
        defer self.mutex.unlock();

        for (self.metrics.items) |m| {
            try writer.print("# HELP {s} {s}\n", .{ m.name, m.help });
            try writer.print("# TYPE {s} {s}\n", .{ m.name, @tagName(m.mtype) });
            try writer.writeAll(m.lines.items);
        }
    }
};

6. Gauge、Counter 与 Histogram 实现

6.1 Counter:只增不减

pub const Counter = struct {
    name: []const u8,
    labels: []const u8, // 如 method="GET",status="200"
    value: std.atomic.Value(u64),

    pub fn init(name: []const u8, labels: []const u8) Counter {
        return .{
            .name = name,
            .labels = labels,
            .value = std.atomic.Value(u64).init(0),
        };
    }

    pub fn inc(self: *Counter, delta: u64) void {
        _ = self.value.fetchAdd(delta, .monotonic);
    }

    pub fn get(self: *Counter) u64 {
        return self.value.load(.monotonic);
    }

    pub fn format(self: *Counter, writer: anytype) !void {
        try writer.print("{s}{s} {d}\n", .{ self.name, self.labels, self.get() });
    }
};

6.2 Gauge:可增可减

pub const Gauge = struct {
    name: []const u8,
    labels: []const u8,
    value: std.atomic.Value(i64),

    pub fn set(self: *Gauge, v: i64) void {
        self.value.store(v, .monotonic);
    }

    pub fn add(self: *Gauge, delta: i64) void {
        _ = self.value.fetchAdd(@intCast(delta), .monotonic);
    }

    pub fn get(self: *Gauge) i64 {
        return self.value.load(.monotonic);
    }
};

6.3 Histogram:延迟分布

pub const Histogram = struct {
    name: []const u8,
    buckets: []const f64,
    counts: []std.atomic.Value(u64),
    sum: std.atomic.Value(f64),
    total_count: std.atomic.Value(u64),

    pub fn observe(self: *Histogram, value: f64) void {
        for (self.buckets, 0..) |bucket, i| {
            if (value <= bucket) {
                _ = self.counts[i].fetchAdd(1, .monotonic);
                break;
            }
        }
        _ = self.total_count.fetchAdd(1, .monotonic);
        // sum 的 CAS 循环(简化版,生产需更严谨的 floating-point CAS)
        var current = self.sum.load(.monotonic);
        while (true) {
            const new_val = current + value;
            if (self.sum.cmpxchgWeak(current, new_val, .monotonic, .monotonic)) |old| {
                current = old;
            } else break;
        }
    }

    pub fn render(self: *Histogram, writer: anytype) !void {
        for (self.buckets, 0..) |bucket, i| {
            try writer.print("{s}_bucket{{le=\"{d}\"}} {d}\n", .{ self.name, bucket, self.counts[i].load(.monotonic) });
        }
        try writer.print("{s}_sum {d}\n", .{ self.name, self.sum.load(.monotonic) });
        try writer.print("{s}_count {d}\n", .{ self.name, self.total_count.load(.monotonic) });
    }
};

6.4 完整 /metrics 端点

// HTTP 处理器 —— Prometheus scrape endpoint
pub fn metricsHandler(writer: anytype, registry: *PrometheusRegistry) !void {
    try writer.writeAll("HTTP/1.1 200 OK\r\nContent-Type: text/plain\r\n\r\n");
    try registry.render(writer);
}

7. 速查表

需求实现方式
基础日志std.log.info/debug/warn/err
自定义格式覆盖 std.log.defaultLog 或 scoped
结构化日志std.json.stringify + 自定义 JsonLogger
日志轮转RotatingLog 按大小切割、保留 N 份
分布式追踪自建 Span + traceparent HTTP 头传播
OTLP 导出HTTP POST /v1/traces 发送 JSON
Prometheus 指标暴露 /metrics HTTP 端点
Counterstd.atomic.Value(u64).fetchAdd
Gaugestd.atomic.Value(i64).store/add
Histogrambucket 数组 + 原子计数

8. 一句话记忆

std.log scoped 分级输出 → JsonLogger 结构化 → RotatingLog 按大小轮转 → Span trace_id+span_id 跨服务传播 → Prometheus /metrics 暴露 counter/gauge/histogram 原子指标 → 可观测三支柱 Logs/Traces/Metrics 在 Zig 中全可自建零依赖。


相关阅读

  • /zig-http-server/ — HTTP 端点服务与路由
  • /zig-json-serialization/ — JSON 序列化与结构化数据
  • /zig-concurrency-atomics/ — 原子操作与并发安全

延伸阅读

  • /zig-debugging-profiling/ — 日志分级与崩溃回溯
  • /zig-crypto-security/ — 安全日志与审计
  • /zig-performance-optimization/ — 遥测开销与性能基准
  • [[zig]] — Zig 系统编程专题

// 完整示例:JSON 结构化日志 + Prometheus Counter/Gauge + /metrics 端点

const std = @import("std");

// ===== 结构化 JSON 日志 =====
pub fn logJson(writer: anytype, level: []const u8, msg: []const u8, fields: anytype) !void {
    try writer.print("{{\"timestamp\":{d},\"level\":\"{s}\",\"message\":\"{s}\"", .{
        std.time.timestamp(), level, msg,
    });
    const fi = @typeInfo(@TypeOf(fields)).Struct;
    inline for (fi.fields) |f| {
        const v = @field(fields, f.name);
        switch (@typeInfo(@TypeOf(v))) {
            .int, .comptime_int => try writer.print(",\"{s}\":{d}", .{f.name, v}),
            else => try writer.print(",\"{s}\":\"{s}\"", .{f.name, v}),
        }
    }
    try writer.writeAll("}\n");
}

// ===== 原子指标 =====
pub const Counter = struct {
    name: []const u8,
    value: std.atomic.Value(u64),
    pub fn inc(self: *Counter, n: u64) void { _ = self.value.fetchAdd(n, .monotonic); }
    pub fn get(self: *Counter) u64 { return self.value.load(.monotonic); }
    pub fn render(self: *Counter, writer: anytype) !void {
        try writer.print("# HELP {s} counter\n# TYPE {s} counter\n{s} {d}\n", .{self.name, self.name, self.name, self.get()});
    }
};

pub const Gauge = struct {
    name: []const u8,
    value: std.atomic.Value(i64),
    pub fn set(self: *Gauge, v: i64) void { self.value.store(v, .monotonic); }
    pub fn render(self: *Gauge, writer: anytype) !void {
        try writer.print("# HELP {s} gauge\n# TYPE {s} gauge\n{s} {d}\n", .{self.name, self.name, self.name, self.value.load(.monotonic)});
    }
};

// ===== /metrics 端点 =====
pub fn metricsHandler(conn: std.net.Stream, req_total: *Counter, active: *Gauge) !void {
    const writer = conn.writer();
    try writer.writeAll("HTTP/1.1 200 OK\r\nContent-Type: text/plain\r\n\r\n");
    try req_total.render(writer);
    try active.render(writer);
}

// ===== 示例服务 =====
pub fn main() !void {
    var req_total = Counter{ .name = "http_requests_total", .value = std.atomic.Value(u64).init(0) };
    var active_conns = Gauge{ .name = "active_connections", .value = std.atomic.Value(i64).init(0) };

    const addr = try std.net.Address.parseIp4("127.0.0.1", 9100);
    var srv = try addr.listen(.{ .reuse_address = true });
    std.debug.print("Metrics on http://127.0.0.1:9100/metrics\n", .{});

    while (true) {
        const conn = try srv.accept();
        req_total.inc(1);
        active_conns.set(active_conns.value.load(.monotonic) + 1);
        defer active_conns.set(active_conns.value.load(.monotonic) - 1);

        var buf: [1024]u8 = undefined;
        _ = try conn.stream.read(&buf);
        try metricsHandler(conn.stream, &req_total, &active_conns);
        conn.stream.close();
    }
}

继续阅读

探索更多技术文章

浏览归档,发现更多关于系统设计、工具链和工程实践的内容。

全部文章 返回首页

「系统编程」更多文章

  1. Zig WebSocket 与实时通信:服务端推送与帧解析
  2. Zig 数据库访问与轻量 ORM:SQLite、PostgreSQL 与自定义 SQL
  3. Zig 密码学与安全编程:AES、ChaCha20Poly1305 与 TLS