From 9878fcf0f7f0cac5314cbe3c068b33377c60fadb Mon Sep 17 00:00:00 2001 From: stqfdyr <89149493+stqfdyr@users.noreply.github.com> Date: Sat, 19 Sep 2026 07:01:18 +0800 Subject: [PATCH 1/6] =?UTF-8?q?feat:=20=E6=B5=81=E9=87=8F=E7=BB=9F?= =?UTF-8?q?=E8=AE=A1=E6=94=AF=E6=8C=81=20--iface=20=E6=8C=87=E5=AE=9A?= =?UTF-8?q?=E7=BD=91=E5=8D=A1=EF=BC=8C=E9=9A=A7=E9=81=93=E4=B8=8E=E5=8F=A0?= =?UTF-8?q?=E5=8A=A0=E8=AE=BE=E5=A4=87=E6=8C=89=E5=86=85=E6=A0=B8=E4=BA=8B?= =?UTF-8?q?=E5=AE=9E=E8=AF=86=E5=88=AB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 名单补 fwln(PVE 防火墙 veth 的另一端)、ifb、gretap、erspan;pppoe- 归入叠加设备 - 新增 counted_elsewhere():有 lower_* 链接的是叠加设备,无 device 且为三层隧道类型或 DEVTYPE=vxlan/geneve 的是隧道,改过名也认得出 - --iface / MONITOR_IFACE:列出的网卡是唯一答案,内置规则不再生效;-name 从本来会计入的里 去掉;末尾 * 匹配前缀 - 设了 --iface 时拼进 boot_id,改设置后 hub 重新对基线,不把两组读数之差记成流量 - 启动打印计入的网卡;修正 --help 续行缩进 --- README.md | 22 +++- src/collect.rs | 334 ++++++++++++++++++++++++++++++++++++++++++------- src/main.rs | 26 +++- 3 files changed, 327 insertions(+), 55 deletions(-) diff --git a/README.md b/README.md index c0f4369..e05fe0f 100644 --- a/README.md +++ b/README.md @@ -34,6 +34,23 @@ monitor-agent --server https://your-hub --token | `--server` | 必填 | hub 地址,也可用 `MONITOR_SERVER` | | `--token` | 必填 | 节点 token,也可用 `MONITOR_TOKEN` | | `--interval` | 1 | 上报间隔(秒),1–3600 | +| `--iface` | 空 | 流量统计的网卡,逗号分隔,也可用 `MONITOR_IFACE`;见下 | + +### 统计哪些网卡的流量 + +默认规则是同一份线上的字节只数一次:lo、容器与虚拟机网卡、隧道不计;bond、网桥、VLAN、PPPoE +这类叠在别的网卡上的设备也不计,只计它们底下的那块。除了按名字,还按内核给出的链路类型、`DEVTYPE` +和 `lower_*` 链接判断,改过名的隧道和网桥同样认得出。 + +转发流量的机器(软路由、桥接了软路由的宿主机)上,同一个包会经过两块真网卡,哪块面向运营商只有 +使用者知道,这时用 `--iface`: + +- `--iface eth1` 或 `--iface pppoe-wan`:只统计列出的网卡,内置规则不再生效 +- `--iface -vxlan100`:从默认结果里去掉一块,`-` 开头的都是排除,排除优先于列出 +- 末尾的 `*` 匹配前缀,如 `--iface 'enp*'` + +启动时打印一行 `counting traffic on: ...`,列出此刻计入的网卡。设了 `--iface` 时它会拼进上报的 +`boot_id`,改设置后 hub 重新对基线,不会把新旧两组网卡读数之差记成流量。 ## 上报字段 @@ -42,8 +59,9 @@ monitor-agent --server https://your-hub --token - **`Facts`** 连接时上报一次:主机名、系统、内核、架构、虚拟化类型、CPU 型号与核数、内存与磁盘总量、本机 IPv4 / IPv6(每族一个,公网地址优先;IPv6 不取临时地址和已废弃地址) - **`Metrics`** 每 `--interval` 秒上报:CPU、负载、内存、swap、磁盘、网卡收发速率与内核累计计数器、TCP / UDP 连接数、进程数、运行时间 -`net_rx_total` / `net_tx_total` 为内核 lifetime 计数器,原样上报;`boot_id` 取自 -`/proc/sys/kernel/random/boot_id`,是 hub 判定主机重启的唯一依据,**不要删**。 +`net_rx_total` / `net_tx_total` 为所计网卡的内核 lifetime 计数器之和,原样上报;`boot_id` 取自 +`/proc/sys/kernel/random/boot_id`(设了 `--iface` 时后接 `/` 与网卡列表),标明这两个读数在哪段区间内 +可以相减,是 hub 判定计数器重新开始的唯一依据,**不要删**。 连接 hub 时逐个尝试解析出的地址,除最后一个外每个限 5 秒。网卡上只有内网 IPv4(NAT)时先连 hub 的 IPv4:NAT 的公网地址不在网卡上,hub 只有看到一条 IPv4 连接才知道它。 diff --git a/src/collect.rs b/src/collect.rs index 057ccfe..776926a 100644 --- a/src/collect.rs +++ b/src/collect.rs @@ -4,6 +4,7 @@ use std::collections::HashMap; use std::fs; use std::net::{IpAddr, Ipv6Addr}; +use std::path::Path; use std::time::Instant; use serde::Serialize; @@ -13,13 +14,20 @@ use serde::Serialize; /// a second time inside their carrier. /// /// `tailscale` belongs to the tunnel group for the same reason `wg` does: it is -/// WireGuard under another name, on an interface not called `wg0`. +/// WireGuard under another name, on an interface not called `wg0`. `fwln` is +/// the half of a Proxmox firewall's veth pair that `fwpr` does not match, and +/// `ifb` mirrors another interface's ingress for traffic shaping. /// -/// ponytail: a name list, so the next tunnel will be missed as well. The kernel -/// knows the answer -- a real NIC has /sys/class/net//device, a virtual -/// one does not -- but a card wrongly failing that test loses a node's entire -/// traffic, where a missing name only doubles it. Worth the swap only after -/// verification against every card in the fleet. +/// `gretap` and `erspan` are GRE carrying Ethernet; the kernel creates one of +/// each, idle, wherever the GRE module is loaded. +/// +/// ponytail: a name list, so a GRE tap or a VPN's tap under a new name will be +/// missed. No kernel attribute separates one from the veth that is an LXC +/// guest's only link: both are Ethernet with no device and no DEVTYPE, and a +/// guest counted as virtual would report no traffic at all. Other tunnels are +/// recognised by [`counted_elsewhere`] whatever their name; the tunnel entries +/// here also keep their addresses out of [`addresses`]. `--iface` covers +/// whatever neither catches. const SKIP_IFACES: &[&str] = &[ "lo", "docker", @@ -35,6 +43,10 @@ const SKIP_IFACES: &[&str] = &[ "podman", "fwbr", "fwpr", + "fwln", + "ifb", + "gretap", + "erspan", "kube", "cali", "nerdctl", @@ -103,8 +115,13 @@ pub struct Facts { #[derive(Serialize, Debug, Clone, Default, PartialEq)] pub struct Metrics { - /// Identifies this boot. It changes on reboot, which is how the hub knows - /// the kernel's byte counters restarted at zero. + /// Names the span over which `net_rx_total` and `net_tx_total` readings + /// are comparable. The hub only tests it for equality, and a change makes + /// it re-baseline rather than book the difference. It is the kernel's boot + /// id, which changes when the counters restart at zero, with `--iface` + /// appended when set, since a different set of interfaces sums different + /// counters: widening it within one boot would otherwise book the added + /// interfaces' lifetime bytes as traffic. pub boot_id: String, pub uptime: u64, pub cpu: f32, @@ -126,15 +143,86 @@ pub struct Metrics { pub procs: u32, } +/// The traffic filter set by `--iface`: interface names separated by commas, +/// each matching a prefix when it ends in `*`. +/// +/// A plain entry makes the list the whole answer: listed interfaces are counted +/// and nothing else is, whatever the built-in rules say. Only the machine's +/// owner knows which port faces the provider on a router, where a forwarded +/// byte crosses two real NICs, or whether a Proxmox host's `vmbr0` alone should +/// count. An entry starting with `-` removes its matches from what is counted +/// otherwise, which one command can apply across machines whose NICs are named +/// differently. Exclusions win over inclusions. +#[derive(Default)] +pub struct Ifaces { + spec: String, + only: Vec, + skip: Vec, +} + +impl Ifaces { + pub fn parse(spec: &str) -> Result { + let entries: Vec<&str> = spec.split(',').map(str::trim).filter(|e| !e.is_empty()).collect(); + let mut ifaces = Self { spec: entries.join(","), ..Self::default() }; + for entry in entries { + let (list, name) = match entry.strip_prefix('-') { + Some(name) => (&mut ifaces.skip, name), + None => (&mut ifaces.only, entry), + }; + // Rejected rather than left to match nothing: either mistake would + // silently remove interfaces from the totals. + if name.is_empty() || name.contains(char::is_whitespace) { + return Err(format!( + "--iface: {entry:?} is not an interface name; separate names with commas" + )); + } + if name.find('*').is_some_and(|i| i + 1 < name.len()) { + return Err(format!("--iface: {entry:?}: `*` may only end a name")); + } + list.push(name.to_owned()); + } + Ok(ifaces) + } + + fn counts(&self, sys: &Path, name: &str) -> bool { + let matches = |p: &String| p.strip_suffix('*').map_or(name == p, |prefix| name.starts_with(prefix)); + if self.skip.iter().any(matches) { + return false; + } + if !self.only.is_empty() { + return self.only.iter().any(matches); + } + !skip_iface(name) && !is_stacked(name) && !counted_elsewhere(sys, name) + } + + /// See [`Metrics::boot_id`]. + fn epoch(&self, boot_id: String) -> String { + if self.spec.is_empty() { + boot_id + } else { + format!("{boot_id}/{}", self.spec) + } + } +} + #[derive(Default)] pub struct Collector { + ifaces: Ifaces, prev_cpu: Option<(u64, u64)>, prev_net: Option<(Instant, u64, u64)>, } impl Collector { - pub fn new() -> Self { - Self::default() + pub fn new(ifaces: Ifaces) -> Self { + Self { ifaces, ..Self::default() } + } + + /// The interfaces the traffic totals include at this moment. + pub fn counted_ifaces(&self) -> Vec { + net_dev(&fs::read_to_string("/proc/net/dev").unwrap_or_default()) + .filter(|(name, ..)| self.ifaces.counts(Path::new(SYS_NET), name)) + .map(|(name, ..)| name.to_owned()) + .collect() } pub fn facts(&self) -> Facts { @@ -164,12 +252,15 @@ impl Collector { let (mem_total, mem_used) = mem_used(&mem); let (swap_total, swap_used) = swap_used(&mem); let (disk_total, disk_used) = disk_usage(&real_mount_points()); - let (rx_total, tx_total) = net_totals(); + let (rx_total, tx_total) = + parse_net_dev(&fs::read_to_string("/proc/net/dev").unwrap_or_default(), |name| { + self.ifaces.counts(Path::new(SYS_NET), name) + }); let (rx, tx) = self.net_rate(rx_total, tx_total, Instant::now()); let (tcp, udp) = conn_counts(); Metrics { - boot_id: read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default(), + boot_id: self.ifaces.epoch(read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default()), uptime: uptime(), cpu: self.cpu_percent(), load: loadavg(), @@ -390,43 +481,88 @@ pub fn is_public(ip: IpAddr) -> bool { } } -/// Sums the kernel's lifetime byte counters, one count per byte on the wire. -fn net_totals() -> (u64, u64) { - parse_net_dev(&fs::read_to_string("/proc/net/dev").unwrap_or_default()) +/// Sums the kernel's lifetime byte counters over the interfaces `counts` +/// accepts, one count per byte on the wire. +fn parse_net_dev(text: &str, counts: impl Fn(&str) -> bool) -> (u64, u64) { + net_dev(text) + .filter(|(name, ..)| counts(name)) + .fold((0, 0), |(rx, tx), (_, r, t)| (rx.saturating_add(r), tx.saturating_add(t))) } -fn parse_net_dev(text: &str) -> (u64, u64) { - let mut rx = 0u64; - let mut tx = 0u64; - for line in text.lines().skip(2) { - let Some((name, rest)) = line.split_once(':') else { continue }; - let name = name.trim(); - if skip_iface(name) || is_stacked(name) { - continue; - } - let f: Vec = rest.split_whitespace().filter_map(|v| v.parse().ok()).collect(); - if f.len() >= 9 { - rx = rx.saturating_add(f[0]); - tx = tx.saturating_add(f[8]); - } - } - (rx, tx) +/// `(name, rx bytes, tx bytes)` for each interface in /proc/net/dev. +fn net_dev(text: &str) -> impl Iterator { + text.lines().skip(2).filter_map(|line| { + let (name, rest) = line.split_once(':')?; + let mut f = rest.split_whitespace().map(|v| v.parse::().ok()); + let rx = f.next()??; + let tx = f.nth(7)??; + Some((name.trim(), rx, tx)) + }) } fn skip_iface(name: &str) -> bool { SKIP_IFACES.iter().any(|p| name.starts_with(p)) } -/// Stacked on top of another interface: bonds, bridges, VLAN children. The -/// kernel books one packet on both, so counting these would double the traffic -/// the hub bills. +/// Stacked on top of another interface: bonds, bridges, VLAN children, and +/// OpenWrt's `pppoe-wan` over its WAN port. The kernel books one packet on +/// both, so counting these would double the traffic the hub bills. `pppoe-` +/// names PPPoE alone: a bare `ppp0` may be an LTE modem's only link. +/// +/// [`counted_elsewhere`] finds the same from the kernel's links under any name. +/// The names still hold where sysfs cannot be read, and PPPoE has no such link +/// to the port it runs over. /// /// A traffic rule only. These interfaces are where a host's own address most -/// often sits -- `vmbr0` on Proxmox, `bond0` where two ports form one link -- -/// and whether bytes were already counted says nothing about address -/// ownership. +/// often sits -- `vmbr0` on Proxmox, `bond0` where two ports form one link, +/// `pppoe-wan` on a router -- and whether bytes were already counted says +/// nothing about address ownership. fn is_stacked(name: &str) -> bool { - name.contains('.') || ["bond", "br", "vlan", "vmbr"].iter().any(|p| name.starts_with(p)) + name.contains('.') || ["bond", "br", "vlan", "vmbr", "pppoe-"].iter().any(|p| name.starts_with(p)) +} + +const SYS_NET: &str = "/sys/class/net"; + +/// Link types of layer-3 tunnels as /sys/class/net//type prints them: +/// none (tun, WireGuard, Tailscale), ipip, ip6tnl, sit, gre, ip6gre. Read off +/// devices of each kind created on a 6.1 kernel. OpenVZ's `venet0`, a +/// container's only link, is void (65535) and stays out. +const TUNNEL_TYPES: &[&str] = &["65534", "768", "769", "776", "778", "823"]; + +/// Layer-2 tunnels that name themselves in `uevent`. GRE taps set no DEVTYPE +/// and are left to [`SKIP_IFACES`]. +const TUNNEL_DEVTYPES: &[&str] = &["DEVTYPE=vxlan", "DEVTYPE=geneve"]; + +/// Whether the kernel shows this interface's bytes counted on another one, +/// whatever it is called: +/// +/// - a `lower_*` link names a device beneath it in this namespace: a bridge +/// with ports, a bond, a VLAN, a macvlan, a DSA switch port over its conduit. +/// The link is absent once a device moves to another namespace, so a +/// container whose only link is a macvlan still counts it. +/// - no hardware behind it, and a tunnel's link type or DEVTYPE: `he-ipv6`, a +/// mesh VPN or `cilium_vxlan` is caught as surely as `wg0`, since its payload +/// leaves again inside a packet the carrier counts. Hardware exempts an LTE +/// modem in raw-IP mode, which shares type none with WireGuard. +/// +/// A traffic rule only, like [`is_stacked`]: a tunnel broker's prefix on +/// `he-ipv6` is this machine's address. Unreadable sysfs leaves the name rules +/// alone in force. +fn counted_elsewhere(sys: &Path, name: &str) -> bool { + let dev = sys.join(name); + // Before the hardware test: a DSA switch port has both. + let stacked = fs::read_dir(&dev).is_ok_and(|mut entries| { + entries.any(|e| e.is_ok_and(|e| e.file_name().as_encoded_bytes().starts_with(b"lower_"))) + }); + if stacked { + return true; + } + if dev.join("device").exists() { + return false; + } + let read = |f: &str| fs::read_to_string(dev.join(f)).unwrap_or_default(); + TUNNEL_TYPES.contains(&read("type").trim()) + || read("uevent").lines().any(|l| TUNNEL_DEVTYPES.contains(&l)) } /// A pseudo filesystem, named outright or as a flavour of one such as @@ -676,7 +812,7 @@ mod tests { // A counter that moved backwards indicates a reboot, not 100% busy. assert_eq!(busy_percent((1000, 925), (500, 400)), 0.0); // The first call has no baseline, so it reports 0. - assert_eq!(Collector::new().cpu_percent(), 0.0); + assert_eq!(Collector::default().cpu_percent(), 0.0); } #[test] @@ -690,8 +826,10 @@ mod tests { } /// One byte on the wire, counted once. Every line but eth0 is that same - /// byte booked a second time: bond, bridge and VLAN are stacked over it, - /// and a tunnel's payload leaves inside a packet eth0 has already counted. + /// byte booked a second time: bond, bridge, VLAN and PPPoE are stacked over + /// it, a tunnel's payload leaves inside a packet eth0 has already counted, + /// `fwln` carries a Proxmox guest's traffic on its way to eth0, and `ifb` + /// mirrors eth0's ingress. /// /// `tailscale0` is listed because it is the same tunnel as `wg0` under a /// different name. @@ -706,12 +844,16 @@ mod tests { tailscale0: 300 1 0 0 0 0 0 0 400 2 0 0 0 0 0 0\n\ tun0: 300 1 0 0 0 0 0 0 400 2 0 0 0 0 0 0\n\ tap0: 300 1 0 0 0 0 0 0 400 2 0 0 0 0 0 0\n\ + fwln100i0: 300 1 0 0 0 0 0 0 400 2 0 0 0 0 0 0\n\ + ifb4eth0: 1000 1 0 0 0 0 0 0 1000 2 0 0 0 0 0 0\n\ bond0: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ br0: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ vmbr0: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ eth0.100: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ - vlan100: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n"; - assert_eq!(parse_net_dev(dev), (1000, 2000)); + vlan100: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ + pppoe-wan: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n"; + let ifaces = Ifaces::default(); + assert_eq!(parse_net_dev(dev, |n| ifaces.counts(Path::new("/nonexistent"), n)), (1000, 2000)); } /// The two questions asked of an interface name, and why one list cannot @@ -719,18 +861,111 @@ mod tests { /// may be this machine's only address. #[test] fn a_stacked_device_loses_its_bytes_but_keeps_its_address() { - for name in ["bond0", "br0", "vmbr0", "eth0.100", "vlan100"] { + for name in ["bond0", "br0", "vmbr0", "eth0.100", "vlan100", "pppoe-wan"] { assert!(is_stacked(name), "{name}: the lower device already counted these bytes"); assert!(!skip_iface(name), "{name} is where a host address lives"); } // Neither this machine's traffic nor its address: container networks, // and tunnels whose payload leaves inside a packet eth0 has counted. - for name in ["lo", "docker0", "veth9a1b2c", "br-6cd9538131d7", "virbr0", "wg0", "tun0", "tap0"] { + for name in [ + "lo", + "docker0", + "veth9a1b2c", + "br-6cd9538131d7", + "virbr0", + "wg0", + "tun0", + "tap0", + "fwln100i0", + "ifb4eth0", + "gretap0", + "erspan0", + ] { assert!(skip_iface(name), "{name} is not this machine"); } assert!(!skip_iface("eth0") && !is_stacked("eth0"), "the wire itself is what gets counted"); } + /// The kernel's account of an interface decides, whatever it is called. + /// Each entry mirrors what /sys/class/net held for that kind on a 6.1 + /// kernel: link type, the `device` link of hardware, the `lower_` link of a + /// stacked device, DEVTYPE in `uevent`. The exempt ones are links a machine + /// depends on that resemble a copy: a raw-IP LTE modem shares WireGuard's + /// type none, OpenVZ's venet0 has no device, and a macvlan moved into a + /// container loses its `lower_` link there. + #[test] + fn the_kernel_tells_a_copy_whatever_the_interface_is_called() { + let sys = std::env::temp_dir().join(format!("monitor-agent-sys-{}", std::process::id())); + for (name, ty, extra) in [ + ("he-ipv6", 776, &[][..]), + ("nebula1", 65534, &[]), + ("gre1", 778, &[]), + ("vx100", 1, &["DEVTYPE=vxlan"]), + ("lan", 1, &["lower_eth0"]), + ("wan", 1, &["device", "lower_eth0"]), + ("wwan0", 65534, &["device"]), + ("venet0", 65535, &[]), + ("mv0", 1, &[]), + ("eth0", 1, &["device"]), + ] { + let dir = sys.join(name); + fs::create_dir_all(&dir).unwrap(); + fs::write(dir.join("type"), format!("{ty}\n")).unwrap(); + for e in extra { + match e.strip_prefix("DEVTYPE=") { + Some(_) => fs::write(dir.join("uevent"), format!("{e}\nINTERFACE={name}\n")).unwrap(), + None => fs::create_dir(dir.join(e)).unwrap(), + } + } + } + // Tunnels by link type and DEVTYPE; a user-named bridge and a DSA + // switch port by their link to the device beneath. + for name in ["he-ipv6", "nebula1", "gre1", "vx100", "lan", "wan"] { + assert!(counted_elsewhere(&sys, name), "{name}: another interface counts these bytes"); + } + for name in ["wwan0", "venet0", "mv0", "eth0", "absent0"] { + assert!(!counted_elsewhere(&sys, name), "{name} is this machine's own link"); + } + fs::remove_dir_all(&sys).unwrap(); + } + + /// A router forwards each byte across two real NICs, so only its owner can + /// name the one facing the provider. What `--iface` lists is counted and + /// nothing else, whatever the built-in rules say; `-` entries come off the + /// top of either. + #[test] + fn iface_names_what_is_counted_over_every_built_in_rule() { + // 1000 bytes downloaded through the router: in on the WAN port inside + // PPPoE, out through the LAN port. + let dev = "header\nheader\n\ + eth0: 50 1 0 0 0 0 0 0 1000 2 0 0 0 0 0 0\n\ + eth1: 1008 1 0 0 0 0 0 0 60 2 0 0 0 0 0 0\n\ + pppoe-wan: 1000 1 0 0 0 0 0 0 52 2 0 0 0 0 0 0\n\ + vmbr0: 7 1 0 0 0 0 0 0 9 2 0 0 0 0 0 0\n"; + let sum = |spec: &str| { + let ifaces = Ifaces::parse(spec).unwrap(); + parse_net_dev(dev, |n| ifaces.counts(Path::new("/nonexistent"), n)) + }; + assert_eq!(sum(""), (1058, 1060), "by default both real ports count the forwarded bytes"); + assert_eq!(sum("pppoe-wan"), (1000, 52), "a listed interface counts though it is stacked"); + assert_eq!(sum("vmbr0"), (7, 9)); + assert_eq!(sum("-eth0"), (1008, 60), "an exclusion comes off the default set"); + assert_eq!(sum("eth*,-eth0"), (1008, 60), "and off a list, which matches prefixes too"); + assert_eq!(sum("eth9"), (0, 0), "an absent interface counts nothing rather than everything"); + + for bad in ["eth0 eth1", "e*h0", "-", "eth0,-"] { + assert!(Ifaces::parse(bad).is_err(), "{bad:?} would silently count nothing"); + } + + // A different set within one boot must make the hub re-baseline rather + // than book the difference between two sums as traffic. + let epoch = |spec: &str| Ifaces::parse(spec).unwrap().epoch("boot".into()); + assert_eq!(epoch(""), "boot", "without --iface the kernel's boot id goes out unchanged"); + assert_ne!(epoch("eth1"), epoch("")); + assert_ne!(epoch("eth1"), epoch("eth0")); + assert_eq!(epoch(" eth1 , "), epoch("eth1"), "spacing alone is not a different set"); + } + #[test] fn the_reported_address_is_the_public_one_whatever_the_kernel_lists_first() { let ip = |s: &str| s.parse::().unwrap(); @@ -807,7 +1042,7 @@ mod tests { #[test] fn net_rate_is_zero_on_first_sample_and_after_a_reboot() { - let mut c = Collector::new(); + let mut c = Collector::default(); let t0 = Instant::now(); assert_eq!(c.net_rate(1000, 2000, t0), (0, 0)); let t1 = t0 + std::time::Duration::from_secs(2); @@ -871,7 +1106,7 @@ mod tests { #[test] fn real_host_collection_is_sane() { - let mut c = Collector::new(); + let mut c = Collector::default(); let f = c.facts(); assert!(!f.hostname.is_empty() && f.cpu_cores >= 1 && f.mem_total > 0); // Whatever this host reports must parse, and a virtual bridge must not @@ -883,6 +1118,9 @@ mod tests { assert!(!f.ipv4.starts_with("172.17."), "a virtual bridge is not this machine's address"); let m = c.collect(); assert!(!m.boot_id.is_empty(), "boot_id drives reboot detection"); + // Read through the real /sys: a link-type test misfiring on this host's + // NIC would leave nothing counted. + assert!(!c.counted_ifaces().is_empty(), "a reachable host counts at least one interface"); assert!(m.mem_used > 0 && m.mem_used < m.mem_total); assert!(m.disk_used <= m.disk_total && m.disk_total > 0); assert!((0.0..=100.0).contains(&m.cpu)); @@ -907,7 +1145,7 @@ mod crosscheck { /// Values are printed, so `cargo test crosscheck -- --nocapture` shows them. #[test] fn memory_and_disk_agree_with_free_and_df_on_this_machine() { - let mut c = Collector::new(); + let mut c = Collector::default(); let m = c.collect(); let gib = |b: u64| b as f64 / 1024.0 / 1024.0 / 1024.0; println!("mem used={:.2}G total={:.2}G", gib(m.mem_used), gib(m.mem_total)); diff --git a/src/main.rs b/src/main.rs index 3bf1457..df3be91 100644 --- a/src/main.rs +++ b/src/main.rs @@ -24,6 +24,7 @@ struct Args { server: String, token: String, interval: u64, + ifaces: collect::Ifaces, /// Permits plain HTTP to a hub reached at ip:port with no TLS in front. /// Off by default: the token would otherwise travel in the clear. insecure: bool, @@ -37,8 +38,12 @@ fn usage() -> ! { --server Hub base URL, e.g. https://hub.example.com\n \ --token Node token from the hub panel\n \ --interval Report interval (default 1)\n \ - --insecure Allow plain ws:// to a remote hub; the token\n \ - travels in the clear. Only for a hub reached\n \ + --iface Count traffic on these interfaces alone, e.g.\n \ + eth1,pppoe-wan. `-name` removes an interface\n \ + from what would be counted; a trailing *\n \ + matches a prefix.\n \ + --insecure Allow plain ws:// to a remote hub; the token\n \ + travels in the clear. Only for a hub reached\n \ at ip:port with no TLS in front.\n", env!("CARGO_PKG_VERSION") ); @@ -46,7 +51,7 @@ fn usage() -> ! { } fn parse_args() -> Result { - let (mut server, mut token, mut interval, mut insecure) = (None, None, 1u64, false); + let (mut server, mut token, mut interval, mut iface, mut insecure) = (None, None, 1u64, None, false); let mut it = std::env::args().skip(1); while let Some(arg) = it.next() { let mut value = || it.next().unwrap_or_else(|| usage()); @@ -54,6 +59,7 @@ fn parse_args() -> Result { "--server" => server = Some(value()), "--token" => token = Some(value()), "--interval" => interval = value().parse().unwrap_or_else(|_| usage()), + "--iface" => iface = Some(value()), "--insecure" => insecure = true, "-h" | "--help" => usage(), other => bail!("unknown argument: {other}"), @@ -61,7 +67,9 @@ fn parse_args() -> Result { } let server = server.or_else(|| std::env::var("MONITOR_SERVER").ok()).unwrap_or_else(|| usage()); let token = token.or_else(|| std::env::var("MONITOR_TOKEN").ok()).unwrap_or_else(|| usage()); - Ok(Args { server, token, interval: interval.clamp(1, 3600), insecure }) + let iface = iface.or_else(|| std::env::var("MONITOR_IFACE").ok()).unwrap_or_default(); + let ifaces = collect::Ifaces::parse(&iface).map_err(anyhow::Error::msg)?; + Ok(Args { server, token, interval: interval.clamp(1, 3600), ifaces, insecure }) } /// `https://host/path` -> `wss://host/path/api/agent/ws`. The token travels in @@ -178,7 +186,15 @@ async fn main() -> Result<()> { for mount in collect::shadowed_mounts(&std::fs::read_to_string("/proc/self/mounts").unwrap_or_default()) { eprintln!("{mount} is covered by another mount and is not counted toward disk totals"); } - let mut collector = Collector::new(); + let mut collector = Collector::new(args.ifaces); + // Reported once at startup: an interface summed twice otherwise shows only + // as a total twice the real one. One listed in --iface but absent here, + // such as a PPPoE link not yet dialled, is counted once it appears. + let counted = collector.counted_ifaces(); + eprintln!( + "counting traffic on: {}", + if counted.is_empty() { "none".to_owned() } else { counted.join(" ") } + ); let mut wait = 0u64; loop { From 67240871f5b8b96da1bad3c324568c15cb395727 Mon Sep 17 00:00:00 2001 From: stqfdyr <89149493+stqfdyr@users.noreply.github.com> Date: Sat, 19 Sep 2026 11:00:57 +0800 Subject: [PATCH 2/6] =?UTF-8?q?fix:=20=E6=89=80=E8=AE=A1=E7=BD=91=E5=8D=A1?= =?UTF-8?q?=E9=9B=86=E5=90=88=E4=B8=80=E5=8F=98=E5=B0=B1=E9=87=8D=E6=96=B0?= =?UTF-8?q?=E5=AF=B9=E5=9F=BA=E7=BA=BF=EF=BC=9B=E7=BD=91=E6=A1=A5=E6=8C=89?= =?UTF-8?q?=20DEVTYPE=20=E5=88=A4=E5=AE=9A=EF=BC=9B--iface=20=E5=89=8D?= =?UTF-8?q?=E7=BC=80=E5=8F=AA=E5=9C=A8=E9=BB=98=E8=AE=A4=E8=A7=84=E5=88=99?= =?UTF-8?q?=E5=86=85=E6=8C=91?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - boot_id 后接所计网卡集合的摘要(原为 --iface 文本):LXD 网桥失去最后一个端口、网卡改名 进出名单时,hub 不再把它的整机读数记成流量 - 网桥、bond 按 uevent 的 DEVTYPE 判定,与有无端口无关 - 新增 Metrics.iface,供面板显示当前设置,不再从 boot_id 里解析 - --iface 的前缀只在默认会计入的网卡里挑,不带上 VLAN 子接口;拒绝单独的 *、-* 与 -- 开头 - 名单补 lxc、cilium(Cilium 的 pod veth 与宿主设备);测试的临时目录开跑前先清理 --- README.md | 19 +++-- src/collect.rs | 223 ++++++++++++++++++++++++++++++------------------- src/main.rs | 3 +- 3 files changed, 151 insertions(+), 94 deletions(-) diff --git a/README.md b/README.md index e05fe0f..4dd57d4 100644 --- a/README.md +++ b/README.md @@ -38,19 +38,20 @@ monitor-agent --server https://your-hub --token ### 统计哪些网卡的流量 -默认规则是同一份线上的字节只数一次:lo、容器与虚拟机网卡、隧道不计;bond、网桥、VLAN、PPPoE +默认规则是同一份线上的字节只数一次:lo、容器与虚拟机网卡、隧道不计;bond、网桥、VLAN、macvlan 这类叠在别的网卡上的设备也不计,只计它们底下的那块。除了按名字,还按内核给出的链路类型、`DEVTYPE` -和 `lower_*` 链接判断,改过名的隧道和网桥同样认得出。 +和 `lower_*` 链接判断,改过名的隧道和网桥同样认得出。PPPoE 只认 OpenWrt 的 `pppoe-wan`:pppd 拨号的 +`ppp0` 默认照计,因为 LTE 拨号时它是唯一的链路,用 pppd 拨 PPPoE 的机器要用 `--iface` 指定。 转发流量的机器(软路由、桥接了软路由的宿主机)上,同一个包会经过两块真网卡,哪块面向运营商只有 使用者知道,这时用 `--iface`: -- `--iface eth1` 或 `--iface pppoe-wan`:只统计列出的网卡,内置规则不再生效 -- `--iface -vxlan100`:从默认结果里去掉一块,`-` 开头的都是排除,排除优先于列出 -- 末尾的 `*` 匹配前缀,如 `--iface 'enp*'` +- `--iface eth1` 或 `--iface pppoe-wan`:只统计列出的网卡,写全名时内置规则不再生效 +- `--iface -eth0`:从本来会计入的网卡里去掉一块(如软路由的 LAN 口),`-` 开头的都是排除,排除优先于列出 +- 末尾的 `*` 匹配前缀,只在默认会计入的网卡里挑:`--iface 'enp*'` 取物理口,不带上 `enp1s0.100` 这类 VLAN -启动时打印一行 `counting traffic on: ...`,列出此刻计入的网卡。设了 `--iface` 时它会拼进上报的 -`boot_id`,改设置后 hub 重新对基线,不会把新旧两组网卡读数之差记成流量。 +启动时打印一行 `counting traffic on: ...`,列出此刻计入的网卡。所计网卡的集合一旦变化(改了 `--iface`、 +网卡增减或被重新归类),上报的 `boot_id` 也跟着变,hub 重新对基线,不会把新旧两组读数之差记成流量。 ## 上报字段 @@ -60,8 +61,8 @@ monitor-agent --server https://your-hub --token - **`Metrics`** 每 `--interval` 秒上报:CPU、负载、内存、swap、磁盘、网卡收发速率与内核累计计数器、TCP / UDP 连接数、进程数、运行时间 `net_rx_total` / `net_tx_total` 为所计网卡的内核 lifetime 计数器之和,原样上报;`boot_id` 取自 -`/proc/sys/kernel/random/boot_id`(设了 `--iface` 时后接 `/` 与网卡列表),标明这两个读数在哪段区间内 -可以相减,是 hub 判定计数器重新开始的唯一依据,**不要删**。 +`/proc/sys/kernel/random/boot_id`,后接 `/` 与所计网卡集合的摘要,标明这两个读数在哪段区间内可以相减, +是 hub 判定计数器重新开始的唯一依据,**不要删**。`iface` 回报当前的 `--iface`,供面板显示与预填。 连接 hub 时逐个尝试解析出的地址,除最后一个外每个限 5 秒。网卡上只有内网 IPv4(NAT)时先连 hub 的 IPv4:NAT 的公网地址不在网卡上,hub 只有看到一条 IPv4 连接才知道它。 diff --git a/src/collect.rs b/src/collect.rs index 776926a..973ac17 100644 --- a/src/collect.rs +++ b/src/collect.rs @@ -19,7 +19,8 @@ use serde::Serialize; /// `ifb` mirrors another interface's ingress for traffic shaping. /// /// `gretap` and `erspan` are GRE carrying Ethernet; the kernel creates one of -/// each, idle, wherever the GRE module is loaded. +/// each, idle, wherever the GRE module is loaded. `lxc` and `cilium` are +/// Cilium's pod veths and host devices, one veth per pod. /// /// ponytail: a name list, so a GRE tap or a VPN's tap under a new name will be /// missed. No kernel attribute separates one from the veth that is an LXC @@ -50,6 +51,8 @@ const SKIP_IFACES: &[&str] = &[ "kube", "cali", "nerdctl", + "lxc", + "cilium", "zt", ]; @@ -118,11 +121,14 @@ pub struct Metrics { /// Names the span over which `net_rx_total` and `net_tx_total` readings /// are comparable. The hub only tests it for equality, and a change makes /// it re-baseline rather than book the difference. It is the kernel's boot - /// id, which changes when the counters restart at zero, with `--iface` - /// appended when set, since a different set of interfaces sums different - /// counters: widening it within one boot would otherwise book the added - /// interfaces' lifetime bytes as traffic. + /// id, which changes when the counters restart at zero, then `/` and a + /// digest of the interfaces summed: an interface joining the sum within one + /// boot -- a reclassified device, a changed `--iface` -- would otherwise + /// have its lifetime bytes booked as traffic. pub boot_id: String, + /// The `--iface` this agent runs with, empty for the default rules. Shown + /// by the panel, which reinstalls with it. + pub iface: String, pub uptime: u64, pub cpu: f32, pub load: [f32; 3], @@ -146,13 +152,15 @@ pub struct Metrics { /// The traffic filter set by `--iface`: interface names separated by commas, /// each matching a prefix when it ends in `*`. /// -/// A plain entry makes the list the whole answer: listed interfaces are counted -/// and nothing else is, whatever the built-in rules say. Only the machine's -/// owner knows which port faces the provider on a router, where a forwarded -/// byte crosses two real NICs, or whether a Proxmox host's `vmbr0` alone should -/// count. An entry starting with `-` removes its matches from what is counted -/// otherwise, which one command can apply across machines whose NICs are named -/// differently. Exclusions win over inclusions. +/// A plain entry makes the list the whole answer: nothing unlisted is counted. +/// Only the machine's owner knows which port faces the provider on a router, +/// where a forwarded byte crosses two real NICs, or whether a Proxmox host's +/// `vmbr0` alone should count. A full name is counted whatever the built-in +/// rules say, since that is how `vmbr0` or `pppoe-wan` is chosen; a prefix +/// picks only among what those rules count, so `enp*` takes the ports and not +/// their VLAN children. An entry starting with `-` removes its matches from what +/// is counted otherwise, which one command can apply across machines whose NICs +/// are named differently. Exclusions win over inclusions. #[derive(Default)] pub struct Ifaces { spec: String, @@ -169,14 +177,15 @@ impl Ifaces { Some(name) => (&mut ifaces.skip, name), None => (&mut ifaces.only, entry), }; - // Rejected rather than left to match nothing: either mistake would - // silently remove interfaces from the totals. - if name.is_empty() || name.contains(char::is_whitespace) { + // Rejected rather than left to match nothing or everything: each + // would silently change the totals. install.sh and the panel refuse + // the same entries. + if name.is_empty() || name.starts_with('-') || name.contains(char::is_whitespace) { return Err(format!( "--iface: {entry:?} is not an interface name; separate names with commas" )); } - if name.find('*').is_some_and(|i| i + 1 < name.len()) { + if name == "*" || name.find('*').is_some_and(|i| i + 1 < name.len()) { return Err(format!("--iface: {entry:?}: `*` may only end a name")); } list.push(name.to_owned()); @@ -185,24 +194,33 @@ impl Ifaces { } fn counts(&self, sys: &Path, name: &str) -> bool { - let matches = |p: &String| p.strip_suffix('*').map_or(name == p, |prefix| name.starts_with(prefix)); - if self.skip.iter().any(matches) { + let by_default = || !skip_iface(name) && !is_stacked(name) && !counted_elsewhere(sys, name); + if self.skip.iter().any(|p| p.strip_suffix('*').map_or(name == p, |prefix| name.starts_with(prefix))) + { return false; } if !self.only.is_empty() { - return self.only.iter().any(matches); + return self.only.iter().any(|p| match p.strip_suffix('*') { + Some(prefix) => name.starts_with(prefix) && by_default(), + None => name == p, + }); } - !skip_iface(name) && !is_stacked(name) && !counted_elsewhere(sys, name) + by_default() } +} - /// See [`Metrics::boot_id`]. - fn epoch(&self, boot_id: String) -> String { - if self.spec.is_empty() { - boot_id - } else { - format!("{boot_id}/{}", self.spec) - } - } +/// See [`Metrics::boot_id`]. The names are sorted, since /proc/net/dev lists a +/// recreated interface in a new position without the set having changed, and +/// hashed with FNV-1a, whose output no Rust release can alter. +fn epoch<'a>(boot_id: &str, names: impl Iterator) -> String { + let mut names: Vec<&str> = names.collect(); + names.sort_unstable(); + // A newline cannot occur in an interface name, so no two sets join alike. + let digest = names + .join("\n") + .bytes() + .fold(0xcbf2_9ce4_8422_2325u64, |h, b| (h ^ u64::from(b)).wrapping_mul(0x0100_0000_01b3)); + format!("{boot_id}/{digest:016x}") } #[derive(Default)] @@ -219,10 +237,12 @@ impl Collector { /// The interfaces the traffic totals include at this moment. pub fn counted_ifaces(&self) -> Vec { - net_dev(&fs::read_to_string("/proc/net/dev").unwrap_or_default()) - .filter(|(name, ..)| self.ifaces.counts(Path::new(SYS_NET), name)) - .map(|(name, ..)| name.to_owned()) - .collect() + let dev = fs::read_to_string("/proc/net/dev").unwrap_or_default(); + self.counted(&dev).into_iter().map(|(name, ..)| name.to_owned()).collect() + } + + fn counted<'a>(&self, dev: &'a str) -> Vec<(&'a str, u64, u64)> { + net_dev(dev).filter(|(name, ..)| self.ifaces.counts(Path::new(SYS_NET), name)).collect() } pub fn facts(&self) -> Facts { @@ -252,15 +272,18 @@ impl Collector { let (mem_total, mem_used) = mem_used(&mem); let (swap_total, swap_used) = swap_used(&mem); let (disk_total, disk_used) = disk_usage(&real_mount_points()); - let (rx_total, tx_total) = - parse_net_dev(&fs::read_to_string("/proc/net/dev").unwrap_or_default(), |name| { - self.ifaces.counts(Path::new(SYS_NET), name) - }); + let dev = fs::read_to_string("/proc/net/dev").unwrap_or_default(); + let counted = self.counted(&dev); + let (rx_total, tx_total) = totals(&counted); let (rx, tx) = self.net_rate(rx_total, tx_total, Instant::now()); let (tcp, udp) = conn_counts(); Metrics { - boot_id: self.ifaces.epoch(read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default()), + boot_id: epoch( + &read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default(), + counted.iter().map(|(name, ..)| *name), + ), + iface: self.ifaces.spec.clone(), uptime: uptime(), cpu: self.cpu_percent(), load: loadavg(), @@ -481,12 +504,10 @@ pub fn is_public(ip: IpAddr) -> bool { } } -/// Sums the kernel's lifetime byte counters over the interfaces `counts` -/// accepts, one count per byte on the wire. -fn parse_net_dev(text: &str, counts: impl Fn(&str) -> bool) -> (u64, u64) { - net_dev(text) - .filter(|(name, ..)| counts(name)) - .fold((0, 0), |(rx, tx), (_, r, t)| (rx.saturating_add(r), tx.saturating_add(t))) +/// Sums the kernel's lifetime byte counters of the counted interfaces, one +/// count per byte on the wire. +fn totals(counted: &[(&str, u64, u64)]) -> (u64, u64) { + counted.iter().fold((0, 0), |(rx, tx), (_, r, t)| (rx.saturating_add(*r), tx.saturating_add(*t))) } /// `(name, rx bytes, tx bytes)` for each interface in /proc/net/dev. @@ -533,36 +554,49 @@ const TUNNEL_TYPES: &[&str] = &["65534", "768", "769", "776", "778", "823"]; /// and are left to [`SKIP_IFACES`]. const TUNNEL_DEVTYPES: &[&str] = &["DEVTYPE=vxlan", "DEVTYPE=geneve"]; +/// Devices that relay their ports' bytes and carry none of their own. Named in +/// `uevent` whether or not a port is attached, unlike the `lower_*` link, so an +/// LXD bridge whose last container has stopped stays out rather than joining +/// the sum with its lifetime counter. +const STACKED_DEVTYPES: &[&str] = &["DEVTYPE=bridge", "DEVTYPE=bond"]; + /// Whether the kernel shows this interface's bytes counted on another one, /// whatever it is called: /// -/// - a `lower_*` link names a device beneath it in this namespace: a bridge -/// with ports, a bond, a VLAN, a macvlan, a DSA switch port over its conduit. -/// The link is absent once a device moves to another namespace, so a +/// - a bridge or bond by DEVTYPE, or a `lower_*` link naming a device beneath +/// it in this namespace: a VLAN, a macvlan, a DSA switch port over its +/// conduit. The link is absent once a device moves to another namespace, so a /// container whose only link is a macvlan still counts it. /// - no hardware behind it, and a tunnel's link type or DEVTYPE: `he-ipv6`, a -/// mesh VPN or `cilium_vxlan` is caught as surely as `wg0`, since its payload -/// leaves again inside a packet the carrier counts. Hardware exempts an LTE -/// modem in raw-IP mode, which shares type none with WireGuard. +/// mesh VPN or a user-named vxlan is caught as surely as `wg0`, since its +/// payload leaves again inside a packet the carrier counts. Hardware exempts +/// an LTE modem in raw-IP mode, which shares type none with WireGuard. /// /// A traffic rule only, like [`is_stacked`]: a tunnel broker's prefix on /// `he-ipv6` is this machine's address. Unreadable sysfs leaves the name rules /// alone in force. +/// +/// ponytail: read afresh every sample, three or four sysfs calls for each +/// interface the name rules leave standing. That is one or two NICs on most +/// hosts; a hundred such interfaces would cost some 400 calls a second. Cache +/// the answer per ifindex if a host like that turns up. fn counted_elsewhere(sys: &Path, name: &str) -> bool { let dev = sys.join(name); + let read = |f: &str| fs::read_to_string(dev.join(f)).unwrap_or_default(); + let uevent = read("uevent"); + let devtype = |set: &[&str]| uevent.lines().any(|l| set.contains(&l)); // Before the hardware test: a DSA switch port has both. - let stacked = fs::read_dir(&dev).is_ok_and(|mut entries| { - entries.any(|e| e.is_ok_and(|e| e.file_name().as_encoded_bytes().starts_with(b"lower_"))) - }); + let stacked = devtype(STACKED_DEVTYPES) + || fs::read_dir(&dev).is_ok_and(|mut entries| { + entries.any(|e| e.is_ok_and(|e| e.file_name().as_encoded_bytes().starts_with(b"lower_"))) + }); if stacked { return true; } if dev.join("device").exists() { return false; } - let read = |f: &str| fs::read_to_string(dev.join(f)).unwrap_or_default(); - TUNNEL_TYPES.contains(&read("type").trim()) - || read("uevent").lines().any(|l| TUNNEL_DEVTYPES.contains(&l)) + TUNNEL_TYPES.contains(&read("type").trim()) || devtype(TUNNEL_DEVTYPES) } /// A pseudo filesystem, named outright or as a flavour of one such as @@ -825,6 +859,13 @@ mod tests { assert_eq!(parse_sockstat("", ""), (0, 0)); } + /// Totals over constructed /proc/net/dev text, with no sysfs to consult. + fn sum(dev: &str, ifaces: &Ifaces) -> (u64, u64) { + totals( + &net_dev(dev).filter(|(n, ..)| ifaces.counts(Path::new("/nonexistent"), n)).collect::>(), + ) + } + /// One byte on the wire, counted once. Every line but eth0 is that same /// byte booked a second time: bond, bridge, VLAN and PPPoE are stacked over /// it, a tunnel's payload leaves inside a packet eth0 has already counted, @@ -852,8 +893,7 @@ mod tests { eth0.100: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ vlan100: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n\ pppoe-wan: 1000 1 0 0 0 0 0 0 2000 2 0 0 0 0 0 0\n"; - let ifaces = Ifaces::default(); - assert_eq!(parse_net_dev(dev, |n| ifaces.counts(Path::new("/nonexistent"), n)), (1000, 2000)); + assert_eq!(sum(dev, &Ifaces::default()), (1000, 2000)); } /// The two questions asked of an interface name, and why one list cannot @@ -880,6 +920,8 @@ mod tests { "ifb4eth0", "gretap0", "erspan0", + "lxc9f2c1e", + "cilium_host", ] { assert!(skip_iface(name), "{name} is not this machine"); } @@ -889,19 +931,25 @@ mod tests { /// The kernel's account of an interface decides, whatever it is called. /// Each entry mirrors what /sys/class/net held for that kind on a 6.1 /// kernel: link type, the `device` link of hardware, the `lower_` link of a - /// stacked device, DEVTYPE in `uevent`. The exempt ones are links a machine + /// stacked device, DEVTYPE in `uevent`. A bridge is known by its DEVTYPE even + /// after its last port has gone and taken the `lower_` link with it. The + /// exempt ones are links a machine /// depends on that resemble a copy: a raw-IP LTE modem shares WireGuard's /// type none, OpenVZ's venet0 has no device, and a macvlan moved into a /// container loses its `lower_` link there. #[test] fn the_kernel_tells_a_copy_whatever_the_interface_is_called() { let sys = std::env::temp_dir().join(format!("monitor-agent-sys-{}", std::process::id())); + // Left behind by a failed run under a reused PID. + let _ = fs::remove_dir_all(&sys); for (name, ty, extra) in [ ("he-ipv6", 776, &[][..]), ("nebula1", 65534, &[]), ("gre1", 778, &[]), ("vx100", 1, &["DEVTYPE=vxlan"]), ("lan", 1, &["lower_eth0"]), + ("lxdbr0", 1, &["DEVTYPE=bridge"]), + ("uplink", 1, &["DEVTYPE=bond"]), ("wan", 1, &["device", "lower_eth0"]), ("wwan0", 65534, &["device"]), ("venet0", 65535, &[]), @@ -919,8 +967,9 @@ mod tests { } } // Tunnels by link type and DEVTYPE; a user-named bridge and a DSA - // switch port by their link to the device beneath. - for name in ["he-ipv6", "nebula1", "gre1", "vx100", "lan", "wan"] { + // switch port by their link to the device beneath; a bridge with no + // port left and a bond by DEVTYPE. + for name in ["he-ipv6", "nebula1", "gre1", "vx100", "lan", "wan", "lxdbr0", "uplink"] { assert!(counted_elsewhere(&sys, name), "{name}: another interface counts these bytes"); } for name in ["wwan0", "venet0", "mv0", "eth0", "absent0"] { @@ -930,40 +979,46 @@ mod tests { } /// A router forwards each byte across two real NICs, so only its owner can - /// name the one facing the provider. What `--iface` lists is counted and - /// nothing else, whatever the built-in rules say; `-` entries come off the - /// top of either. + /// name the one facing the provider. A full name in `--iface` is counted + /// whatever the built-in rules say; a prefix picks among what they count; + /// `-` entries come off the top of either. #[test] fn iface_names_what_is_counted_over_every_built_in_rule() { // 1000 bytes downloaded through the router: in on the WAN port inside - // PPPoE, out through the LAN port. + // PPPoE, out through the LAN port. eth1.7 is a VLAN on the WAN port. let dev = "header\nheader\n\ eth0: 50 1 0 0 0 0 0 0 1000 2 0 0 0 0 0 0\n\ eth1: 1008 1 0 0 0 0 0 0 60 2 0 0 0 0 0 0\n\ + eth1.7: 500 1 0 0 0 0 0 0 30 2 0 0 0 0 0 0\n\ pppoe-wan: 1000 1 0 0 0 0 0 0 52 2 0 0 0 0 0 0\n\ vmbr0: 7 1 0 0 0 0 0 0 9 2 0 0 0 0 0 0\n"; - let sum = |spec: &str| { - let ifaces = Ifaces::parse(spec).unwrap(); - parse_net_dev(dev, |n| ifaces.counts(Path::new("/nonexistent"), n)) - }; - assert_eq!(sum(""), (1058, 1060), "by default both real ports count the forwarded bytes"); - assert_eq!(sum("pppoe-wan"), (1000, 52), "a listed interface counts though it is stacked"); - assert_eq!(sum("vmbr0"), (7, 9)); - assert_eq!(sum("-eth0"), (1008, 60), "an exclusion comes off the default set"); - assert_eq!(sum("eth*,-eth0"), (1008, 60), "and off a list, which matches prefixes too"); - assert_eq!(sum("eth9"), (0, 0), "an absent interface counts nothing rather than everything"); - - for bad in ["eth0 eth1", "e*h0", "-", "eth0,-"] { - assert!(Ifaces::parse(bad).is_err(), "{bad:?} would silently count nothing"); + let with = |spec: &str| sum(dev, &Ifaces::parse(spec).unwrap()); + assert_eq!(with(""), (1058, 1060), "by default both real ports count the forwarded bytes"); + assert_eq!(with("pppoe-wan"), (1000, 52), "a listed interface counts though it is stacked"); + assert_eq!(with("vmbr0"), (7, 9)); + assert_eq!(with("eth1.7"), (500, 30)); + assert_eq!(with("-eth0"), (1008, 60), "an exclusion comes off the default set"); + // The VLAN shares the prefix, but its bytes are already on eth1. + assert_eq!(with("eth*,-eth0"), (1008, 60), "a prefix picks among what the rules count"); + assert_eq!(with("eth9"), (0, 0), "an absent interface counts nothing rather than everything"); + + // Each would count nothing, everything, or not what it says. + for bad in ["eth0 eth1", "e*h0", "-", "eth0,-", "*", "-*", "--eth0"] { + assert!(Ifaces::parse(bad).is_err(), "{bad:?} must be refused"); } + } - // A different set within one boot must make the hub re-baseline rather - // than book the difference between two sums as traffic. - let epoch = |spec: &str| Ifaces::parse(spec).unwrap().epoch("boot".into()); - assert_eq!(epoch(""), "boot", "without --iface the kernel's boot id goes out unchanged"); - assert_ne!(epoch("eth1"), epoch("")); - assert_ne!(epoch("eth1"), epoch("eth0")); - assert_eq!(epoch(" eth1 , "), epoch("eth1"), "spacing alone is not a different set"); + /// Any change to which interfaces are summed, within one boot, must make the + /// hub re-baseline rather than book the difference between two sums: a + /// device the rules reclassify, a changed `--iface`. + #[test] + fn the_epoch_changes_whenever_the_summed_set_does() { + let e = |names: &[&str]| epoch("boot", names.iter().copied()); + assert!(e(&["eth0"]).starts_with("boot/"), "a reboot still changes it"); + assert_ne!(e(&["eth0"]), e(&["eth0", "lxdbr0"])); + assert_ne!(e(&["eth0"]), e(&["eth1"])); + assert_ne!(e(&["eth0"]), e(&[])); + assert_eq!(e(&["eth1", "eth0"]), e(&["eth0", "eth1"]), "listing order is not a different set"); } #[test] diff --git a/src/main.rs b/src/main.rs index df3be91..c6861cf 100644 --- a/src/main.rs +++ b/src/main.rs @@ -41,7 +41,8 @@ fn usage() -> ! { --iface Count traffic on these interfaces alone, e.g.\n \ eth1,pppoe-wan. `-name` removes an interface\n \ from what would be counted; a trailing *\n \ - matches a prefix.\n \ + picks by prefix among the interfaces counted\n \ + by default.\n \ --insecure Allow plain ws:// to a remote hub; the token\n \ travels in the clear. Only for a hub reached\n \ at ip:port with no TLS in front.\n", From 60e7cc646b86c9301775198a7d4da3628e08658a Mon Sep 17 00:00:00 2001 From: stqfdyr <89149493+stqfdyr@users.noreply.github.com> Date: Sat, 19 Sep 2026 15:37:50 +0800 Subject: [PATCH 3/6] =?UTF-8?q?fix:=20=E6=8C=82=E5=9C=A8=E7=BD=91=E6=A1=A5?= =?UTF-8?q?=E4=B8=8A=E7=9A=84=E8=99=9A=E6=8B=9F=E7=BD=91=E5=8D=A1=E4=B8=8D?= =?UTF-8?q?=E8=AE=A1=E6=B5=81=E9=87=8F=EF=BC=9B=E6=89=80=E8=AE=A1=E7=BD=91?= =?UTF-8?q?=E5=8D=A1=E9=9B=86=E5=90=88=E5=8F=98=E5=8C=96=E6=97=B6=E9=80=9F?= =?UTF-8?q?=E7=8E=87=E4=B8=8D=E5=86=8D=E5=87=BA=E7=8E=B0=E5=B0=96=E5=B3=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 没有硬件、又挂在网桥(或其它 master)下的设备不计:libvirt 的 vnetN 等虚拟机 tap、自定义名字的 容器 veth。虚拟机发出的流量在物理口上已经记过一次。只凭 tun_flags 判 tap 不可取:rootless 容器的 pasta 网络里,唯一的网卡就是一块不挂网桥的 tap。 - 集合变化时 net_rx / net_tx 按新旧两组读数之差计算,会把新加入网卡的全部历史读数当成一秒内的流量, 写进历史图。现在纪元不同就不算速率,与 hub 重新对基线一致。 - epoch 的 ponytail 标注:集合每变一次,hub 丢一个间隔的流量,以及消除它的办法。 --- README.md | 5 ++-- src/collect.rs | 79 ++++++++++++++++++++++++++++++++------------------ 2 files changed, 54 insertions(+), 30 deletions(-) diff --git a/README.md b/README.md index 4dd57d4..8be889c 100644 --- a/README.md +++ b/README.md @@ -39,8 +39,9 @@ monitor-agent --server https://your-hub --token ### 统计哪些网卡的流量 默认规则是同一份线上的字节只数一次:lo、容器与虚拟机网卡、隧道不计;bond、网桥、VLAN、macvlan -这类叠在别的网卡上的设备也不计,只计它们底下的那块。除了按名字,还按内核给出的链路类型、`DEVTYPE` -和 `lower_*` 链接判断,改过名的隧道和网桥同样认得出。PPPoE 只认 OpenWrt 的 `pppoe-wan`:pppd 拨号的 +这类叠在别的网卡上的设备也不计,只计它们底下的那块。除了按名字,还按内核给出的链路类型、`DEVTYPE`、 +`lower_*` 链接和所属网桥判断,改过名的隧道、网桥,以及挂在网桥上的虚拟机 tap(如 libvirt 的 `vnet0`) +和容器 veth 同样认得出。PPPoE 只认 OpenWrt 的 `pppoe-wan`:pppd 拨号的 `ppp0` 默认照计,因为 LTE 拨号时它是唯一的链路,用 pppd 拨 PPPoE 的机器要用 `--iface` 指定。 转发流量的机器(软路由、桥接了软路由的宿主机)上,同一个包会经过两块真网卡,哪块面向运营商只有 diff --git a/src/collect.rs b/src/collect.rs index 973ac17..fd8ac2a 100644 --- a/src/collect.rs +++ b/src/collect.rs @@ -22,13 +22,14 @@ use serde::Serialize; /// each, idle, wherever the GRE module is loaded. `lxc` and `cilium` are /// Cilium's pod veths and host devices, one veth per pod. /// -/// ponytail: a name list, so a GRE tap or a VPN's tap under a new name will be -/// missed. No kernel attribute separates one from the veth that is an LXC -/// guest's only link: both are Ethernet with no device and no DEVTYPE, and a -/// guest counted as virtual would report no traffic at all. Other tunnels are -/// recognised by [`counted_elsewhere`] whatever their name; the tunnel entries -/// here also keep their addresses out of [`addresses`]. `--iface` covers -/// whatever neither catches. +/// ponytail: a name list, so a GRE tap, or a tap or veth that is no bridge's +/// port, under a new name will be missed. Nothing in the kernel separates one +/// from a container's only link -- an LXC guest's veth, the tap of a rootless +/// container's pasta network -- and a guest counted as virtual would report no +/// traffic at all. Bridge ports and other tunnels are recognised by +/// [`counted_elsewhere`] whatever their name; the tunnel entries here also keep +/// their addresses out of [`addresses`]. `--iface` covers whatever neither +/// catches. const SKIP_IFACES: &[&str] = &[ "lo", "docker", @@ -212,6 +213,14 @@ impl Ifaces { /// See [`Metrics::boot_id`]. The names are sorted, since /proc/net/dev lists a /// recreated interface in a new position without the set having changed, and /// hashed with FNV-1a, whose output no Rust release can alter. +/// +/// ponytail: every change of the set costs the hub one interval on every +/// interface, including a freshly created one, whose counter starts at zero and +/// would have added correctly. Where counted interfaces come and go -- pods under +/// names no rule knows, pppN on a VPN server -- each event loses one interval. +/// Per-interface counters in the report, summed by the hub, would remove the +/// loss; an offset kept here would not, as it dies with the process and takes the +/// traffic of the downtime with it. fn epoch<'a>(boot_id: &str, names: impl Iterator) -> String { let mut names: Vec<&str> = names.collect(); names.sort_unstable(); @@ -227,7 +236,7 @@ fn epoch<'a>(boot_id: &str, names: impl Iterator) -> String { pub struct Collector { ifaces: Ifaces, prev_cpu: Option<(u64, u64)>, - prev_net: Option<(Instant, u64, u64)>, + prev_net: Option<(String, Instant, u64, u64)>, } impl Collector { @@ -275,14 +284,15 @@ impl Collector { let dev = fs::read_to_string("/proc/net/dev").unwrap_or_default(); let counted = self.counted(&dev); let (rx_total, tx_total) = totals(&counted); - let (rx, tx) = self.net_rate(rx_total, tx_total, Instant::now()); + let boot_id = epoch( + &read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default(), + counted.iter().map(|(name, ..)| *name), + ); + let (rx, tx) = self.net_rate(&boot_id, rx_total, tx_total, Instant::now()); let (tcp, udp) = conn_counts(); Metrics { - boot_id: epoch( - &read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default(), - counted.iter().map(|(name, ..)| *name), - ), + boot_id, iface: self.ifaces.spec.clone(), uptime: uptime(), cpu: self.cpu_percent(), @@ -314,9 +324,11 @@ impl Collector { pct } - fn net_rate(&mut self, rx: u64, tx: u64, now: Instant) -> (u64, u64) { + fn net_rate(&mut self, epoch: &str, rx: u64, tx: u64, now: Instant) -> (u64, u64) { let rate = match self.prev_net { - Some((t, prx, ptx)) => { + // Sums over different interface sets: an interface joining would + // read as its lifetime counter crossing the wire in one interval. + Some((ref e, t, prx, ptx)) if e == epoch => { let secs = now.saturating_duration_since(t).as_secs_f64(); if secs <= 0.0 { (0, 0) @@ -329,9 +341,9 @@ impl Collector { ) } } - None => (0, 0), + _ => (0, 0), }; - self.prev_net = Some((now, rx, tx)); + self.prev_net = Some((epoch.to_owned(), now, rx, tx)); rate } } @@ -571,6 +583,10 @@ const STACKED_DEVTYPES: &[&str] = &["DEVTYPE=bridge", "DEVTYPE=bond"]; /// mesh VPN or a user-named vxlan is caught as surely as `wg0`, since its /// payload leaves again inside a packet the carrier counts. Hardware exempts /// an LTE modem in raw-IP mode, which shares type none with WireGuard. +/// - no hardware behind it, and a `master`: a VM's tap such as libvirt's +/// `vnet0`, or a container's veth, as a bridge's port. What the guest sends +/// out crosses the physical port as well. A container's own only link is no +/// bridge's port inside the container and stays counted. /// /// A traffic rule only, like [`is_stacked`]: a tunnel broker's prefix on /// `he-ipv6` is this machine's address. Unreadable sysfs leaves the name rules @@ -596,7 +612,7 @@ fn counted_elsewhere(sys: &Path, name: &str) -> bool { if dev.join("device").exists() { return false; } - TUNNEL_TYPES.contains(&read("type").trim()) || devtype(TUNNEL_DEVTYPES) + dev.join("master").exists() || TUNNEL_TYPES.contains(&read("type").trim()) || devtype(TUNNEL_DEVTYPES) } /// A pseudo filesystem, named outright or as a flavour of one such as @@ -933,10 +949,10 @@ mod tests { /// kernel: link type, the `device` link of hardware, the `lower_` link of a /// stacked device, DEVTYPE in `uevent`. A bridge is known by its DEVTYPE even /// after its last port has gone and taken the `lower_` link with it. The - /// exempt ones are links a machine - /// depends on that resemble a copy: a raw-IP LTE modem shares WireGuard's - /// type none, OpenVZ's venet0 has no device, and a macvlan moved into a - /// container loses its `lower_` link there. + /// exempt ones are links a machine depends on that resemble a copy: a raw-IP + /// LTE modem shares WireGuard's type none, OpenVZ's venet0 has no device, a + /// macvlan moved into a container loses its `lower_` link there, and a NIC + /// in a bridge has a `master` as a VM's tap does. #[test] fn the_kernel_tells_a_copy_whatever_the_interface_is_called() { let sys = std::env::temp_dir().join(format!("monitor-agent-sys-{}", std::process::id())); @@ -955,6 +971,8 @@ mod tests { ("venet0", 65535, &[]), ("mv0", 1, &[]), ("eth0", 1, &["device"]), + ("vnet0", 1, &["master"]), + ("eno1", 1, &["device", "master"]), ] { let dir = sys.join(name); fs::create_dir_all(&dir).unwrap(); @@ -968,11 +986,11 @@ mod tests { } // Tunnels by link type and DEVTYPE; a user-named bridge and a DSA // switch port by their link to the device beneath; a bridge with no - // port left and a bond by DEVTYPE. - for name in ["he-ipv6", "nebula1", "gre1", "vx100", "lan", "wan", "lxdbr0", "uplink"] { + // port left and a bond by DEVTYPE; a VM's tap by its bridge. + for name in ["he-ipv6", "nebula1", "gre1", "vx100", "lan", "wan", "lxdbr0", "uplink", "vnet0"] { assert!(counted_elsewhere(&sys, name), "{name}: another interface counts these bytes"); } - for name in ["wwan0", "venet0", "mv0", "eth0", "absent0"] { + for name in ["wwan0", "venet0", "mv0", "eth0", "eno1", "absent0"] { assert!(!counted_elsewhere(&sys, name), "{name} is this machine's own link"); } fs::remove_dir_all(&sys).unwrap(); @@ -1099,12 +1117,17 @@ mod tests { fn net_rate_is_zero_on_first_sample_and_after_a_reboot() { let mut c = Collector::default(); let t0 = Instant::now(); - assert_eq!(c.net_rate(1000, 2000, t0), (0, 0)); + assert_eq!(c.net_rate("a", 1000, 2000, t0), (0, 0)); let t1 = t0 + std::time::Duration::from_secs(2); - assert_eq!(c.net_rate(1200, 2400, t1), (100, 200)); + assert_eq!(c.net_rate("a", 1200, 2400, t1), (100, 200)); // Counter restarted: no negative value, no spurious spike. let t2 = t1 + std::time::Duration::from_secs(2); - assert_eq!(c.net_rate(50, 60, t2), (0, 0)); + assert_eq!(c.net_rate("a", 50, 60, t2), (0, 0)); + // An interface joined with its lifetime counter: not a burst of traffic. + let t3 = t2 + std::time::Duration::from_secs(2); + assert_eq!(c.net_rate("b", 9_000_000, 9_000_000, t3), (0, 0)); + let t4 = t3 + std::time::Duration::from_secs(2); + assert_eq!(c.net_rate("b", 9_000_200, 9_000_400, t4), (100, 200)); } /// Two independent guards reject a mount: its filesystem type, and whether From 87b08a7431bf9221bece4158ac0f9cc29f00f431 Mon Sep 17 00:00:00 2001 From: stqfdyr <89149493+stqfdyr@users.noreply.github.com> Date: Sat, 19 Sep 2026 16:23:09 +0800 Subject: [PATCH 4/6] =?UTF-8?q?fix:=20=E5=8F=AA=E6=8E=92=E9=99=A4=E7=BD=91?= =?UTF-8?q?=E6=A1=A5=E7=AB=AF=E5=8F=A3=EF=BC=9B=E9=80=9F=E7=8E=87=E9=80=90?= =?UTF-8?q?=E7=BD=91=E5=8D=A1=E6=B1=82=E5=B7=AE?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 上一版按 master 判定,范围比说明宽:VRF、Open vSwitch 下的设备也有 master,可能正是这台机器的 上联。改按 brport/ 判定,只认网桥端口,与 README 的「挂在网桥上」一致。 - 所计网卡集合变化那一次不再整体把速率记成 0:速率只存在内存里,逐网卡与自己的上一次读数相减, 只算两次采样里都在的网卡。新加入的网卡下一次采样才计入,计数器重新开始的网卡这一次记 0, 其余网卡照常。总流量仍按 boot_id 由 hub 重新对基线,不变。 --- src/collect.rs | 80 +++++++++++++++++++++++++++----------------------- 1 file changed, 44 insertions(+), 36 deletions(-) diff --git a/src/collect.rs b/src/collect.rs index fd8ac2a..82494a0 100644 --- a/src/collect.rs +++ b/src/collect.rs @@ -236,7 +236,9 @@ fn epoch<'a>(boot_id: &str, names: impl Iterator) -> String { pub struct Collector { ifaces: Ifaces, prev_cpu: Option<(u64, u64)>, - prev_net: Option<(String, Instant, u64, u64)>, + /// When the last sample was taken, and each counted interface's counters. + prev_net_at: Option, + prev_net: HashMap, } impl Collector { @@ -288,7 +290,7 @@ impl Collector { &read_trim("/proc/sys/kernel/random/boot_id").unwrap_or_default(), counted.iter().map(|(name, ..)| *name), ); - let (rx, tx) = self.net_rate(&boot_id, rx_total, tx_total, Instant::now()); + let (rx, tx) = self.net_rate(&counted, Instant::now()); let (tcp, udp) = conn_counts(); Metrics { @@ -324,26 +326,31 @@ impl Collector { pct } - fn net_rate(&mut self, epoch: &str, rx: u64, tx: u64, now: Instant) -> (u64, u64) { - let rate = match self.prev_net { - // Sums over different interface sets: an interface joining would - // read as its lifetime counter crossing the wire in one interval. - Some((ref e, t, prx, ptx)) if e == epoch => { + /// Per interface, over those in both samples: one joining brings a lifetime + /// counter that is not this interval's traffic, and one whose counter + /// restarted moved backwards. Either would otherwise read as a burst in the + /// history. Kept in memory only; a restarted agent reports no rate once. + fn net_rate(&mut self, counted: &[(&str, u64, u64)], now: Instant) -> (u64, u64) { + let rate = match self.prev_net_at { + Some(t) => { let secs = now.saturating_duration_since(t).as_secs_f64(); + let (rx, tx) = counted + .iter() + .filter_map(|(name, rx, tx)| { + let (prx, ptx) = self.prev_net.get(*name)?; + Some((rx.saturating_sub(*prx), tx.saturating_sub(*ptx))) + }) + .fold((0u64, 0u64), |(a, b), (r, t)| (a.saturating_add(r), b.saturating_add(t))); if secs <= 0.0 { (0, 0) } else { - // A counter that moved backwards means a reboot or a wrap: - // report no rate rather than a spurious spike. - ( - (rx.saturating_sub(prx) as f64 / secs) as u64, - (tx.saturating_sub(ptx) as f64 / secs) as u64, - ) + ((rx as f64 / secs) as u64, (tx as f64 / secs) as u64) } } - _ => (0, 0), + None => (0, 0), }; - self.prev_net = Some((epoch.to_owned(), now, rx, tx)); + self.prev_net_at = Some(now); + self.prev_net = counted.iter().map(|(n, r, t)| ((*n).to_owned(), (*r, *t))).collect(); rate } } @@ -583,10 +590,11 @@ const STACKED_DEVTYPES: &[&str] = &["DEVTYPE=bridge", "DEVTYPE=bond"]; /// mesh VPN or a user-named vxlan is caught as surely as `wg0`, since its /// payload leaves again inside a packet the carrier counts. Hardware exempts /// an LTE modem in raw-IP mode, which shares type none with WireGuard. -/// - no hardware behind it, and a `master`: a VM's tap such as libvirt's -/// `vnet0`, or a container's veth, as a bridge's port. What the guest sends -/// out crosses the physical port as well. A container's own only link is no -/// bridge's port inside the container and stays counted. +/// - no hardware behind it, and a bridge's port (`brport/`): a VM's tap such as +/// libvirt's `vnet0`, or a container's veth. What the guest sends out crosses +/// the physical port as well. A container's own only link is no bridge's port +/// inside the container and stays counted, as does a device under another +/// master -- a VRF, Open vSwitch -- whose uplink may be this very device. /// /// A traffic rule only, like [`is_stacked`]: a tunnel broker's prefix on /// `he-ipv6` is this machine's address. Unreadable sysfs leaves the name rules @@ -612,7 +620,7 @@ fn counted_elsewhere(sys: &Path, name: &str) -> bool { if dev.join("device").exists() { return false; } - dev.join("master").exists() || TUNNEL_TYPES.contains(&read("type").trim()) || devtype(TUNNEL_DEVTYPES) + dev.join("brport").exists() || TUNNEL_TYPES.contains(&read("type").trim()) || devtype(TUNNEL_DEVTYPES) } /// A pseudo filesystem, named outright or as a flavour of one such as @@ -951,8 +959,9 @@ mod tests { /// after its last port has gone and taken the `lower_` link with it. The /// exempt ones are links a machine depends on that resemble a copy: a raw-IP /// LTE modem shares WireGuard's type none, OpenVZ's venet0 has no device, a - /// macvlan moved into a container loses its `lower_` link there, and a NIC - /// in a bridge has a `master` as a VM's tap does. + /// macvlan moved into a container loses its `lower_` link there, a NIC in a + /// bridge is a bridge's port as a VM's tap is, and a VRF's member has a + /// master but is no bridge's port. #[test] fn the_kernel_tells_a_copy_whatever_the_interface_is_called() { let sys = std::env::temp_dir().join(format!("monitor-agent-sys-{}", std::process::id())); @@ -971,8 +980,9 @@ mod tests { ("venet0", 65535, &[]), ("mv0", 1, &[]), ("eth0", 1, &["device"]), - ("vnet0", 1, &["master"]), - ("eno1", 1, &["device", "master"]), + ("vnet0", 1, &["brport"]), + ("eno1", 1, &["device", "brport"]), + ("up1", 1, &["master"]), ] { let dir = sys.join(name); fs::create_dir_all(&dir).unwrap(); @@ -990,7 +1000,7 @@ mod tests { for name in ["he-ipv6", "nebula1", "gre1", "vx100", "lan", "wan", "lxdbr0", "uplink", "vnet0"] { assert!(counted_elsewhere(&sys, name), "{name}: another interface counts these bytes"); } - for name in ["wwan0", "venet0", "mv0", "eth0", "eno1", "absent0"] { + for name in ["wwan0", "venet0", "mv0", "eth0", "eno1", "up1", "absent0"] { assert!(!counted_elsewhere(&sys, name), "{name} is this machine's own link"); } fs::remove_dir_all(&sys).unwrap(); @@ -1114,20 +1124,18 @@ mod tests { } #[test] - fn net_rate_is_zero_on_first_sample_and_after_a_reboot() { + fn net_rate_counts_each_interface_against_its_own_last_reading() { let mut c = Collector::default(); let t0 = Instant::now(); - assert_eq!(c.net_rate("a", 1000, 2000, t0), (0, 0)); - let t1 = t0 + std::time::Duration::from_secs(2); - assert_eq!(c.net_rate("a", 1200, 2400, t1), (100, 200)); + let at = |secs| t0 + std::time::Duration::from_secs(secs); + assert_eq!(c.net_rate(&[("eth0", 1000, 2000)], t0), (0, 0)); + assert_eq!(c.net_rate(&[("eth0", 1200, 2400)], at(2)), (100, 200)); // Counter restarted: no negative value, no spurious spike. - let t2 = t1 + std::time::Duration::from_secs(2); - assert_eq!(c.net_rate("a", 50, 60, t2), (0, 0)); - // An interface joined with its lifetime counter: not a burst of traffic. - let t3 = t2 + std::time::Duration::from_secs(2); - assert_eq!(c.net_rate("b", 9_000_000, 9_000_000, t3), (0, 0)); - let t4 = t3 + std::time::Duration::from_secs(2); - assert_eq!(c.net_rate("b", 9_000_200, 9_000_400, t4), (100, 200)); + assert_eq!(c.net_rate(&[("eth0", 50, 60)], at(4)), (0, 0)); + // An interface joined with its lifetime counter: not a burst of + // traffic, and the others keep their rate. + assert_eq!(c.net_rate(&[("eth0", 250, 460), ("eth1", 9_000_000, 9_000_000)], at(6)), (100, 200)); + assert_eq!(c.net_rate(&[("eth0", 450, 860), ("eth1", 9_000_200, 9_000_400)], at(8)), (200, 400)); } /// Two independent guards reject a mount: its filesystem type, and whether From f8db509585d48c6affd8616364de1ab8d7bb3469 Mon Sep 17 00:00:00 2001 From: stqfdyr <89149493+stqfdyr@users.noreply.github.com> Date: Sat, 19 Sep 2026 16:57:38 +0800 Subject: [PATCH 5/6] =?UTF-8?q?refactor:=20--iface=20=E5=8F=AA=E8=AE=A4?= =?UTF-8?q?=E5=AE=8C=E6=95=B4=E7=BD=91=E5=8D=A1=E5=90=8D=EF=BC=8C=E5=8E=BB?= =?UTF-8?q?=E6=8E=89=E6=9C=AB=E5=B0=BE=20*=20=E7=9A=84=E5=89=8D=E7=BC=80?= =?UTF-8?q?=E5=8C=B9=E9=85=8D?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 前缀规则是这套语法里最绕的一条(只在默认会计入的网卡里挑),也让 agent、install.sh、面板三处的 校验最复杂,而有了 `-` 排除之后,基本没有非它不可的场景。现在只有两种写法:列出要统计的网卡, 或在名字前加 `-` 排除。写了 `*` 直接报错退出,不会静默地什么都匹配不到。 --- README.md | 5 +++-- src/collect.rs | 48 +++++++++++++++++++++--------------------------- src/main.rs | 4 +--- 3 files changed, 25 insertions(+), 32 deletions(-) diff --git a/README.md b/README.md index 8be889c..6c91f3d 100644 --- a/README.md +++ b/README.md @@ -48,8 +48,9 @@ monitor-agent --server https://your-hub --token 使用者知道,这时用 `--iface`: - `--iface eth1` 或 `--iface pppoe-wan`:只统计列出的网卡,写全名时内置规则不再生效 -- `--iface -eth0`:从本来会计入的网卡里去掉一块(如软路由的 LAN 口),`-` 开头的都是排除,排除优先于列出 -- 末尾的 `*` 匹配前缀,只在默认会计入的网卡里挑:`--iface 'enp*'` 取物理口,不带上 `enp1s0.100` 这类 VLAN +- `--iface -eth0`:从本来会计入的网卡里去掉一块(如软路由的 LAN 口),`-` 开头的都是排除,排除优先于列出。 + 一批机器都有一块同名的内网网卡时,一条批量命令就能在每台上去掉它 +- 只认完整的网卡名,写 `eth*` 这类通配会直接报错退出 启动时打印一行 `counting traffic on: ...`,列出此刻计入的网卡。所计网卡的集合一旦变化(改了 `--iface`、 网卡增减或被重新归类),上报的 `boot_id` 也跟着变,hub 重新对基线,不会把新旧两组读数之差记成流量。 diff --git a/src/collect.rs b/src/collect.rs index 82494a0..62488fe 100644 --- a/src/collect.rs +++ b/src/collect.rs @@ -150,18 +150,17 @@ pub struct Metrics { pub procs: u32, } -/// The traffic filter set by `--iface`: interface names separated by commas, -/// each matching a prefix when it ends in `*`. +/// The traffic filter set by `--iface`: full interface names separated by +/// commas. /// /// A plain entry makes the list the whole answer: nothing unlisted is counted. /// Only the machine's owner knows which port faces the provider on a router, /// where a forwarded byte crosses two real NICs, or whether a Proxmox host's -/// `vmbr0` alone should count. A full name is counted whatever the built-in -/// rules say, since that is how `vmbr0` or `pppoe-wan` is chosen; a prefix -/// picks only among what those rules count, so `enp*` takes the ports and not -/// their VLAN children. An entry starting with `-` removes its matches from what -/// is counted otherwise, which one command can apply across machines whose NICs -/// are named differently. Exclusions win over inclusions. +/// `vmbr0` alone should count. A listed name is counted whatever the built-in +/// rules say, since that is how `vmbr0` or `pppoe-wan` is chosen. An entry +/// starting with `-` removes that interface from what is counted otherwise, +/// which one batch command can apply across machines whose other NICs are named +/// differently. Exclusions win over inclusions. #[derive(Default)] pub struct Ifaces { spec: String, @@ -181,32 +180,29 @@ impl Ifaces { // Rejected rather than left to match nothing or everything: each // would silently change the totals. install.sh and the panel refuse // the same entries. - if name.is_empty() || name.starts_with('-') || name.contains(char::is_whitespace) { + // `*` included: a wildcard would match no interface, where the + // intent was plainly a set of them. + if name.is_empty() + || name.starts_with('-') + || name.contains(|c: char| c.is_whitespace() || c == '*') + { return Err(format!( - "--iface: {entry:?} is not an interface name; separate names with commas" + "--iface: {entry:?} is not an interface name; give full names separated by commas" )); } - if name == "*" || name.find('*').is_some_and(|i| i + 1 < name.len()) { - return Err(format!("--iface: {entry:?}: `*` may only end a name")); - } list.push(name.to_owned()); } Ok(ifaces) } fn counts(&self, sys: &Path, name: &str) -> bool { - let by_default = || !skip_iface(name) && !is_stacked(name) && !counted_elsewhere(sys, name); - if self.skip.iter().any(|p| p.strip_suffix('*').map_or(name == p, |prefix| name.starts_with(prefix))) - { + if self.skip.iter().any(|n| n == name) { return false; } if !self.only.is_empty() { - return self.only.iter().any(|p| match p.strip_suffix('*') { - Some(prefix) => name.starts_with(prefix) && by_default(), - None => name == p, - }); + return self.only.iter().any(|n| n == name); } - by_default() + !skip_iface(name) && !is_stacked(name) && !counted_elsewhere(sys, name) } } @@ -1007,9 +1003,8 @@ mod tests { } /// A router forwards each byte across two real NICs, so only its owner can - /// name the one facing the provider. A full name in `--iface` is counted - /// whatever the built-in rules say; a prefix picks among what they count; - /// `-` entries come off the top of either. + /// name the one facing the provider. A name in `--iface` is counted whatever + /// the built-in rules say; `-` entries come off the top of either. #[test] fn iface_names_what_is_counted_over_every_built_in_rule() { // 1000 bytes downloaded through the router: in on the WAN port inside @@ -1026,12 +1021,11 @@ mod tests { assert_eq!(with("vmbr0"), (7, 9)); assert_eq!(with("eth1.7"), (500, 30)); assert_eq!(with("-eth0"), (1008, 60), "an exclusion comes off the default set"); - // The VLAN shares the prefix, but its bytes are already on eth1. - assert_eq!(with("eth*,-eth0"), (1008, 60), "a prefix picks among what the rules count"); + assert_eq!(with("eth0,eth1,-eth0"), (1008, 60), "an exclusion wins over a listing"); assert_eq!(with("eth9"), (0, 0), "an absent interface counts nothing rather than everything"); // Each would count nothing, everything, or not what it says. - for bad in ["eth0 eth1", "e*h0", "-", "eth0,-", "*", "-*", "--eth0"] { + for bad in ["eth0 eth1", "eth*", "e*h0", "-", "eth0,-", "*", "-*", "--eth0"] { assert!(Ifaces::parse(bad).is_err(), "{bad:?} must be refused"); } } diff --git a/src/main.rs b/src/main.rs index c6861cf..a887734 100644 --- a/src/main.rs +++ b/src/main.rs @@ -40,9 +40,7 @@ fn usage() -> ! { --interval Report interval (default 1)\n \ --iface Count traffic on these interfaces alone, e.g.\n \ eth1,pppoe-wan. `-name` removes an interface\n \ - from what would be counted; a trailing *\n \ - picks by prefix among the interfaces counted\n \ - by default.\n \ + from what would be counted. Full names only.\n \ --insecure Allow plain ws:// to a remote hub; the token\n \ travels in the clear. Only for a hub reached\n \ at ip:port with no TLS in front.\n", From f33e9f30d688e21b5796a2d8e354ff909bbc8db7 Mon Sep 17 00:00:00 2001 From: stqfdyr <89149493+stqfdyr@users.noreply.github.com> Date: Sat, 19 Sep 2026 20:48:47 +0800 Subject: [PATCH 6/6] =?UTF-8?q?refactor:=20--iface=20=E4=B8=8D=E5=86=8D?= =?UTF-8?q?=E5=8D=95=E7=8B=AC=E6=8B=92=E7=BB=9D=20*?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 通配匹配已经去掉,`*` 没有理由比写错的网卡名多一条规则:两者都匹配不到网卡,启动日志的 `counting traffic on:` 和面板的「当前」都能看出来。install.sh 与面板的字符白名单照样不收 `*`。 --- README.md | 2 +- src/collect.rs | 9 ++------- 2 files changed, 3 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 6c91f3d..a8cb6f0 100644 --- a/README.md +++ b/README.md @@ -50,7 +50,7 @@ monitor-agent --server https://your-hub --token - `--iface eth1` 或 `--iface pppoe-wan`:只统计列出的网卡,写全名时内置规则不再生效 - `--iface -eth0`:从本来会计入的网卡里去掉一块(如软路由的 LAN 口),`-` 开头的都是排除,排除优先于列出。 一批机器都有一块同名的内网网卡时,一条批量命令就能在每台上去掉它 -- 只认完整的网卡名,写 `eth*` 这类通配会直接报错退出 +- 只认完整的网卡名,不支持通配 启动时打印一行 `counting traffic on: ...`,列出此刻计入的网卡。所计网卡的集合一旦变化(改了 `--iface`、 网卡增减或被重新归类),上报的 `boot_id` 也跟着变,hub 重新对基线,不会把新旧两组读数之差记成流量。 diff --git a/src/collect.rs b/src/collect.rs index 62488fe..4bb985d 100644 --- a/src/collect.rs +++ b/src/collect.rs @@ -180,12 +180,7 @@ impl Ifaces { // Rejected rather than left to match nothing or everything: each // would silently change the totals. install.sh and the panel refuse // the same entries. - // `*` included: a wildcard would match no interface, where the - // intent was plainly a set of them. - if name.is_empty() - || name.starts_with('-') - || name.contains(|c: char| c.is_whitespace() || c == '*') - { + if name.is_empty() || name.starts_with('-') || name.contains(char::is_whitespace) { return Err(format!( "--iface: {entry:?} is not an interface name; give full names separated by commas" )); @@ -1025,7 +1020,7 @@ mod tests { assert_eq!(with("eth9"), (0, 0), "an absent interface counts nothing rather than everything"); // Each would count nothing, everything, or not what it says. - for bad in ["eth0 eth1", "eth*", "e*h0", "-", "eth0,-", "*", "-*", "--eth0"] { + for bad in ["eth0 eth1", "-", "eth0,-", "--eth0"] { assert!(Ifaces::parse(bad).is_err(), "{bad:?} must be refused"); } }