From: Hannes Laimer <h.laimer@proxmox.com>
To: pve-devel@lists.proxmox.com
Subject: [PATCH proxmox-ebpf v2 1/3] add the shared tc subsystem code
Date: Fri, 4 Sep 2026 11:04:56 +0200 [thread overview]
Message-ID: <20260904090458.990888-2-h.laimer@proxmox.com> (raw)
In-Reply-To: <20260904090458.990888-1-h.laimer@proxmox.com>
A subsystem owns tc classifier programs and their maps, attached per
interface and pinned in bpffs together with an object fingerprint and
a schema version, so state survives between the one-shot calls from
the consumer driving it and re-running never interrupts traffic.
Apply paths coordinate through a shared flock, loading takes it
exclusively.
Signed-off-by: Hannes Laimer <h.laimer@proxmox.com>
---
src/lib.rs | 3 +
src/subsystem.rs | 476 +++++++++++++++++++++++++++++++++++++++++++++++
src/tc.rs | 187 +++++++++++++++++++
3 files changed, 666 insertions(+)
create mode 100644 src/subsystem.rs
create mode 100644 src/tc.rs
diff --git a/src/lib.rs b/src/lib.rs
index 244f44c..572d861 100644
--- a/src/lib.rs
+++ b/src/lib.rs
@@ -1,2 +1,5 @@
//! eBPF subsystems, one per concern, each behind a cargo feature of its name. Nothing here runs
//! on its own, consumers pull in only the subsystem they drive.
+
+pub mod subsystem;
+pub mod tc;
diff --git a/src/subsystem.rs b/src/subsystem.rs
new file mode 100644
index 0000000..853e17b
--- /dev/null
+++ b/src/subsystem.rs
@@ -0,0 +1,476 @@
+//! The shared subsystem code. A subsystem's tc programs, one per direction it declares, with
+//! their maps and links, pinned under `/sys/fs/bpf/proxmox-ebpf/<name>/<schema>/`, the programs
+//! by the fingerprint of their object, so the pinned state itself says which build it belongs to.
+//! The loaded BPF stays in the kernel between invocations, so [`ApplyLock::ensure_loaded`] loads
+//! and verifies only on the first run and on a version change. Everything else attaches links and
+//! syncs maps against what is already there. Each subsystem owns one [`TcPrograms`].
+
+use std::{collections::HashSet, fs::File, io::ErrorKind, marker::PhantomData, path::PathBuf};
+
+use anyhow::Context;
+use aya::{EbpfLoader, programs::SchedClassifier};
+use nix::fcntl::{Flock, FlockArg};
+
+use crate::tc::{self, Direction};
+
+// throwaway pin roots for the verify-at-load step, one per subsystem
+pub(crate) const VERIFY_ROOT: &str = "/sys/fs/bpf/proxmox-ebpf-test";
+
+const PIN_ROOT: &str = "/sys/fs/bpf/proxmox-ebpf";
+const RUN_ROOT: &str = "/run/proxmox-ebpf";
+
+/// The pinned tc programs for one subsystem.
+///
+/// Holds only the immutable description. The loaded BPF lives in the kernel, pinned, and is
+/// reached back through those pins, so normal operation runs no verifier.
+pub struct TcPrograms {
+ name: &'static str,
+ obj: &'static [u8],
+ fingerprint: u64,
+ prog_name: fn(Direction) -> &'static str,
+ directions: &'static [Direction],
+ schema_version: u32,
+}
+
+impl TcPrograms {
+ pub fn new(
+ name: &'static str,
+ obj: &'static [u8],
+ fingerprint: u64,
+ prog_name: fn(Direction) -> &'static str,
+ directions: &'static [Direction],
+ schema_version: u32,
+ ) -> Self {
+ Self {
+ name,
+ obj,
+ fingerprint,
+ prog_name,
+ directions,
+ schema_version,
+ }
+ }
+
+ fn pin_root(&self) -> PathBuf {
+ PathBuf::from(PIN_ROOT).join(self.name)
+ }
+ fn schema_root(&self) -> PathBuf {
+ self.pin_root().join(self.schema_version.to_string())
+ }
+ fn links_dir(&self) -> PathBuf {
+ self.schema_root().join("links")
+ }
+ fn link_pin_path(&self, ifindex: u32, dir: Direction) -> PathBuf {
+ self.links_dir().join(tc::pin_filename(ifindex, dir))
+ }
+ fn progs_dir(&self) -> PathBuf {
+ self.schema_root().join("prog")
+ }
+ fn fingerprint_dir_name(&self) -> String {
+ format!("{:016x}", self.fingerprint)
+ }
+ fn prog_dir(&self) -> PathBuf {
+ self.progs_dir().join(self.fingerprint_dir_name())
+ }
+ fn prog_pin_path(&self, dir: Direction) -> PathBuf {
+ self.prog_dir().join(dir.as_str())
+ }
+
+ fn hash_map<K: aya::Pod, V: aya::Pod>(
+ &self,
+ name: &str,
+ ) -> anyhow::Result<aya::maps::HashMap<aya::maps::MapData, K, V>> {
+ let map = aya::maps::MapData::from_pin(self.schema_root().join(name))
+ .with_context(|| format!("{}: open pinned map {name}", self.name))?;
+ Ok(aya::maps::HashMap::try_from(aya::maps::Map::HashMap(map))?)
+ }
+
+ /// Open one direction's pinned program. A plain `BPF_OBJ_GET`, no verifier, since the program
+ /// was checked once when [`ensure_loaded`](ApplyLock::ensure_loaded) installed it.
+ fn program(&self, dir: Direction) -> anyhow::Result<SchedClassifier> {
+ SchedClassifier::from_pin(self.prog_pin_path(dir))
+ .with_context(|| format!("{}: open pinned program {}", self.name, dir.as_str()))
+ }
+
+ /// FNV-1a over the embedded object. A `const fn`, so each subsystem folds it into a `const` at
+ /// compile time and the per-invocation path never rehashes a constant.
+ pub const fn obj_fingerprint(obj: &[u8]) -> u64 {
+ let mut hash = 0xcbf29ce484222325u64;
+ let mut i = 0;
+ while i < obj.len() {
+ hash ^= obj[i] as u64;
+ hash = hash.wrapping_mul(0x100000001b3);
+ i += 1;
+ }
+ hash
+ }
+
+ /// The per-subsystem apply lock under `/run`, one file taken in two modes. A full apply (and
+ /// any install/teardown) takes it [exclusively](Self::lock_exclusive) so its enumerate, detach
+ /// and attach run as one unit. An additive single-interface apply takes it
+ /// [shared](Self::lock_shared) so guest plugs run concurrently and only block while a full
+ /// apply holds it. The kernel drops the lock if the process dies. Everything that touches the
+ /// pinned state is a method of the returned guard, so the mode a step needs is checked at
+ /// compile time.
+ fn lock<M>(&self, arg: FlockArg) -> anyhow::Result<ApplyLock<'_, M>> {
+ let dir = PathBuf::from(RUN_ROOT).join(self.name);
+ std::fs::create_dir_all(&dir)?;
+ let file = File::create(dir.join("lock"))?;
+ let lock = Flock::lock(file, arg).map_err(|(_, e)| anyhow::Error::new(e))?;
+ Ok(ApplyLock {
+ programs: self,
+ _lock: lock,
+ _mode: PhantomData,
+ })
+ }
+
+ /// Take the apply lock in shared mode, for an additive single-interface apply.
+ pub fn lock_shared(&self) -> anyhow::Result<ApplyLock<'_, Shared>> {
+ self.lock(FlockArg::LockShared)
+ }
+
+ /// Take the apply lock exclusively, for a full apply or an install/teardown.
+ pub fn lock_exclusive(&self) -> anyhow::Result<ApplyLock<'_, Exclusive>> {
+ self.lock(FlockArg::LockExclusive)
+ }
+
+ fn ensure_loaded(&self) -> anyhow::Result<bool> {
+ if self.is_current() {
+ return Ok(false);
+ }
+ let empty = !self.any_programs_pinned();
+
+ // any other schema pinned here is old state whose map layout this build cannot bind to,
+ // so it goes once the new code verified. A newer one means a newer build already runs
+ // here and this process is the stale one, it must not tear that down
+ let others = self.other_schemas()?;
+ if let Some(newer) = others
+ .iter()
+ .flatten()
+ .copied()
+ .find(|&s| s > self.schema_version)
+ {
+ anyhow::bail!(
+ "{}: pinned state is of schema {newer}, newer than this build's {}, clear it to \
+ downgrade",
+ self.name,
+ self.schema_version
+ );
+ }
+ let schema_changed = !others.is_empty();
+ if schema_changed {
+ log::warn!("{}: pinned state of another schema, rebuilding", self.name);
+ // verify the new code loads against throw-away state before tearing the old down, so a
+ // verifier rejection can't leave us with the old state wiped and nothing to replace it
+ self.verify().with_context(|| {
+ format!(
+ "{}: new BPF code does not load against fresh state",
+ self.name
+ )
+ })?;
+ self.tear_down().context("tear_down")?;
+ }
+
+ let reclaimed = self.install().context("install")?;
+ Ok(empty || schema_changed || reclaimed > 0)
+ }
+
+ /// True when every direction's program is pinned under this build's schema and fingerprint. A
+ /// lock-free hint for picking the lock mode. A stale answer is harmless, the exclusive path
+ /// checks again and the shared guard cannot install, its attach fails on the missing pins.
+ pub fn is_current(&self) -> bool {
+ self.directions
+ .iter()
+ .all(|&dir| self.prog_pin_path(dir).exists())
+ }
+
+ fn any_programs_pinned(&self) -> bool {
+ std::fs::read_dir(self.progs_dir())
+ .map(|mut entries| entries.next().is_some())
+ .unwrap_or(false)
+ }
+
+ // the schema versions pinned here other than this build's, a directory that is no version at
+ // all is foreign state and counts as unknown
+ fn other_schemas(&self) -> anyhow::Result<Vec<Option<u32>>> {
+ let entries = match std::fs::read_dir(self.pin_root()) {
+ Ok(entries) => entries,
+ Err(e) if e.kind() == ErrorKind::NotFound => return Ok(Vec::new()),
+ Err(e) => return Err(e.into()),
+ };
+ let mut others = Vec::new();
+ for entry in entries {
+ let name = entry?.file_name();
+ let schema = name.to_str().and_then(|s| s.parse::<u32>().ok());
+ if schema != Some(self.schema_version) {
+ others.push(schema);
+ }
+ }
+ Ok(others)
+ }
+
+ fn verify(&self) -> anyhow::Result<()> {
+ let names: Vec<&str> = self
+ .directions
+ .iter()
+ .map(|&d| (self.prog_name)(d))
+ .collect();
+ tc::verify(self.name, &[self.obj], &names)
+ }
+
+ fn tear_down(&self) -> anyhow::Result<()> {
+ match std::fs::remove_dir_all(self.pin_root()) {
+ Ok(()) => Ok(()),
+ Err(e) if e.kind() == ErrorKind::NotFound => Ok(()),
+ Err(e) => Err(e.into()),
+ }
+ }
+
+ /// Load and verify the object, move the pinned links onto it, then pin its programs and drop
+ /// those of other builds. The verifier runs here and in the throwaway [`verify`](Self::verify),
+ /// nowhere else. The programs are pinned last since their pins are what marks this build as
+ /// current, so a failure before that point leaves the previous build in charge and the next
+ /// run retries. A link the update cannot reach has lost its netdev, so it is unpinned and the
+ /// next reconcile attaches the interface fresh if it is still wanted. Returns how many links
+ /// went that way. The `bpf` handle is dropped at the end, the pinned programs and maps stay
+ /// resident in the kernel.
+ fn install(&self) -> anyhow::Result<usize> {
+ std::fs::create_dir_all(self.links_dir())?;
+ let mut bpf = EbpfLoader::new()
+ .map_pin_path(self.schema_root())
+ .load(self.obj)?;
+ // load everything before touching any link, so a verifier rejection of one direction
+ // leaves the old state whole
+ for &dir in self.directions {
+ let name = (self.prog_name)(dir);
+ let prog: &mut SchedClassifier = bpf
+ .program_mut(name)
+ .with_context(|| format!("{}: program {name} not found in object", self.name))?
+ .try_into()?;
+ prog.load()?;
+ }
+
+ let live = self.live_links()?;
+ let mut reclaimed = 0usize;
+ for &dir in self.directions {
+ let prog = self.loaded(&mut bpf, dir)?;
+ for &(ifindex, _) in live.iter().filter(|&&(_, d)| d == dir) {
+ let path = self.link_pin_path(ifindex, dir);
+ if let Err(e) = tc::swap_pinned_link(prog, &path) {
+ log::warn!(
+ "{}: {ifindex}-{} swap failed ({e:#}), reclaiming",
+ self.name,
+ dir.as_str()
+ );
+ self.reclaim_link(ifindex, dir);
+ reclaimed += 1;
+ }
+ }
+ }
+
+ // a run that died after pinning some of the directions left a partial set behind
+ match std::fs::remove_dir_all(self.prog_dir()) {
+ Ok(()) => {}
+ Err(e) if e.kind() == ErrorKind::NotFound => {}
+ Err(e) => return Err(e).context("remove partial program pins"),
+ }
+ std::fs::create_dir_all(self.prog_dir())?;
+ for &dir in self.directions {
+ self.loaded(&mut bpf, dir)?.pin(self.prog_pin_path(dir))?;
+ }
+
+ // the links are on this build now, the other builds' programs go with their pins
+ let mine = self.fingerprint_dir_name();
+ for entry in std::fs::read_dir(self.progs_dir())? {
+ let entry = entry?;
+ if entry.file_name().to_string_lossy() != mine.as_str()
+ && let Err(e) = std::fs::remove_dir_all(entry.path())
+ {
+ log::warn!(
+ "{}: remove stale program pins {}: {e:#}",
+ self.name,
+ entry.path().display()
+ );
+ }
+ }
+ Ok(reclaimed)
+ }
+
+ // one direction's program out of the loaded object, resolved once by the load loop
+ fn loaded<'b>(
+ &self,
+ bpf: &'b mut aya::Ebpf,
+ dir: Direction,
+ ) -> anyhow::Result<&'b mut SchedClassifier> {
+ let prog = bpf
+ .program_mut((self.prog_name)(dir))
+ .expect("every program was resolved by the load loop")
+ .try_into()?;
+ Ok(prog)
+ }
+
+ fn live_links(&self) -> anyhow::Result<Vec<(u32, Direction)>> {
+ tc::read_pinned_links(&self.links_dir())
+ }
+
+ /// Unpin a defunct link. The pin file goes even when the kernel object behind it cannot be
+ /// opened any more.
+ fn reclaim_link(&self, ifindex: u32, dir: Direction) {
+ let path = self.link_pin_path(ifindex, dir);
+ if let Err(e) = tc::detach_pinned_link(&path) {
+ log::warn!(
+ "{}: reclaim {ifindex}-{}: unpin stale link: {e:#}",
+ self.name,
+ dir.as_str()
+ );
+ let _ = std::fs::remove_file(&path);
+ }
+ }
+
+ fn reconcile(&self, desired: &HashSet<u32>) -> anyhow::Result<()> {
+ let live: HashSet<(u32, Direction)> = self.live_links()?.into_iter().collect();
+
+ for &(ifidx, dir) in &live {
+ if desired.contains(&ifidx) {
+ continue;
+ }
+ log::debug!("{}: detach {ifidx}-{}", self.name, dir.as_str());
+ if let Err(e) = tc::detach_pinned_link(&self.link_pin_path(ifidx, dir)) {
+ log::error!("{}: detach {ifidx}-{}: {e:#}", self.name, dir.as_str());
+ } else {
+ log::info!("{}: detached {ifidx}-{}", self.name, dir.as_str());
+ }
+ }
+
+ let mut failed = 0usize;
+ for &dir in self.directions {
+ let mut prog = self.program(dir)?;
+ for &ifidx in desired {
+ if live.contains(&(ifidx, dir)) {
+ continue;
+ }
+ let path = self.link_pin_path(ifidx, dir);
+ if let Err(e) = tc::attach_and_pin(&mut prog, ifidx, dir, &path) {
+ log::error!("{}: attach {ifidx}-{}: {e:#}", self.name, dir.as_str());
+ failed += 1;
+ }
+ }
+ }
+ if failed > 0 {
+ anyhow::bail!("{}: {failed} interface(s) failed to attach", self.name);
+ }
+ Ok(())
+ }
+
+ fn attach_iface(&self, ifindex: u32) -> anyhow::Result<()> {
+ for &dir in self.directions {
+ // opened ahead of the pin check so a missing program fails here and never counts as a
+ // dead link
+ let mut prog = self.program(dir)?;
+ let path = self.link_pin_path(ifindex, dir);
+ if path.exists() {
+ match tc::swap_pinned_link(&mut prog, &path) {
+ Ok(()) => continue,
+ Err(e) => {
+ log::warn!(
+ "{}: {ifindex}-{} swap failed ({e:#}), rebuilding",
+ self.name,
+ dir.as_str()
+ );
+ self.reclaim_link(ifindex, dir);
+ }
+ }
+ }
+ match tc::attach_and_pin(&mut prog, ifindex, dir, &path) {
+ Ok(()) => {}
+ Err(e) if e.downcast_ref::<tc::PinExists>().is_some() => {
+ log::debug!(
+ "{}: {ifindex}-{} attached concurrently",
+ self.name,
+ dir.as_str()
+ );
+ }
+ Err(e) => {
+ return Err(e).with_context(|| {
+ format!("{}: attach {ifindex}-{}", self.name, dir.as_str())
+ });
+ }
+ }
+ }
+ Ok(())
+ }
+
+ /// Drop the pinned state of this subsystem, links included, under the exclusive lock, which
+ /// stays in place for whoever waits on it. For package removal, leaving links attached would
+ /// keep the programs running with nothing left to update them.
+ pub fn clear(&self) -> anyhow::Result<()> {
+ let _lock = self.lock_exclusive()?;
+ // the lock file under /run stays, unlinking it under a waiter would let that waiter run
+ // unserialized
+ self.tear_down()
+ }
+}
+
+/// Marker for a guard holding the apply lock shared.
+pub struct Shared;
+
+/// Marker for a guard holding the apply lock exclusively.
+pub struct Exclusive;
+
+/// The apply lock of one subsystem, held for as long as the guard lives. The steps that rewrite
+/// pinned state, loading and the full pass, exist only on the [`Exclusive`] guard, so nothing can
+/// run them next to shared holders. There is no upgrade, a caller that took the shared lock and
+/// finds an install needed drops it and takes the exclusive one.
+pub struct ApplyLock<'a, M> {
+ programs: &'a TcPrograms,
+ _lock: Flock<File>,
+ _mode: PhantomData<M>,
+}
+
+impl<M> ApplyLock<'_, M> {
+ /// Open a pinned BPF hash map by name, for the owning subsystem to sync. Valid once
+ /// [`ensure_loaded`](ApplyLock::ensure_loaded) has run, which is every caller's first step.
+ pub fn hash_map<K: aya::Pod, V: aya::Pod>(
+ &self,
+ name: &str,
+ ) -> anyhow::Result<aya::maps::HashMap<aya::maps::MapData, K, V>> {
+ self.programs.hash_map(name)
+ }
+
+ /// Attach the programs to a single interface, additively, without touching others.
+ ///
+ /// If a pin already exists, swap the program in place with no traffic gap. The swap only
+ /// succeeds while the link is still on a live netdev, a defunct pin is reclaimed and the
+ /// interface attached fresh. Two concurrent attaches of the same interface run under the
+ /// shared lock, the one that loses the pin race finds the other's link in place and is done.
+ /// A failed attach propagates so the caller can decide what an interface without the program
+ /// means for it.
+ pub fn attach_iface(&self, ifindex: u32) -> anyhow::Result<()> {
+ self.programs.attach_iface(ifindex)
+ }
+}
+
+impl ApplyLock<'_, Exclusive> {
+ /// Make sure the programs are loaded and pinned. The load runs only when there is no current
+ /// pin, when the schema version changed (a rebuild), or when the object changed (a refresh).
+ /// Otherwise the programs and maps pinned by an earlier run are reused untouched. Returns
+ /// whether the pinned state was found incomplete, so the caller knows its links and maps have
+ /// to be re-established. A clean refresh keeps both and reports false. A subsystem without
+ /// maps never bumps its schema and so never tears down.
+ pub fn ensure_loaded(&self) -> anyhow::Result<bool> {
+ self.programs.ensure_loaded()
+ }
+
+ /// Make the attached set match `desired`. Detach interfaces no longer wanted, attach the ones
+ /// missing. Refreshing existing links onto new code is done by
+ /// [`ensure_loaded`](Self::ensure_loaded). The live set it samples cannot change under it,
+ /// every other writer waits on the exclusive lock.
+ ///
+ /// Returns an error if any interface failed to attach (after attempting all of them), an
+ /// interface left without its program is the subsystem not doing its job there, so that
+ /// surfaces as a failed apply. A failed detach only leaves a program running longer than
+ /// wanted and is logged.
+ pub fn reconcile(&self, desired: &HashSet<u32>) -> anyhow::Result<()> {
+ self.programs.reconcile(desired)
+ }
+}
diff --git a/src/tc.rs b/src/tc.rs
new file mode 100644
index 0000000..a01dcaa
--- /dev/null
+++ b/src/tc.rs
@@ -0,0 +1,187 @@
+//! TC link plumbing shared by all subsystems. A `Direction` enum, attach/swap/detach free
+//! functions over tcx links (kernel 6.6 and newer), and a uniform pin-filename layout
+//! `{ifindex}-{direction}`.
+
+use std::{
+ io::ErrorKind,
+ path::{Path, PathBuf},
+ str::FromStr,
+};
+
+use anyhow::Context;
+use aya::{
+ EbpfLoader,
+ pin::PinError,
+ programs::{
+ SchedClassifier, TcAttachType,
+ links::{FdLink, LinkOrder, PinnedLink},
+ tc::TcAttachOptions,
+ },
+};
+
+use crate::subsystem::VERIFY_ROOT;
+
+/// An attach that found its pin already present, so a caller racing another attach of the same
+/// interface can tell that outcome from a real failure.
+#[derive(Debug)]
+pub struct PinExists;
+
+impl std::fmt::Display for PinExists {
+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+ f.write_str("pin path already exists")
+ }
+}
+
+impl std::error::Error for PinExists {}
+
+#[derive(Copy, Clone, Hash, PartialEq, Eq, Debug)]
+pub enum Direction {
+ Ingress,
+ Egress,
+}
+
+pub const DIRECTIONS: [Direction; 2] = [Direction::Ingress, Direction::Egress];
+
+impl Direction {
+ pub fn as_str(self) -> &'static str {
+ match self {
+ Self::Ingress => "ingress",
+ Self::Egress => "egress",
+ }
+ }
+ pub fn aya_type(self) -> TcAttachType {
+ match self {
+ Self::Ingress => TcAttachType::Ingress,
+ Self::Egress => TcAttachType::Egress,
+ }
+ }
+}
+
+impl FromStr for Direction {
+ type Err = ();
+ fn from_str(s: &str) -> Result<Self, Self::Err> {
+ match s {
+ "ingress" => Ok(Self::Ingress),
+ "egress" => Ok(Self::Egress),
+ _ => Err(()),
+ }
+ }
+}
+
+/// RAII handle for a subsystem's verify root, so the throwaway state is gone even when the verify
+/// body panics or returns Err.
+struct VerifyRoot(PathBuf);
+
+impl VerifyRoot {
+ fn new(subsystem: &str) -> anyhow::Result<Self> {
+ let path = Path::new(VERIFY_ROOT).join(subsystem);
+ let _ = std::fs::remove_dir_all(&path);
+ std::fs::create_dir_all(&path)
+ .with_context(|| format!("create verify root {}", path.display()))?;
+ Ok(Self(path))
+ }
+ fn path(&self) -> &Path {
+ &self.0
+ }
+}
+
+impl Drop for VerifyRoot {
+ fn drop(&mut self) {
+ if let Err(e) = std::fs::remove_dir_all(&self.0) {
+ log::warn!("failed to clean up verify root {}: {e:#}", self.0.display());
+ }
+ }
+}
+
+/// Loads every named program in each object against a throwaway pin root of the subsystem,
+/// catching verifier regressions before any real state is touched.
+pub fn verify(subsystem: &str, objects: &[&[u8]], program_names: &[&str]) -> anyhow::Result<()> {
+ let root = VerifyRoot::new(subsystem)?;
+ for &obj in objects {
+ let mut bpf = EbpfLoader::new().map_pin_path(root.path()).load(obj)?;
+ for &name in program_names {
+ let p: &mut SchedClassifier = bpf
+ .program_mut(name)
+ .with_context(|| format!("program {name} not found in object"))?
+ .try_into()?;
+ p.load()?;
+ }
+ }
+ Ok(())
+}
+
+pub fn pin_filename(ifindex: u32, dir: Direction) -> String {
+ format!("{ifindex}-{}", dir.as_str())
+}
+
+/// Attach `prog` in `dir` as a tcx link and pin it. tcx (kernel 6.6 and newer) hangs the
+/// program on the netdev hook itself, no qdisc is involved, so nothing here competes with the
+/// `ingress` qdisc the rate limiter installs on the same interface. Programs of several
+/// subsystems stack in attach order, each new link goes last.
+pub fn attach_and_pin(
+ prog: &mut SchedClassifier,
+ ifindex: u32,
+ dir: Direction,
+ pin_path: &Path,
+) -> anyhow::Result<()> {
+ let name = nix::net::if_::if_indextoname(ifindex)?;
+ let name = name.to_str()?;
+ let link_id = prog.attach_with_options(
+ name,
+ dir.aya_type(),
+ TcAttachOptions::TcxOrder(LinkOrder::last()),
+ )?;
+ let link = prog.take_link(link_id)?;
+ let fd_link: FdLink = link.try_into()?;
+ // a lost race against another attach of the same interface shows up here, the unpinned link
+ // detaches again when its fd drops
+ match fd_link.pin(pin_path) {
+ Ok(_) => Ok(()),
+ Err(PinError::SyscallError(e)) if e.io_error.kind() == ErrorKind::AlreadyExists => {
+ Err(PinExists.into())
+ }
+ Err(e) => Err(e.into()),
+ }
+}
+
+/// Rebind a pinned link to `prog` via LINK_UPDATE. Atomic, traffic sees no detach/reattach gap.
+pub fn swap_pinned_link(prog: &mut SchedClassifier, pin_path: &Path) -> anyhow::Result<()> {
+ let pinned = PinnedLink::from_pin(pin_path)?;
+ let fd_link: FdLink = pinned.into();
+ let link = fd_link.try_into()?;
+ let new_id = prog.attach_to_link(link)?;
+ // take the handle out of aya's internal tracking, we have the pin
+ let _ = prog.take_link(new_id)?;
+ Ok(())
+}
+
+pub fn detach_pinned_link(pin_path: &Path) -> anyhow::Result<()> {
+ let pinned = PinnedLink::from_pin(pin_path)?;
+ let _fd_link = pinned.unpin()?;
+ Ok(())
+}
+
+/// Read and parse every pin file in `links_dir` as `{ifindex}-{direction}`. Unrecognized names are
+/// skipped with a warning.
+pub fn read_pinned_links(links_dir: &Path) -> anyhow::Result<Vec<(u32, Direction)>> {
+ let mut out = Vec::new();
+ let dir = match std::fs::read_dir(links_dir) {
+ Ok(d) => d,
+ Err(e) if e.kind() == ErrorKind::NotFound => return Ok(out),
+ Err(e) => return Err(e.into()),
+ };
+ for entry in dir {
+ let entry = entry?;
+ let name = entry.file_name();
+ let name = name.to_string_lossy();
+ let Some((ifidx_str, dir_str)) = name.split_once('-') else {
+ log::warn!("unrecognized pin file in links dir: {name}");
+ continue;
+ };
+ match (ifidx_str.parse::<u32>(), dir_str.parse::<Direction>()) {
+ (Ok(ifidx), Ok(d)) => out.push((ifidx, d)),
+ _ => log::warn!("unrecognized pin file in links dir: {name}"),
+ }
+ }
+ Ok(out)
+}
--
2.47.3
next prev parent reply other threads:[~2026-09-04 9:05 UTC|newest]
Thread overview: 4+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-04 9:04 [PATCH proxmox-ebpf v2 0/3] add proxmox-ebpf library Hannes Laimer
2026-09-04 9:04 ` Hannes Laimer [this message]
2026-09-04 9:04 ` [PATCH proxmox-ebpf v2 2/3] tests: add a native harness for the BPF C programs Hannes Laimer
2026-09-04 9:04 ` [PATCH proxmox-ebpf v2 3/3] debian: package the crate as a rust library Hannes Laimer
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260904090458.990888-2-h.laimer@proxmox.com \
--to=h.laimer@proxmox.com \
--cc=pve-devel@lists.proxmox.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.