Cross-compilation and macOS compatibility fixes (#1)

- Ethernet ioctl: cfg-gate the ioctl request parameter type — c_int on
  musl, c_ulong on gnu — so the same code compiles on both
  x86_64-unknown-linux-gnu and x86_64-unknown-linux-musl targets

- Chaos harness macOS support: replace direct host 'ip link' invocations
  with a privileged Docker container helper (--net=host --pid=host) that
  shares the Docker VM's namespaces, making veth pair setup work on both
  Linux and macOS (where host 'ip' is unavailable and container PIDs
  live inside the Docker Desktop VM)

- Python 3.9 compat: add 'from __future__ import annotations' to
  docker_exec.py for X|Y union syntax support

- Add deploy/ to .gitignore

Co-authored-by: origami74 <origami74@gmail.com>
This commit is contained in:
Arjen
2026-02-26 03:48:51 -08:00
committed by GitHub
co-authored by origami74
parent d29da442ac
commit c1589c02af
4 changed files with 103 additions and 19 deletions
+2
View File
@@ -13,3 +13,5 @@
# Claude Code # Claude Code
.claude/ .claude/
deploy/
+10 -2
View File
@@ -329,7 +329,11 @@ fn get_mac_addr(fd: RawFd, if_index: i32) -> Result<[u8; 6], TransportError> {
); );
} }
let ret = unsafe { libc::ioctl(fd, libc::SIOCGIFHWADDR as libc::c_ulong, &ifr) }; #[cfg(target_env = "musl")]
let ioctl_req = libc::SIOCGIFHWADDR as libc::c_int;
#[cfg(not(target_env = "musl"))]
let ioctl_req = libc::SIOCGIFHWADDR as libc::c_ulong;
let ret = unsafe { libc::ioctl(fd, ioctl_req, &ifr) };
if ret < 0 { if ret < 0 {
return Err(TransportError::StartFailed(format!( return Err(TransportError::StartFailed(format!(
"ioctl(SIOCGIFHWADDR) failed: {}", "ioctl(SIOCGIFHWADDR) failed: {}",
@@ -375,7 +379,11 @@ fn get_if_mtu(fd: RawFd, if_index: i32) -> Result<u16, TransportError> {
); );
} }
let ret = unsafe { libc::ioctl(fd, libc::SIOCGIFMTU as libc::c_ulong, &ifr) }; #[cfg(target_env = "musl")]
let ioctl_req = libc::SIOCGIFMTU as libc::c_int;
#[cfg(not(target_env = "musl"))]
let ioctl_req = libc::SIOCGIFMTU as libc::c_ulong;
let ret = unsafe { libc::ioctl(fd, ioctl_req, &ifr) };
if ret < 0 { if ret < 0 {
return Err(TransportError::StartFailed(format!( return Err(TransportError::StartFailed(format!(
"ioctl(SIOCGIFMTU) failed: {}", "ioctl(SIOCGIFMTU) failed: {}",
+2
View File
@@ -1,5 +1,7 @@
"""Thin wrapper around subprocess for docker exec calls.""" """Thin wrapper around subprocess for docker exec calls."""
from __future__ import annotations
import subprocess import subprocess
import logging import logging
+89 -17
View File
@@ -9,6 +9,24 @@ each container's network namespace. Naming:
After creation, the container-side MAC addresses are queried and After creation, the container-side MAC addresses are queried and
stored in SimNode.ethernet_macs for use in config generation. stored in SimNode.ethernet_macs for use in config generation.
Implementation note
-------------------
All ``ip link`` operations that manipulate the host network stack are
executed inside a short-lived privileged Docker container that shares
the host network and PID namespaces (``--net=host --pid=host``). This
works on both Linux and macOS:
* **Linux** the helper container shares the real host network/PID
namespaces, so ``ip link set ... netns <pid>`` behaves identically to
running ``ip`` directly on the host.
* **macOS** Docker containers run inside a Linux VM; the helper
container shares *that* VM's namespaces, which is exactly where the
simulation containers live. Running ``ip`` on the macOS host would
never work because the container PIDs are in the VM, not macOS.
The helper image is resolved from the running simulation containers
(which already have ``iproute2`` from the chaos Dockerfile).
""" """
from __future__ import annotations from __future__ import annotations
@@ -21,7 +39,6 @@ from .topology import SimTopology, veth_interface_name
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
class VethManager: class VethManager:
"""Manages veth pairs for Ethernet-transport edges.""" """Manages veth pairs for Ethernet-transport edges."""
@@ -30,6 +47,34 @@ class VethManager:
# Track created host-side temp names for cleanup # Track created host-side temp names for cleanup
self._host_pairs: list[tuple[str, str, str, str]] = [] self._host_pairs: list[tuple[str, str, str, str]] = []
# (node_a, node_b, host_name_a, host_name_b) # (node_a, node_b, host_name_a, host_name_b)
self._ip_image: str | None = None
def _get_image(self) -> str:
"""Resolve the Docker image to use for ip(8) helper containers.
Uses the image of the first simulation container (which already has
iproute2 installed via the chaos Dockerfile). Must be called after
containers are started.
"""
if self._ip_image is not None:
return self._ip_image
first_node = next(iter(sorted(self.topology.nodes)))
container = self.topology.container_name(first_node)
result = subprocess.run(
["docker", "inspect", "-f", "{{.Config.Image}}", container],
capture_output=True,
text=True,
timeout=10,
)
if result.returncode != 0 or not result.stdout.strip():
raise RuntimeError(
f"Cannot determine Docker image for ip(8) helper "
f"(docker inspect {container} failed): {result.stderr.strip()}"
)
self._ip_image = result.stdout.strip()
log.debug("Using sim image %s for ip(8) helper", self._ip_image)
return self._ip_image
def setup_all(self): def setup_all(self):
"""Create veth pairs for all Ethernet edges. """Create veth pairs for all Ethernet edges.
@@ -45,10 +90,11 @@ class VethManager:
if not eth_edges: if not eth_edges:
return return
log.info("Setting up %d Ethernet veth pairs...", len(eth_edges)) image = self._get_image()
log.info("Setting up %d Ethernet veth pairs (helper image: %s)...", len(eth_edges), image)
for a, b in eth_edges: for a, b in eth_edges:
self._create_veth_pair(a, b) self._create_veth_pair(a, b, image)
log.info( log.info(
"Veth setup complete: %d pairs", "Veth setup complete: %d pairs",
@@ -62,6 +108,7 @@ class VethManager:
destroyed. We re-create the veth pairs for all Ethernet edges destroyed. We re-create the veth pairs for all Ethernet edges
involving this node. involving this node.
""" """
image = self._get_image()
for a, b in self.topology.ethernet_edges(): for a, b in self.topology.ethernet_edges():
if a != node_id and b != node_id: if a != node_id and b != node_id:
continue continue
@@ -69,17 +116,18 @@ class VethManager:
nn_a = a.replace("n", "") nn_a = a.replace("n", "")
nn_b = b.replace("n", "") nn_b = b.replace("n", "")
host_a = f"vh{nn_a}{nn_b}a" host_a = f"vh{nn_a}{nn_b}a"
_run_host(["ip", "link", "delete", host_a], check=False) _run_host(["ip", "link", "delete", host_a], image, check=False)
# Re-create # Re-create
self._create_veth_pair(a, b) self._create_veth_pair(a, b, image)
def teardown_all(self): def teardown_all(self):
"""Clean up all veth pairs.""" """Clean up all veth pairs."""
image = self._get_image()
for _, _, host_a, _ in self._host_pairs: for _, _, host_a, _ in self._host_pairs:
_run_host(["ip", "link", "delete", host_a], check=False) _run_host(["ip", "link", "delete", host_a], image, check=False)
self._host_pairs.clear() self._host_pairs.clear()
def _create_veth_pair(self, node_a: str, node_b: str): def _create_veth_pair(self, node_a: str, node_b: str, image: str):
"""Create a single veth pair between two containers.""" """Create a single veth pair between two containers."""
container_a = self.topology.container_name(node_a) container_a = self.topology.container_name(node_a)
container_b = self.topology.container_name(node_b) container_b = self.topology.container_name(node_b)
@@ -102,19 +150,19 @@ class VethManager:
final_b = veth_interface_name(node_b, node_a) final_b = veth_interface_name(node_b, node_a)
# Clean up any stale pair # Clean up any stale pair
_run_host(["ip", "link", "delete", host_a], check=False) _run_host(["ip", "link", "delete", host_a], image, check=False)
# Create veth pair on host # Create veth pair on host
ok = _run_host([ ok = _run_host([
"ip", "link", "add", host_a, "type", "veth", "peer", "name", host_b, "ip", "link", "add", host_a, "type", "veth", "peer", "name", host_b,
]) ], image)
if not ok: if not ok:
log.warning("Failed to create veth pair %s/%s", host_a, host_b) log.warning("Failed to create veth pair %s/%s", host_a, host_b)
return return
# Move into container namespaces # Move into container namespaces
_run_host(["ip", "link", "set", host_a, "netns", str(pid_a)]) _run_host(["ip", "link", "set", host_a, "netns", str(pid_a)], image)
_run_host(["ip", "link", "set", host_b, "netns", str(pid_b)]) _run_host(["ip", "link", "set", host_b, "netns", str(pid_b)], image)
# Rename and bring up inside containers # Rename and bring up inside containers
docker_exec_quiet( docker_exec_quiet(
@@ -175,19 +223,43 @@ def _get_mac_in_container(container: str, iface: str) -> str | None:
return None return None
def _run_host(cmd: list[str], check: bool = True) -> bool: def _run_host(cmd: list[str], image: str, check: bool = True) -> bool:
"""Run a command on the host. Returns True on success.""" """Run an ``ip`` command via a privileged Docker container.
Uses ``--net=host --pid=host --privileged`` so the container shares
the Docker host's (or Docker Desktop VM's) network and PID
namespaces. This makes ``ip link set ... netns <pid>`` work
correctly on both Linux and macOS.
``image`` should be a Docker image that has ``iproute2`` installed
(e.g. the simulation's own image built from the chaos Dockerfile).
``--entrypoint ip`` overrides the image's default entrypoint so the
simulation entrypoint script is not executed.
"""
docker_cmd = [
"docker", "run", "--rm",
"--privileged",
"--net=host",
"--pid=host",
"--entrypoint", "ip",
image,
] + cmd[1:] # cmd[0] is "ip", skip it since it's now the entrypoint
try: try:
result = subprocess.run( result = subprocess.run(
cmd, docker_cmd,
capture_output=True, capture_output=True,
text=True, text=True,
timeout=10, timeout=30,
) )
if check and result.returncode != 0: if check and result.returncode != 0:
log.debug("Host cmd failed: %s -> %s", " ".join(cmd), result.stderr.strip()) log.debug(
"ip cmd failed: %s -> %s",
" ".join(cmd),
result.stderr.strip(),
)
return False return False
return result.returncode == 0 return result.returncode == 0
except subprocess.TimeoutExpired: except subprocess.TimeoutExpired:
log.warning("Host cmd timed out: %s", " ".join(cmd)) log.warning("ip cmd timed out: %s", " ".join(cmd))
return False return False