From f1d6df7d5137db60997d3d632d8a7b2eb20f378c Mon Sep 17 00:00:00 2001 From: alexion Date: Sat, 25 Jul 2026 23:02:58 -0400 Subject: [PATCH] test(guests): boot a guest end to end in a VM (task 0010) Add a NixOS VM integration test to the flake's checks, so nix flake check boots the network foundation and one guest on a virtual L2 segment and asserts the three behaviors a VM can honestly reproduce: the guest presents its own MAC distinct from the host's, gets its own IP on a tagged VLAN across the segment, and writes to a bind mount owned by the shared storage group. A tagged router node serving DHCP only on VLAN 10 makes the address and reverse ping prove 802.1Q tagging end to end, not plain reachability. Booting a networked guest surfaced a latent defect: the guest's own networkd default-enables systemd-resolved, which conflicts with the nested-container default of inheriting the host's resolv.conf, failing the guest toplevel build. Fix it in the guest networking realization so a networked guest keeps its own resolver. --- .../tasks/0010-guest-vm-integration-test.md | 40 ++++ flake.nix | 11 +- lib.nix | 5 + tests/guest-integration.nix | 192 ++++++++++++++++++ 4 files changed, 246 insertions(+), 2 deletions(-) create mode 100644 .claude/tasks/0010-guest-vm-integration-test.md create mode 100644 tests/guest-integration.nix diff --git a/.claude/tasks/0010-guest-vm-integration-test.md b/.claude/tasks/0010-guest-vm-integration-test.md new file mode 100644 index 0000000..9dbd5ca --- /dev/null +++ b/.claude/tasks/0010-guest-vm-integration-test.md @@ -0,0 +1,40 @@ +--- +spec: guests +blocked-by: [0004-guest-networking-placement, 0006-guest-storage-placement] +--- + +## What to build + +One NixOS VM integration test, added to the flake's `checks` so `nix flake check` runs it, exercising the external observable behavior of a Guest and its foundations rather than the internal shape of the generated nested-container config. +A single harness boots the `modules.network` foundation and one sample Guest and asserts the three hard requirements together: the Guest presents its own MAC, gets its own IP on a tagged VLAN across a virtual L2 segment, and can write to a bind-mounted directory owned by the shared `storage` group. +The upstream NixOS test suite's nested-container networking cases (macvlan, extra-veth) are the model. +The two behaviors a VM cannot honestly reproduce — real 802.1Q against the physical switch and real ZFS identity-mapped writes on the pool — are out of this test and verified manually on the target Host. + +## Acceptance criteria + +- [x] A NixOS VM test is added to the flake's `checks` and runs as part of `nix flake check`. +- [x] The test boots `modules.network` and one sample Guest on a virtual L2 segment. +- [x] The test asserts the Guest presents its own MAC distinct from the Host's. +- [x] The test asserts the Guest gets its own IP on the correct tagged VLAN across the virtual segment. +- [x] The test asserts a guest service can write to a bind-mounted directory owned by the shared `storage` group. + +## Implementation Notes + +The test lives in `tests/guest-integration.nix` and is merged into `checks.x86_64-linux` as `guest-integration`, built through `pkgs.testers.runNixOSTest`. +The flake builds its nodes with the same `my`/`inputs` special arguments every configuration gets, passed through `node.specialArgs`, so the host node imports the real `modules/network.nix` and the real `my.guest` builder rather than a hand-rolled stand-in. + +The harness is two nodes on one test-framework segment, the "virtual L2 segment". +The host runs `modules.network` with `trunk = "eth1"` and `vlans = [10]`, and one guest placed on VLAN 10. +A second `router` node speaks VLAN 10 only on a tagged `eth1.10` sub-interface and serves DHCP there, so the guest getting a `10.0.10.x` lease and the reverse `router → guest` ping succeed only when 802.1Q tagging works end to end across the segment. +This exercises the tagged path honestly rather than plain co-segment reachability. +The MAC assertion checks the guest's `eth0` equals its derived placement MAC and differs from the host trunk, and the storage assertion relies on the identity map (`privateUsers = "no"`) carrying gid 10000 through unshifted, so `stat -c %G` reading `storage` on the host is the "no permission errors" mechanism under test. + +Booting a networked guest surfaced a latent defect in the guest networking foundation: enabling the guest's own networkd default-enables `systemd-resolved`, which conflicts with the nested-container default of inheriting the host's `resolv.conf`, and the guest's toplevel failed to build with "Using host resolv.conf is not supported with systemd-resolved". +Task 0004 never hit this because it only evaluated derived values, never built a networked guest's toplevel, and no committed host places a networked guest. +The fix is one line in `guestNet` (`networking.useHostResolvConf = false`), so any networked guest keeps its own resolver. + +The guest interior here carries a storage-writing service as test scaffolding, since a real guest seals its own interior and the sample guest carries no service. +The host node also orders `container@sample` after the bridge's device unit, since the container enslaves its veth to the bridge at start and the upstream containers module orders only after `network.target`, not after the specific `hostBridge`. + +Following the pattern of tasks 0003, 0004, and 0006, no committed host places a networked or pool-mounted guest, since the repo's only host is a wifi laptop with no bridge or pool. +The behaviors a VM cannot honestly reproduce — real 802.1Q against the physical switch and real ZFS identity-mapped writes on the pool — stay out of this test and are verified manually on the target Host, as the spec's testing decisions direct. diff --git a/flake.nix b/flake.nix index 68d7157..8e70b36 100644 --- a/flake.nix +++ b/flake.nix @@ -79,9 +79,16 @@ # Every host under hosts/ is discovered and built. nixosConfigurations = my.mkHosts (self + "/hosts"); - # `nix flake check` builds each host's toplevel. + # `nix flake check` builds each host's toplevel, and boots one guest + # end to end in a VM to exercise its externally observable behavior. checks.x86_64-linux = lib.mapAttrs ( _name: host: host.config.system.build.toplevel - ) self.nixosConfigurations; + ) self.nixosConfigurations + // { + guest-integration = import ./tests/guest-integration.nix { + inherit inputs self; + system = "x86_64-linux"; + }; + }; }; } diff --git a/lib.nix b/lib.nix index 8f0d51f..d57f36d 100644 --- a/lib.nix +++ b/lib.nix @@ -139,6 +139,11 @@ let { config = lib.mkIf networked { networking.useNetworkd = true; + + # networkd default-enables resolved, which owns the guest's resolv.conf. + # The nested-container default of inheriting the host's file conflicts with that, so the guest keeps its own. + networking.useHostResolvConf = false; + systemd.network.networks."20-eth0" = { matchConfig.Name = "eth0"; linkConfig.MACAddress = cfg.mac; diff --git a/tests/guest-integration.nix b/tests/guest-integration.nix new file mode 100644 index 0000000..af62cac --- /dev/null +++ b/tests/guest-integration.nix @@ -0,0 +1,192 @@ +{ + inputs, + self, + system, +}: +# Boots the network foundation and one guest end to end in a VM, asserting the +# externally observable guest behaviors a VM can honestly reproduce. +let + pkgs = import inputs.nixpkgs { inherit system; }; + + vlan = 10; + subnet = "10.0.10"; + routerAddress = "${subnet}.1"; + + # The tagged sub-interface the router speaks VLAN 10 on, so the guest reaches + # it only when frames are tagged correctly across the wire. + routerVlanLink = "eth1.${toString vlan}"; + + # A guest interior that writes a marker file as the shared storage group, so + # the host can observe the write landing on its bind mount as that group. + # This is test scaffolding, since a real guest seals its own interior. + storageWriter = + { ... }: + { + users.users.svc = { + isSystemUser = true; + group = "storage"; + }; + + systemd.services.storage-writer = { + wantedBy = [ "multi-user.target" ]; + after = [ "local-fs.target" ]; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + User = "svc"; + }; + script = "echo guest-wrote-this > /data/marker"; + }; + }; +in +pkgs.testers.runNixOSTest { + name = "guest-integration"; + + # The host node evaluates the flake's own modules, so it needs the same + # special arguments the flake builds every configuration with. + node.specialArgs = { + my = self.lib; + inherit inputs; + }; + + nodes.host = + { my, lib, ... }: + { + imports = [ + (inputs.self + "/modules/network.nix") + (my.guest { + name = "sample"; + interior = storageWriter; + }) + # The guest realization declares sops.secrets, so the option must exist + # even though this guest names no secrets. + inputs.sops-nix.nixosModules.sops + ]; + + # The trunk the network foundation tags VLANs onto, kept address-free so + # the foundation owns it entirely. + virtualisation.interfaces.eth1.vlan = 1; + networking.useNetworkd = true; + networking.useDHCP = false; + + # The shared write group at the fixed gid every guest carries, so an + # identity-mapped guest write lands on the host as this same group. + users.groups.storage.gid = 10000; + + # The bind-mount target, group-owned by storage and group-writable with the + # setgid bit, so a storage-group process in the guest can create files here. + systemd.tmpfiles.rules = [ "d /srv/shared 2770 root storage - -" ]; + + # The container enslaves its veth to the bridge at start, so it must wait + # for the foundation to have created that bridge. + systemd.services."container@sample" = + let + bridgeDevice = + "sys-subsystem-net-devices-" + + lib.replaceStrings [ "-" ] [ "\\x2d" ] (my.bridgeName vlan) + + ".device"; + in + { + after = [ bridgeDevice ]; + wants = [ bridgeDevice ]; + }; + + modules.network = { + enable = true; + trunk = "eth1"; + vlans = [ vlan ]; + }; + + guests.sample = { + enable = true; + vlan = vlan; + mounts."/data".hostPath = "/srv/shared"; + }; + }; + + # A peer on the same virtual segment that speaks only tagged VLAN 10 and hands + # out addresses on it, so the guest reaching it proves the tagged path works. + nodes.router = + { ... }: + { + virtualisation.interfaces.eth1.vlan = 1; + networking.useNetworkd = true; + networking.useDHCP = false; + networking.firewall.enable = false; + + systemd.network = { + enable = true; + + netdevs."40-${routerVlanLink}" = { + netdevConfig = { + Name = routerVlanLink; + Kind = "vlan"; + }; + vlanConfig.Id = vlan; + }; + + networks = { + "30-eth1" = { + matchConfig.Name = "eth1"; + networkConfig.LinkLocalAddressing = "no"; + linkConfig.RequiredForOnline = "no"; + vlan = [ routerVlanLink ]; + }; + "40-${routerVlanLink}" = { + matchConfig.Name = routerVlanLink; + networkConfig = { + Address = "${routerAddress}/24"; + DHCPServer = true; + }; + dhcpServerConfig = { + PoolOffset = 100; + PoolSize = 10; + }; + }; + }; + }; + }; + + testScript = + { nodes, ... }: + let + guestMac = nodes.host.guests.sample.mac; + in + '' + import re + + start_all() + + host.wait_for_unit("multi-user.target") + router.wait_for_unit("systemd-networkd.service") + router.wait_until_succeeds("ip -4 addr show ${routerVlanLink} | grep -q ${routerAddress}") + + with subtest("the guest container comes up"): + host.wait_until_succeeds("nixos-container status sample | grep -q up") + + with subtest("the guest presents its own MAC, distinct from the host's"): + guest_mac = host.succeed( + "nixos-container run sample -- cat /sys/class/net/eth0/address" + ).strip() + assert guest_mac == "${guestMac}", \ + f"guest eth0 MAC {guest_mac} != configured ${guestMac}" + host_mac = host.succeed("cat /sys/class/net/eth1/address").strip() + assert guest_mac != host_mac, \ + f"guest MAC {guest_mac} must differ from host trunk MAC {host_mac}" + + with subtest("the guest gets its own IP on the tagged VLAN across the segment"): + host.wait_until_succeeds( + "nixos-container run sample -- ip -4 -o addr show eth0 | grep -q 'inet ${subnet}\\.'" + ) + out = host.succeed("nixos-container run sample -- ip -4 -o addr show eth0") + match = re.search(r"inet (${subnet}\.\d+)", out) + assert match is not None, f"no VLAN address on guest eth0: {out}" + router.succeed(f"ping -n -c 1 -w 30 {match.group(1)}") + + with subtest("a guest service writes to the storage-group bind mount"): + host.wait_for_file("/srv/shared/marker") + host.succeed("grep -q guest-wrote-this /srv/shared/marker") + group = host.succeed("stat -c %G /srv/shared/marker").strip() + assert group == "storage", f"marker file group {group} != storage" + ''; +}