diff --git a/nixos/tests/ceph-multi-node-bluestore.nix b/nixos/tests/ceph-multi-node-bluestore.nix index 73c120a246fd..19e0d0d5e4ac 100644 --- a/nixos/tests/ceph-multi-node-bluestore.nix +++ b/nixos/tests/ceph-multi-node-bluestore.nix @@ -25,19 +25,16 @@ let osd0 = { name = "0"; ip = "192.168.1.2"; - key = "AQBCEJNa3s8nHRAANvdsr93KqzBznuIWm2gOGg=="; uuid = "55ba2294-3e24-478f-bee0-9dca4c231dd9"; }; osd1 = { name = "1"; ip = "192.168.1.3"; - key = "AQBEEJNac00kExAAXEgy943BGyOpVH1LLlHafQ=="; uuid = "5e97a838-85b6-43b0-8950-cb56d554d1e5"; }; osd2 = { name = "2"; ip = "192.168.1.4"; - key = "AQAdyhZeIaUlARAAGRoidDAmS6Vkp546UFEf5w=="; uuid = "ea999274-13d0-4dd5-9af9-ad25a324f72f"; }; # Client that mounts CephFS using the in-kernel client. @@ -58,6 +55,9 @@ let monHost = cfg.monA.ip; monInitialMembers = cfg.monA.name; }; + extraConfig = { + mon_host = "v2:${cfg.monA.ip}:3300 v1:${cfg.monA.ip}:6789"; + }; } // daemonConfig; @@ -310,14 +310,15 @@ let "sudo -u ceph ceph-authtool --create-keyring /tmp/ceph.mon.keyring --gen-key -n mon. --cap mon 'allow *'", "sudo -u ceph ceph-authtool --create-keyring /etc/ceph/ceph.client.admin.keyring --gen-key -n client.admin --cap mon 'allow *' --cap osd 'allow *' --cap mds 'allow *' --cap mgr 'allow *'", "sudo -u ceph ceph-authtool /tmp/ceph.mon.keyring --import-keyring /etc/ceph/ceph.client.admin.keyring", - "monmaptool --create --add ${cfg.monA.name} ${cfg.monA.ip} --fsid ${cfg.clusterId} /tmp/monmap", + # Creating the mon with v2 (and a legacy v1) address right away removes the need for running `enable-msgr2` later on. + # It is also makes the test more consistent by fixing the address to a known value instead of letting it derive the address. + "monmaptool --create --addv ${cfg.monA.name} '[v2:${cfg.monA.ip}:3300,v1:${cfg.monA.ip}:6789]' --auth-allowed-ciphers aes256k --auth-preferred-cipher aes256k --auth-service-cipher aes256k --fsid ${cfg.clusterId} /tmp/monmap", "sudo -u ceph ceph-mon --mkfs -i ${cfg.monA.name} --monmap /tmp/monmap --keyring /tmp/ceph.mon.keyring", "sudo -u ceph mkdir -p /var/lib/ceph/mgr/ceph-${cfg.monA.name}/", "sudo -u ceph touch /var/lib/ceph/mon/ceph-${cfg.monA.name}/done", "systemctl start ceph-mon-${cfg.monA.name}", ) monA.wait_for_unit("ceph-mon-${cfg.monA.name}") - monA.succeed("ceph mon enable-msgr2") monA.succeed("ceph config set mon auth_allow_insecure_global_id_reclaim false") # Can't check ceph status until a mon is up @@ -387,6 +388,9 @@ let monA.wait_until_succeeds("ceph -s | grep 'HEALTH_OK'") monA.succeed( + # Autoscaling will cause PGs to be peering, causing the tests to become flakey. + "ceph osd pool set noautoscale", + "ceph osd pool create multi-node-test 32 32", "ceph osd pool ls | grep 'multi-node-test'", @@ -536,45 +540,50 @@ let # Create a CephFS. monA.succeed( - "ceph osd pool create cephfs-data 32 32", - "ceph osd pool create cephfs-metadata 32 32", - "ceph fs new cephfs cephfs-metadata cephfs-data", + "ceph fs volume create testing", + "ceph osd pool set cephfs.testing.data pg_num 32", + "ceph osd pool set cephfs.testing.meta pg_num 32", ) # Wait for the MDS to claim the filesystem and become active. - monA.wait_until_succeeds("ceph fs status cephfs | grep -e 'active'", timeout=60) + monA.wait_until_succeeds("ceph fs status testing | grep -e 'active'", timeout=60) - # Distribute the admin keyring (and a plain secret file for the kernel - # client) to both client machines, so that they can authenticate. + # Create a subvolume, issue credentials, then distribute those credentials. monA.succeed( - "cp /etc/ceph/ceph.client.admin.keyring /tmp/shared", - "ceph-authtool -p /etc/ceph/ceph.client.admin.keyring > /tmp/shared/admin.secret", + "ceph fs subvolumegroup create testing group", + "ceph fs subvolume create testing subvolume --group_name group", + "ceph fs subvolume authorize testing subvolume kclient group", + "ceph fs subvolume authorize testing subvolume fuseclient group", + "ceph auth get client.kclient -o /tmp/shared/ceph.client.kclient.keyring", + "ceph auth get client.fuseclient -o /tmp/shared/ceph.client.fuseclient.keyring", ) - kclient.succeed("cp /tmp/shared/ceph.client.admin.keyring /etc/ceph") - fuseclient.succeed("cp /tmp/shared/ceph.client.admin.keyring /etc/ceph") - kclient.succeed("cp /tmp/shared/admin.secret /etc/ceph/admin.secret") + kclient.succeed("cp /tmp/shared/ceph.client.kclient.keyring /etc/ceph") + fuseclient.succeed("cp /tmp/shared/ceph.client.fuseclient.keyring /etc/ceph") + + # Get the volume path generated by Ceph. + volume_path = monA.succeed("ceph fs subvolume getpath testing subvolume group | tee /dev/stderr").strip() # Mount CephFS on the kernel client. # We force the messenger v2 protocol via "ms_mode=secure"; the cluster - # has msgr2 enabled (see "ceph mon enable-msgr2" above) and the legacy v1 + # has msgr2 enabled (the monmap is created with a v2 address above) and the legacy v1 # protocol apparently does not reconnect reliably after the servers are restarted. # The msgr2 monitor listens on port 3300 (instead of legacy v1 port 6789), # so we have to point the device string at that port explicitly. # `recover_session=clean` makes the kernel client automatically reconnect # (discarding its stale session) after the whole cluster has been down, - # which would otherwise leave the mount blocklisted and hanging forever. + # which would otherwise leave the mount blocklisted and hanging. # Real CephFS use may not prefer hanging `recover_session=clean`, and # prefer manual de-blocklisting to avoid any failed OS syscalls, # but for this test, discarding stale sessions is good enough. kclient.succeed("mkdir -p /mnt/cephfs") kclient.wait_until_succeeds( - "mount -t ceph ${cfg.monA.ip}:3300:/ /mnt/cephfs -o name=admin,secretfile=/etc/ceph/admin.secret,ms_mode=secure,recover_session=clean" + f"mount -t ceph kclient@.testing={volume_path} /mnt/cephfs -o ms_mode=secure,recover_session=clean" ) kclient.succeed("mountpoint /mnt/cephfs") # Mount CephFS on the FUSE client using ceph-fuse. fuseclient.succeed("mkdir -p /mnt/cephfs") fuseclient.wait_until_succeeds( - "ceph-fuse --id admin -m ${cfg.monA.ip}:6789 /mnt/cephfs" + f"ceph-fuse --id fuseclient -m ${cfg.monA.ip}:3300 -r {volume_path} /mnt/cephfs" ) fuseclient.succeed("mountpoint /mnt/cephfs") @@ -602,24 +611,40 @@ let osd1.crash() osd2.crash() - # Start it up + # Start the mon first and mark the OSDs as down. + # Since the heartbeats are pretty high by default, the OSDs would otherwise be marked as up still. + # However we do not want to lower the heartbeats since this might cause flakey tests. + monA.start() + monA.wait_for_unit("ceph-mon-${cfg.monA.name}") + monA.wait_until_succeeds("ceph osd down all") + # Then start the OSDs as normal. osd0.start() osd1.start() osd2.start() - monA.start() + # Ensure they are all up. + osd0.wait_for_unit("network.target") + osd1.wait_for_unit("network.target") + osd2.wait_for_unit("network.target") + + # FIXME: dmcrypt OSDs currently do not work out of the box. + # For a potential long-term fix see: https://github.com/NixOS/nixpkgs/pull/512912#discussion_r3140295546 + osd1.succeed( + "ceph-volume lvm activate --no-tmpfs --no-systemd ${cfg.osd1.name} ${cfg.osd1.uuid}", + "systemctl start ceph-osd-${cfg.osd1.name}", + ) # Ensure the cluster comes back up again. # See the note above on why this uses `wait_until_succeeds`. monA.wait_until_succeeds("ceph -s | grep 'mon: 1 daemons'") monA.wait_until_succeeds("ceph -s | grep 'quorum ${cfg.monA.name}'") - monA.wait_until_succeeds("ceph osd stat | grep -e '3 osds: 3 up[^,]*, 3 in'") monA.wait_until_succeeds("ceph -s | grep 'mgr: ${cfg.monA.name}(active,'") + monA.wait_until_succeeds("ceph osd stat | grep -e '3 osds: 3 up[^,]*, 3 in'") monA.wait_until_succeeds("ceph -s | grep 'HEALTH_OK'", timeout=60) # Ensure the MDS/CephFS comes back up again, too. monA.wait_for_unit("ceph-mds-${cfg.monA.name}") - monA.wait_until_succeeds("ceph fs status cephfs | grep -e 'active'", timeout=60) + monA.wait_until_succeeds("ceph fs status testing | grep -e 'active'", timeout=60) monA.wait_until_succeeds("ceph -s | grep 'HEALTH_OK'") # The clients kept running across the outage, so their CephFS mounts