From 239a32de717d2195a69cb9b95b7505c0acdc9737 Mon Sep 17 00:00:00 2001 From: Lucian Petrut Date: Fri, 18 Sep 2026 14:39:28 +0000 Subject: [PATCH 1/2] Add Linux-guest HotAdd transport SCSI-attach source VMDKs onto a proxy guest with ReconfigureVM and read them as local SCSI devices, including NVMe/SATA sources remapped onto proxy SCSI. Co-authored-by: Cursor --- README.md | 24 +- docs/hotadd.md | 81 +++ docs/reverse_engineering_procedure.md | 16 +- openvixdisklib/hotadd.py | 724 ++++++++++++++++++++++++++ openvixdisklib/openvixdisklib.py | 102 ++-- tests/integration/base.py | 59 ++- tests/integration/hotadd_proxy.py | 139 +++++ tests/integration/hotadd_remote.py | 84 +++ tests/integration/test_hotadd.py | 184 +++++++ tests/perf/hotadd_remote.py | 80 +++ tests/perf/test_compare.py | 148 +++++- tests/unit/test_hotadd.py | 276 ++++++++++ 12 files changed, 1844 insertions(+), 73 deletions(-) create mode 100644 docs/hotadd.md create mode 100644 openvixdisklib/hotadd.py create mode 100644 tests/integration/hotadd_proxy.py create mode 100644 tests/integration/hotadd_remote.py create mode 100644 tests/integration/test_hotadd.py create mode 100644 tests/perf/hotadd_remote.py create mode 100644 tests/unit/test_hotadd.py diff --git a/README.md b/README.md index 254794b..8b9364f 100644 --- a/README.md +++ b/README.md @@ -13,21 +13,25 @@ Python naming). VIM login and inventory use [pyVmomi](https://github.com/vmware/pyvmomi). The NFC ticket, ESXi authd handshake, and disk I/O were reverse-engineered -from VDDK 8 NBD traffic; see `docs/`. +from VDDK 8 NBD traffic; see `docs/`. Linux HotAdd uses the public +vSphere `ReconfigureVM` API (see `docs/hotadd.md`). ## Status Implemented against vCenter 8 / ESXi 8. Default transport is `nbdssl` -(`nbd` is still available): +(`nbd` is still available). Linux guests can also use `hotadd`: - `VixDiskLib_ConnectEx` (UID credentials) - `VixDiskLib_Open` (datastore path, read-only or read-write) - `VixDiskLib_Read` (optional ``skip_decompression`` packs FastLZ extras) - `VixDiskLib_Write` +- HotAdd on a Linux VMware guest (SCSI, NVMe, or SATA source disks, + attached onto a proxy SCSI controller) Not implemented: compression open flags other than FastLZ, CBT / allocated-block queries, disk geometry (`DDB_GET`), encrypted disks, -and direct ESXi `ha-nfc` without vCenter `vpxa-nfc`. +direct ESXi `ha-nfc` without vCenter `vpxa-nfc`, SAN / file transports, +Windows HotAdd, and HotAdd onto a proxy NVMe controller. Requires Python 3.10 or later. @@ -73,6 +77,7 @@ VDDK-shaped handle. | `openvixdisklib/openvixdisklib.py` | Drop-in handle (`connect` / `open` / `read` / `write`) | | `openvixdisklib/nfc_auth.py` | VIM login, NFC ticket, authd on 902 | | `openvixdisklib/nfc_open.py` | Classic NFC handshake, AIO open, sector read/write | +| `openvixdisklib/hotadd.py` | Linux-guest SCSI HotAdd attach, local block I/O | | `openvixdisklib/fastlz.py` | FastLZ NFC adapter (pip `pyfastlz`) | | `tests/integration/` | Live pytest suite against a lab vCenter | | `tests/perf/` | Throughput comparison of OpenVixDiskLib vs VDDK | @@ -96,11 +101,16 @@ password: secret allow_untrusted: true datacenter: Datacenter datastore: datastore0 +hotadd_proxy: + host: hotadd-proxy.example.com + user: root ``` A session-scoped pytest fixture creates an empty VM with a 10 GiB thin disk on that datastore and tears it down when the session ends. Tests -write known patterns and read them back. +write known patterns and read them back. HotAdd tests SSH into +`hotadd_proxy` (a Linux guest on the same datastore) and skip if SSH +fails. ```bash tox -e integration @@ -117,7 +127,10 @@ tox -e integration -- --runslow Compare write/read throughput of OpenVixDiskLib and native VDDK (`64KiB`, 129-sector, and `32MiB` transfers; `nbdssl` and `nbd`; plain, FastLZ, and OpenVixDiskLib FastLZ ``skip_decompression``; -AIO sessions 64 KiB×1, 1 MiB×1, 2 MiB×1, and 2 MiB×4). +AIO sessions 64 KiB×1, 1 MiB×1, 2 MiB×1, and 2 MiB×4). The same sizes +are also timed over Linux-guest ``hotadd`` (plain OpenVixDiskLib I/O +on `hotadd_proxy`; FastLZ and NFC AIO do not apply) and skipped if +SSH to the proxy fails. ```bash tox -e perf @@ -145,5 +158,6 @@ Lint and typecheck: `tox -e pep8`, `tox -e mypy`. | `docs/nfc_open.md` | Classic NFC and AIO open | | `docs/nfc_read.md` | AIO IO / `VixDiskLib_Read` | | `docs/nfc_write.md` | AIO IO / `VixDiskLib_Write` | +| `docs/hotadd.md` | Linux-guest SCSI HotAdd | | `docs/ssl_hook.md` | TLS intercept used for capture | | `docs/reverse_engineering_procedure.md` | How the protocol was recovered | diff --git a/docs/hotadd.md b/docs/hotadd.md new file mode 100644 index 0000000..6e14ca3 --- /dev/null +++ b/docs/hotadd.md @@ -0,0 +1,81 @@ +# HotAdd transport + +OpenVixDiskLib can SCSI-HotAdd a VMDK onto the Linux guest that is +running the library, then read and write it as a local block device. +This is not an NFC protocol: it uses public VIM `ReconfigureVM` plus +guest SCSI I/O. There is no VixTransport linked clone and no VMDK +parser; ESXi presents a single SCSI LUN. + +NBD and NBDSSL remain the default. `transport_modes=None` is still +`nbdssl`. `hotadd` is advertised and selected only when the process is +a VMware guest (`/sys/class/dmi/id/sys_vendor`). + +## Mapping from VDDK + +| VDDK behaviour | OpenVixDiskLib | +| -------------- | -------------- | +| Run inside a proxy VM | Same. DMI UUID is matched to `config.uuid`. | +| SCSI HotAdd of the source VMDK | `ReconfigureVM` add of an existing backing onto a **SCSI** controller on the proxy | +| Linked clone via VixTransport | Not implemented. The snapshot or base VMDK is attached directly. | +| Open as a whole-disk VMDK | Open `/dev/sdX` with `pread` / `pwrite` | +| IDE disks | Not supported (same as VDDK) | +| NVMe / SATA source disks | Supported. The backing file is attached onto proxy SCSI; the guest sees `/dev/sdX`, not `/dev/nvme*`. | +| HotAdd onto a proxy NVMe controller | Not implemented | + +Colon lists such as `file:san:hotadd:nbdssl:nbd` pick the first **usable** +mode. On a bare-metal host that is `nbdssl`. Inside a guest it is +`hotadd`. `"hotadd"` alone on bare metal raises `NotImplementedError`. + +## Attach and detach + +1. Find this guest in vCenter (`SearchIndex.FindByUuid`). +2. Resolve `disk_path` on the source VM. SCSI, NVMe + (`VirtualNVMEController`), and SATA (`VirtualAHCIController`) are + accepted. IDE and RDM are rejected. A powered-on source VM requires + `snapshot_ref`; a powered-off VM may attach the base disk. +3. Add the existing VMDK to a free SCSI unit on the proxy (unit 7 is + skipped). If every unit is taken, a PVSCSI controller is added. + Read-only opens use `independent_nonpersistent` (redo log, source + stays clean). Writable opens use `persistent`. +4. Rescan SCSI hosts and wait for the device. Matching prefers sysfs + `bus:0:unit:0`, then `*:0:unit:0` when `unit != 0`. +5. `close` detaches with `Operation.remove` and **no** `fileOperation`. + The source VMDK must not be deleted. Leftover attachments of the + same backing are detached before a new open. + +Never HotAdd the proxy's own boot disk. Never use “newest `sdX`” as the +only match when a unique SCSI address exists. + +Do not remove the source VM or its snapshot while the disk is still +attached. Independent-nonpersistent attaches create a redo log on the +source datastore; detach is what cleans it up. + +## API + +`VixDiskLibHandle.connect(..., transport_modes="hotadd")` then +`open` / `read` / `write` / `close` as for NBD. Compression open flags +and NFC `skip_decompression` do not apply; FastLZ flags on a HotAdd +open raise `NotImplementedError`. `readinto` returns a `ReadResult` +with empty `fragments`. + +Implementation: `openvixdisklib.hotadd`. + +## Lab + +Live tests SSH into a Linux proxy that shares the lab datastore and +run `tests/integration/hotadd_remote.py` there. Configure +`.test_config.yaml`: + +```yaml +hotadd_proxy: + host: hotadd-proxy.example.com + user: root + # identity_file: /home/user/.ssh/id_ed25519 +``` + +Tests skip when SSH is unavailable. The session lab VM (PVSCSI) and a +function-scoped NVMe VM are HotAdded onto the proxy, written, and +checked again over `nbdssl` from the runner. `tox -e perf` times the +same transfer sizes over HotAdd (plain I/O; FastLZ and NFC AIO do not +apply). Dependencies on the proxy are installed into +`/tmp/openvixdisklib-hotadd/.venv`, not the system Python. diff --git a/docs/reverse_engineering_procedure.md b/docs/reverse_engineering_procedure.md index b988f5c..c9cce38 100644 --- a/docs/reverse_engineering_procedure.md +++ b/docs/reverse_engineering_procedure.md @@ -9,9 +9,11 @@ NFC work can follow the same loop instead of rediscovering it. Scope so far: `VixDiskLib_ConnectEx` + `VixDiskLib_Open` + `VixDiskLib_Read` + `VixDiskLib_Write` against lab vCenter 8.0.1 / -ESXi 8, transports `nbd` and `nbdssl`. Validation method: -`tests/integration/` (the session-scoped `lab` fixture creates a temporary -empty VM with a 10 GiB disk and destroys it when the pytest session ends). +ESXi 8, transports `nbd`, `nbdssl`, and Linux-guest `hotadd`. +Validation method: `tests/integration/` (the session-scoped `lab` +fixture creates a temporary empty VM with a 10 GiB disk and destroys it +when the pytest session ends). HotAdd live tests also SSH into a Linux +proxy guest; see `docs/hotadd.md`. Rule from `AGENTS.md`: reuse pyVmomi for every public VIM operation. Only reimplement what pyVmomi does not expose. @@ -409,3 +411,11 @@ Not yet reversed, same loop as above: - `VixDiskLib_GetInfo` capacity - Host-switch AIO messages - Direct ESXi `ha-nfc` without vCenter `vpxa-nfc` + +## HotAdd (not NFC) + +HotAdd does not use the capture loop above. VDDK SCSI-attaches the +source VMDK to the proxy VM and opens a local whole disk. OpenVixDiskLib +reuses pyVmomi `ReconfigureVM` for attach/detach and `pread`/`pwrite` on +the Linux SCSI device. NVMe and SATA source disks are remapped onto a +proxy SCSI controller. Details: `docs/hotadd.md`. diff --git a/openvixdisklib/hotadd.py b/openvixdisklib/hotadd.py new file mode 100644 index 0000000..19be8a9 --- /dev/null +++ b/openvixdisklib/hotadd.py @@ -0,0 +1,724 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Linux-guest HotAdd transport: SCSI-attach a VMDK and read it locally. + +The source disk may sit on SCSI, NVMe, or SATA in the backup VM. This +module always HotAdds that backing onto a SCSI controller of the proxy +VM (the guest running this process), then I/Os ``/dev/sdX``. IDE disks +and RDMs are not supported. +""" + +from __future__ import annotations + +import logging +import os +import time +from collections.abc import Callable, Iterable +from dataclasses import dataclass + +from pyVmomi import vim + +from openvixdisklib.nfc_open import ReadResult + +LOG = logging.getLogger(__name__) + +SECTOR_SIZE = 512 +SCSI_RESERVED_UNIT = 7 +SCSI_MAX_UNIT = 15 +SCSI_MAX_BUS = 3 +SCSI_CHANNEL = 0 +SCSI_LUN = 0 +TASK_POLL_S = 0.5 +TASK_TIMEOUT_S = 300 +DEVICE_WAIT_S = 120 +DEVICE_POLL_S = 0.5 +DMI_VENDOR_PATH = "/sys/class/dmi/id/sys_vendor" +DMI_UUID_PATH = "/sys/class/dmi/id/product_uuid" +SCSI_HOST_DIR = "/sys/class/scsi_host" +SCSI_DEVICE_DIR = "/sys/bus/scsi/devices" + + +def is_vmware_guest() -> bool: + """Return True when this process is running in a VMware guest.""" + vendor = _read_sysfs(DMI_VENDOR_PATH) + return vendor is not None and "vmware" in vendor.lower() + + +def _read_sysfs(path: str) -> str | None: + try: + with open(path, encoding="utf-8") as handle: + return handle.read().strip() + except OSError: + return None + + +def guest_uuid() -> str: + """Return the SMBIOS UUID of this guest, or raise if it is missing.""" + uuid = _read_sysfs(DMI_UUID_PATH) + if not uuid: + raise RuntimeError(f"cannot read guest UUID from {DMI_UUID_PATH}") + return uuid + + +def _byteswap_uuid(uuid: str) -> str: + hexpart = uuid.replace("-", "") + if len(hexpart) != 32: + return uuid + + def _rev(field: str) -> str: + return "".join(reversed([field[i : i + 2] for i in range(0, len(field), 2)])) + + swapped = ( + _rev(hexpart[0:8]) + _rev(hexpart[8:12]) + _rev(hexpart[12:16]) + hexpart[16:] + ) + return ( + f"{swapped[0:8]}-{swapped[8:12]}-{swapped[12:16]}-" + f"{swapped[16:20]}-{swapped[20:]}" + ) + + +def find_proxy_vm(si: vim.ServiceInstance) -> vim.VirtualMachine: + """Locate the VM this process is running in via the BIOS UUID.""" + uuid = guest_uuid() + search = si.RetrieveContent().searchIndex + candidates = (uuid, uuid.lower(), uuid.upper(), _byteswap_uuid(uuid)) + seen: set[str] = set() + for candidate in candidates: + if candidate in seen: + continue + seen.add(candidate) + for instance_uuid in (False, True): + vm = search.FindByUuid(None, candidate, True, instance_uuid) + if vm is not None: + return vm + raise RuntimeError(f"no VM in this vCenter has UUID {uuid}") + + +def _controller_map( + devices: Iterable[vim.vm.device.VirtualDevice], +) -> dict[int, vim.vm.device.VirtualController]: + return { + device.key: device + for device in devices + if isinstance(device, vim.vm.device.VirtualController) + } + + +def _is_file_backed(disk: vim.vm.device.VirtualDisk) -> bool: + backing = disk.backing + if backing is None or not getattr(backing, "fileName", None): + return False + name = type(backing).__name__ + return "RawDisk" not in name + + +def _controller_supported(controller: vim.vm.device.VirtualController | None) -> bool: + if controller is None: + return False + if isinstance(controller, vim.vm.device.VirtualIDEController): + return False + return isinstance( + controller, + ( + vim.vm.device.VirtualSCSIController, + vim.vm.device.VirtualNVMEController, + vim.vm.device.VirtualAHCIController, + ), + ) + + +def _snapshot_moref(snapshot_ref: str) -> str: + if "=" in snapshot_ref: + kind, value = snapshot_ref.split("=", 1) + if kind.lower() != "moref" or not value: + raise ValueError(f"unsupported snapshot_ref: {snapshot_ref}") + return value + return snapshot_ref + + +def _walk_snapshots( + trees: list[vim.vm.SnapshotTree] | None, moref: str +) -> vim.vm.SnapshotTree | None: + for tree in trees or []: + if tree.snapshot._moId == moref: + return tree + found = _walk_snapshots(tree.childSnapshotList, moref) + if found is not None: + return found + return None + + +def source_devices( + vm: vim.VirtualMachine, snapshot_ref: str | None +) -> list[vim.vm.device.VirtualDevice]: + """Return hardware devices of ``vm``, or of ``snapshot_ref`` when set.""" + if snapshot_ref: + if vm.snapshot is None: + raise RuntimeError(f"{vm._moId} has no snapshots") + moref = _snapshot_moref(snapshot_ref) + tree = _walk_snapshots(vm.snapshot.rootSnapshotList, moref) + if tree is None: + raise RuntimeError(f"snapshot {moref} not found on {vm._moId}") + return list(tree.config.hardware.device) + return list(vm.config.hardware.device) + + +def find_source_disk( + devices: Iterable[vim.vm.device.VirtualDevice], disk_path: str +) -> vim.vm.device.VirtualDisk: + """Return the file-backed disk whose backing path is ``disk_path``. + + SCSI, NVMe, and SATA controllers are accepted. IDE and RDM backings + raise ``NotImplementedError``. + """ + controllers = _controller_map(devices) + for device in devices: + if not isinstance(device, vim.vm.device.VirtualDisk): + continue + backing = device.backing + if getattr(backing, "fileName", None) != disk_path: + continue + if not _is_file_backed(device): + raise NotImplementedError( + f"HotAdd does not support RDM or raw backings: {disk_path}" + ) + controller = controllers.get(device.controllerKey) + if isinstance(controller, vim.vm.device.VirtualIDEController): + raise NotImplementedError(f"HotAdd does not support IDE disks: {disk_path}") + if not _controller_supported(controller): + kind = type(controller).__name__ if controller else "missing controller" + raise NotImplementedError( + f"HotAdd does not support {kind} disks: {disk_path}" + ) + return device + raise FileNotFoundError(f"no virtual disk with backing {disk_path!r}") + + +def _scsi_controllers( + devices: Iterable[vim.vm.device.VirtualDevice], +) -> list[vim.vm.device.VirtualSCSIController]: + return [ + device + for device in devices + if isinstance(device, vim.vm.device.VirtualSCSIController) + ] + + +def _used_units( + devices: Iterable[vim.vm.device.VirtualDevice], controller_key: int +) -> set[int]: + return { + device.unitNumber + for device in devices + if isinstance(device, vim.vm.device.VirtualDisk) + and device.controllerKey == controller_key + and device.unitNumber is not None + } + + +def pick_scsi_slot( + devices: Iterable[vim.vm.device.VirtualDevice], +) -> tuple[vim.vm.device.VirtualSCSIController | None, int, int]: + """Return ``(controller, bus, unit)`` for a free SCSI slot. + + ``controller`` is ``None`` when a new PVSCSI controller must be + added on ``bus``; ``unit`` is then 0. + """ + device_list = list(devices) + for controller in sorted(_scsi_controllers(device_list), key=lambda c: c.busNumber): + used = _used_units(device_list, controller.key) + for unit in range(SCSI_MAX_UNIT + 1): + if unit == SCSI_RESERVED_UNIT: + continue + if unit not in used: + return controller, int(controller.busNumber), unit + used_buses = {controller.busNumber for controller in _scsi_controllers(device_list)} + for bus in range(SCSI_MAX_BUS + 1): + if bus not in used_buses: + return None, bus, 0 + raise RuntimeError("no free SCSI controller bus on the HotAdd proxy") + + +@dataclass +class AttachPlan: + """ReconfigureVM spec plus the SCSI address the guest should see.""" + + spec: vim.vm.ConfigSpec + bus_number: int + unit_number: int + file_name: str + + +def build_attach_spec( + devices: Iterable[vim.vm.device.VirtualDevice], + file_name: str, + read_only: bool, + capacity_kb: int | None = None, +) -> AttachPlan: + """Build a SCSI HotAdd spec for an existing VMDK backing. + + Never sets ``fileOperation`` (the VMDK already exists). Read-only + opens use ``independent_nonpersistent``; writable opens use + ``persistent``. + """ + device_list = list(devices) + controller, bus, unit = pick_scsi_slot(device_list) + changes: list[vim.vm.device.VirtualDeviceSpec] = [] + if controller is None: + new_controller = vim.vm.device.ParaVirtualSCSIController() + new_controller.key = -101 + new_controller.busNumber = bus + new_controller.sharedBus = vim.vm.device.VirtualSCSIController.Sharing.noSharing + if hasattr(new_controller, "hotAddRemove"): + new_controller.hotAddRemove = True + controller_spec = vim.vm.device.VirtualDeviceSpec() + controller_spec.operation = vim.vm.device.VirtualDeviceSpec.Operation.add + controller_spec.device = new_controller + changes.append(controller_spec) + controller_key = new_controller.key + else: + controller_key = controller.key + + backing = vim.vm.device.VirtualDisk.FlatVer2BackingInfo() + backing.fileName = file_name + backing.diskMode = "independent_nonpersistent" if read_only else "persistent" + + disk = vim.vm.device.VirtualDisk() + disk.key = -201 + disk.controllerKey = controller_key + disk.unitNumber = unit + disk.backing = backing + if capacity_kb: + disk.capacityInKB = capacity_kb + disk.deviceInfo = vim.Description() + disk.deviceInfo.label = "openvixdisklib-hotadd" + disk.deviceInfo.summary = file_name + + disk_spec = vim.vm.device.VirtualDeviceSpec() + disk_spec.operation = vim.vm.device.VirtualDeviceSpec.Operation.add + disk_spec.device = disk + changes.append(disk_spec) + + spec = vim.vm.ConfigSpec() + spec.deviceChange = changes + return AttachPlan(spec=spec, bus_number=bus, unit_number=unit, file_name=file_name) + + +def build_detach_spec(device: vim.vm.device.VirtualDisk) -> vim.vm.ConfigSpec: + """Build a remove spec that detaches ``device`` without deleting files.""" + change = vim.vm.device.VirtualDeviceSpec() + change.operation = vim.vm.device.VirtualDeviceSpec.Operation.remove + change.device = device + spec = vim.vm.ConfigSpec() + spec.deviceChange = [change] + return spec + + +def _wait_for_task(task: vim.Task) -> object: + deadline = time.monotonic() + TASK_TIMEOUT_S + while task.info.state in (vim.TaskInfo.State.running, vim.TaskInfo.State.queued): + if time.monotonic() > deadline: + raise TimeoutError(f"timed out waiting for vSphere task {task}") + time.sleep(TASK_POLL_S) + if task.info.state != vim.TaskInfo.State.success: + raise RuntimeError(f"vSphere task failed: {task.info.error}") + return task.info.result + + +def _boot_disk_path(devices: Iterable[vim.vm.device.VirtualDevice]) -> str | None: + for device in devices: + if isinstance(device, vim.vm.device.VirtualDisk): + return getattr(device.backing, "fileName", None) + return None + + +def _disks_with_backing( + devices: Iterable[vim.vm.device.VirtualDevice], file_name: str +) -> list[vim.vm.device.VirtualDisk]: + matches = [] + for device in devices: + if not isinstance(device, vim.vm.device.VirtualDisk): + continue + if getattr(device.backing, "fileName", None) == file_name: + matches.append(device) + return matches + + +def _reconfigure(vm: vim.VirtualMachine, spec: vim.vm.ConfigSpec) -> None: + _wait_for_task(vm.ReconfigVM_Task(spec)) + vm.Reload() + + +def _detach_device(vm: vim.VirtualMachine, device: vim.vm.device.VirtualDisk) -> None: + LOG.info( + "HotAdd detach %s unit=%s from %s", + getattr(device.backing, "fileName", None), + device.unitNumber, + vm._moId, + ) + _reconfigure(vm, build_detach_spec(device)) + + +def _scsi_sysfs(bus: int, unit: int) -> str: + return os.path.join( + SCSI_DEVICE_DIR, f"{bus}:{SCSI_CHANNEL}:{unit}:{SCSI_LUN}", "block" + ) + + +def _block_names(sysfs_dir: str) -> list[str]: + try: + return [ + name + for name in os.listdir(sysfs_dir) + if not name.startswith(".") and os.path.isdir(os.path.join(sysfs_dir, name)) + ] + except OSError: + return [] + + +def _scsi_block_dirs(unit: int) -> list[str]: + """Return sysfs ``block`` dirs for SCSI target ``unit`` (any host).""" + found: list[str] = [] + try: + names = os.listdir(SCSI_DEVICE_DIR) + except OSError: + return found + suffix = f":{SCSI_CHANNEL}:{unit}:{SCSI_LUN}" + for name in names: + if not name.endswith(suffix): + continue + block = os.path.join(SCSI_DEVICE_DIR, name, "block") + if os.path.isdir(block): + found.append(block) + return found + + +def list_scsi_block_devices() -> set[str]: + """Return guest ``/dev`` paths for every SCSI block device.""" + found: set[str] = set() + try: + names = os.listdir(SCSI_DEVICE_DIR) + except OSError: + return found + for name in names: + block = os.path.join(SCSI_DEVICE_DIR, name, "block") + for dev in _block_names(block): + found.add(f"/dev/{dev}") + return found + + +def find_scsi_block_device( + bus: int, unit: int, before: set[str] | None = None +) -> str | None: + """Return ``/dev/sdX`` for the HotAdded SCSI disk, if present. + + Linux SCSI host numbers often do not match VMware bus numbers. + Matching uses ``/sys/bus/scsi/devices/:0::0/block``. + """ + exact = _scsi_sysfs(bus, unit) + names = _block_names(exact) + if names: + return f"/dev/{names[0]}" + matches = _scsi_block_dirs(unit) + candidates: list[str] = [] + for block_dir in matches: + candidates.extend(f"/dev/{name}" for name in _block_names(block_dir)) + if before is not None: + new = [path for path in candidates if path not in before] + if len(new) == 1: + return new[0] + appeared = list_scsi_block_devices() - before + if len(appeared) == 1: + return appeared.pop() + if len(candidates) == 1: + return candidates[0] + return None + + +def rescan_scsi_hosts() -> None: + """Ask every SCSI host to scan for new LUNs.""" + if not os.path.isdir(SCSI_HOST_DIR): + return + for host in os.listdir(SCSI_HOST_DIR): + scan = os.path.join(SCSI_HOST_DIR, host, "scan") + try: + with open(scan, "w", encoding="ascii") as handle: + handle.write("- - -\n") + except OSError: + continue + + +def wait_for_scsi_device( + bus: int, + unit: int, + timeout_s: float = DEVICE_WAIT_S, + before: set[str] | None = None, +) -> str: + """Rescan SCSI and wait until the HotAdded disk has a block device.""" + deadline = time.monotonic() + timeout_s + last: str | None = None + while time.monotonic() < deadline: + rescan_scsi_hosts() + last = find_scsi_block_device(bus, unit, before=before) + if last and os.path.exists(last): + return last + time.sleep(DEVICE_POLL_S) + raise TimeoutError( + f"HotAdded disk did not appear at SCSI {bus}:0:{unit}:0 ({last})" + ) + + +def _offline_scsi_unit(unit: int) -> None: + """Ask Linux to drop SCSI devices with target ``unit``.""" + for block_dir in _scsi_block_dirs(unit): + delete_path = os.path.join(os.path.dirname(block_dir), "delete") + try: + with open(delete_path, "w", encoding="ascii") as handle: + handle.write("1\n") + except OSError: + continue + + +def wait_scsi_device_gone( + bus: int, unit: int, timeout_s: float = DEVICE_WAIT_S +) -> None: + """Wait until the SCSI device sysfs node disappears after detach.""" + del bus + _offline_scsi_unit(unit) + deadline = time.monotonic() + timeout_s + while time.monotonic() < deadline: + if not _scsi_block_dirs(unit): + return + _offline_scsi_unit(unit) + time.sleep(DEVICE_POLL_S) + raise TimeoutError(f"HotAdded disk still present at SCSI unit {unit}") + + +def _pread_all(fd: int, size: int, offset: int) -> bytes: + chunks = bytearray() + remaining = size + pos = offset + while remaining: + data = os.pread(fd, remaining, pos) + if not data: + raise OSError(f"short read at offset {pos}: got {len(chunks)} of {size}") + chunks.extend(data) + remaining -= len(data) + pos += len(data) + return bytes(chunks) + + +def _pwrite_all(fd: int, data: bytes, offset: int) -> None: + remaining = memoryview(data) + pos = offset + while remaining: + written = os.pwrite(fd, remaining, pos) + if written <= 0: + raise OSError(f"short write at offset {pos}") + remaining = remaining[written:] + pos += written + + +class HotAddDisk: + """A locally attached HotAdd VMDK opened as a SCSI block device.""" + + def __init__( + self, + fd: int, + dev_path: str, + bus_number: int, + unit_number: int, + detach: Callable[[], None], + sector_size: int = SECTOR_SIZE, + ) -> None: + """Wrap an open block-device fd and a detach callback. + + Args: + fd: File descriptor for the SCSI disk. + dev_path: Guest path such as ``/dev/sdb``. + bus_number: VMware SCSI bus of the attached disk. + unit_number: VMware SCSI unit of the attached disk. + detach: Called from ``close`` after the fd is closed. + sector_size: Sector size in bytes (VDDK uses 512). + """ + self._fd = fd + self.dev_path = dev_path + self.bus_number = bus_number + self.unit_number = unit_number + self._detach = detach + self.sector_size = sector_size + self._closed = False + + def readinto( + self, + start_sector: int, + num_sectors: int, + buf: bytearray | memoryview, + skip_decompression: bool = False, + ) -> ReadResult: + """Read ``num_sectors`` into ``buf`` starting at ``start_sector``. + + ``skip_decompression`` is an NFC option and is ignored; HotAdd + has no compressed extras. ``fragments`` is always empty. + + Args: + start_sector: Sector offset from the start of the disk. + num_sectors: Number of sectors to read. + buf: Destination buffer. + skip_decompression: Ignored; accepted for API compatibility. + """ + del skip_decompression + if num_sectors < 1: + raise ValueError("num_sectors must be at least 1") + length = num_sectors * self.sector_size + view = buf if isinstance(buf, memoryview) else memoryview(buf) + if view.readonly: + raise TypeError("read buffer is read-only") + raw = view.cast("B") if view.format != "B" else view + if len(raw) < length: + raise RuntimeError(f"read buffer is {len(raw)} bytes, need {length}") + data = _pread_all(self._fd, length, start_sector * self.sector_size) + raw[:length] = data + return ReadResult( + uncompressed_length=length, compressed_length=length, fragments=() + ) + + def write(self, start_sector: int, num_sectors: int, data: bytes) -> None: + """Write ``num_sectors`` starting at ``start_sector``. + + Args: + start_sector: Sector offset from the start of the disk. + num_sectors: Number of sectors to write. + data: Bytes to write; length must be ``num_sectors * sector_size``. + """ + if num_sectors < 1: + raise ValueError("num_sectors must be at least 1") + length = num_sectors * self.sector_size + if len(data) != length: + raise ValueError(f"write data is {len(data)} bytes, need {length}") + _pwrite_all(self._fd, data, start_sector * self.sector_size) + os.fsync(self._fd) + + def close(self) -> None: + """Close the block device and detach the VMDK from the proxy.""" + if self._closed: + return + try: + os.close(self._fd) + except OSError: + pass + self._detach() + self._closed = True + + def __enter__(self) -> HotAddDisk: + return self + + def __exit__(self, exc_type, exc, tb) -> None: + self.close() + + +def open_disk( + si: vim.ServiceInstance, + source_vm: vim.VirtualMachine, + disk_path: str, + snapshot_ref: str | None = None, + read_only: bool = True, +) -> HotAddDisk: + """HotAdd ``disk_path`` from ``source_vm`` onto this guest and open it. + + The source VM must be powered off, or ``snapshot_ref`` must name a + snapshot whose hardware contains ``disk_path``. The disk is always + attached to a SCSI controller on the proxy. + + Args: + si: Logged-in VIM session. + source_vm: VM that owns ``disk_path``. + disk_path: Datastore path of the VMDK. + snapshot_ref: Snapshot moref required when ``source_vm`` is on. + read_only: Independent-nonpersistent attach when True. + """ + if not is_vmware_guest(): + raise RuntimeError("HotAdd requires a VMware guest (the backup proxy)") + if ( + source_vm.runtime.powerState == vim.VirtualMachinePowerState.poweredOn + and not snapshot_ref + ): + raise RuntimeError( + "snapshot_ref is required to HotAdd a powered-on virtual machine" + ) + + proxy = find_proxy_vm(si) + proxy_devices = list(proxy.config.hardware.device) + boot_path = _boot_disk_path(proxy_devices) + if boot_path == disk_path: + raise RuntimeError("refusing to HotAdd the proxy VM's boot disk") + + for leftover in _disks_with_backing(proxy_devices, disk_path): + LOG.warning("detaching leftover HotAdd disk %s from %s", disk_path, proxy._moId) + _detach_device(proxy, leftover) + proxy_devices = list(proxy.config.hardware.device) + + devices = source_devices(source_vm, snapshot_ref) + source = find_source_disk(devices, disk_path) + capacity = getattr(source, "capacityInKB", None) + plan = build_attach_spec( + proxy_devices, disk_path, read_only=read_only, capacity_kb=capacity + ) + LOG.info( + "HotAdd attach %s onto %s SCSI %s:%s read_only=%s", + disk_path, + proxy._moId, + plan.bus_number, + plan.unit_number, + read_only, + ) + attached: vim.vm.device.VirtualDisk | None = None + before = list_scsi_block_devices() + try: + _reconfigure(proxy, plan.spec) + matches = _disks_with_backing(proxy.config.hardware.device, disk_path) + if len(matches) != 1: + raise RuntimeError( + f"expected one attached disk {disk_path!r}, found {len(matches)}" + ) + attached = matches[0] + bus = plan.bus_number + unit = plan.unit_number + controllers = _controller_map(proxy.config.hardware.device) + controller = controllers.get(attached.controllerKey) + if isinstance(controller, vim.vm.device.VirtualSCSIController): + bus = int(controller.busNumber) + unit = int(attached.unitNumber) + dev_path = wait_for_scsi_device(bus, unit, before=before) + flags = os.O_RDONLY if read_only else os.O_RDWR + fd = os.open(dev_path, flags) + except Exception: + victim = attached + if victim is None: + leftovers = _disks_with_backing(proxy.config.hardware.device, disk_path) + victim = leftovers[0] if leftovers else None + if victim is not None: + _detach_best_effort(proxy, victim) + raise + + def _detach() -> None: + try: + _detach_device(proxy, attached) + finally: + wait_scsi_device_gone(bus, unit) + + return HotAddDisk(fd, dev_path, bus, unit, _detach) + + +def _detach_best_effort( + vm: vim.VirtualMachine, device: vim.vm.device.VirtualDisk +) -> None: + """Detach ``device`` after a failed open; log and ignore errors.""" + try: + _detach_device(vm, device) + unit = device.unitNumber + if unit is not None: + _offline_scsi_unit(int(unit)) + except Exception: + LOG.exception("HotAdd cleanup after failed open") diff --git a/openvixdisklib/openvixdisklib.py b/openvixdisklib/openvixdisklib.py index cafb7f9..cb91948 100644 --- a/openvixdisklib/openvixdisklib.py +++ b/openvixdisklib/openvixdisklib.py @@ -10,7 +10,8 @@ ``VixDiskLibHandle.connect`` / ``open`` / ``read`` match the VDDK wrapper in ``tests/integration/vixdisklib.py``. VIM login uses pyVmomi; NFC ticket, -authd, and disk I/O use ``nfc_auth`` and ``nfc_open``. +authd, and disk I/O use ``nfc_auth`` and ``nfc_open``. Linux HotAdd uses +``hotadd``. """ from __future__ import annotations @@ -24,7 +25,7 @@ from pyVim.connect import Disconnect from pyVmomi import vim -from openvixdisklib import nfc_auth, nfc_open +from openvixdisklib import hotadd, nfc_auth, nfc_open ReadResult = nfc_open.ReadResult ReadFragment = nfc_open.ReadFragment @@ -84,20 +85,31 @@ def _parse_vm_moref(vmx_spec: str | None) -> str: return vmx_spec +def _available_transports() -> list[str]: + """Return transports this process can use, in advertisement order.""" + modes = ["nbdssl", "nbd"] + if hotadd.is_vmware_guest(): + modes.append("hotadd") + return modes + + def _select_transport(transport_modes: str | None) -> str: - """Return the first requested transport this replacement implements. + """Return the first requested transport this replacement can use. ``None`` defaults to ``nbdssl``. A colon-separated list (VDDK - style, for example ``file:nbdssl:nbd``) picks the first of - ``nbdssl`` or ``nbd``. + style, for example ``file:san:hotadd:nbdssl:nbd``) picks the first + of ``nbdssl``, ``nbd``, and ``hotadd`` that is usable here. + ``hotadd`` is usable only inside a VMware guest. """ if transport_modes is None: return "nbdssl" + usable = set(_available_transports()) for mode in transport_modes.split(":"): - if mode in ("nbdssl", "nbd"): + if mode in usable: return mode raise NotImplementedError( - f"supported transports are nbdssl and nbd, got {transport_modes!r}" + f"supported transports are {' and '.join(_available_transports())}, " + f"got {transport_modes!r}" ) @@ -124,9 +136,14 @@ def __init__( class _DiskHandle: - """Opened NFC disk plus the authd TLS socket it was taken from.""" + """Opened disk (NFC or HotAdd) plus an optional authd TLS socket.""" - def __init__(self, disk: nfc_open.NfcDisk, authd_sock, transport_mode: str) -> None: + def __init__( + self, + disk: nfc_open.NfcDisk | hotadd.HotAddDisk, + transport_mode: str, + authd_sock=None, + ) -> None: self.disk = disk self.authd_sock = authd_sock self.transport_mode = transport_mode @@ -178,8 +195,8 @@ def get_vix_disklib_name(cls) -> str: return "libvixDiskLib.so" def get_transport_modes(self) -> list[str]: - """Return the transport modes this replacement implements.""" - return ["nbdssl", "nbd"] + """Return the transport modes this process can use.""" + return _available_transports() def get_transport_mode(self, disk_handle: _DiskHandle) -> str: """Return the transport used for ``disk_handle``.""" @@ -214,11 +231,13 @@ def connect( username: VIM user name. password: VIM password. vmx_spec: VM selector, ``moref=vm-…``. - snapshot_ref: Snapshot moref; unused on the NFC ticket. + snapshot_ref: Snapshot moref. Unused on the NFC ticket. + Required for HotAdd when the source VM is powered on. read_only: When False, the disk may be opened for write. - transport_modes: ``nbdssl``, ``nbd``, or a colon list. The - first supported mode is used; ``None`` defaults to - ``nbdssl``. + transport_modes: ``nbdssl``, ``nbd``, ``hotadd``, or a colon + list. The first usable mode is used; ``None`` defaults + to ``nbdssl``. ``hotadd`` is usable only in a VMware + guest. port: HTTPS port, usually 443. allow_untrusted: Skip management TLS verification when True. When False with no ``thumbprint``, the system CA store @@ -270,13 +289,14 @@ def open( aio_buffer_size: int = nfc_open.NFC_AIO_BUFFER_SIZE, aio_buffer_count: int = nfc_open.NFC_AIO_BUFFER_COUNT, ) -> Iterator[_DiskHandle]: - """Open ``disk_path`` over NFC. Matches ``VixDiskLib_Open``. + """Open ``disk_path`` over NFC or HotAdd. Matches ``VixDiskLib_Open``. - Read-only opens request ``NfcGetVmFiles`` (VM only). The VMDK - path, including a snapshot parent such as ``…-000007.vmdk``, is - sent on NFC ``OPEN_FILE``. Writable opens use + Read-only NFC opens request ``NfcGetVmFiles`` (VM only). The + VMDK path, including a snapshot parent such as ``…-000007.vmdk``, + is sent on NFC ``OPEN_FILE``. Writable NFC opens use ``NfcRandomAccessOpenDisk`` and resolve a device key from the - disk's backing chain. + disk's backing chain. ``hotadd`` SCSI-attaches the VMDK to this + guest (Linux proxy) and opens the local block device. Args: conn: Connection from ``connect``. @@ -284,14 +304,16 @@ def open( flags: Open flags. ``VIXDISKLIB_FLAG_OPEN_READ_ONLY`` opens the disk read-only; omit it for write. ``VIXDISKLIB_FLAG_OPEN_COMPRESSION_FASTLZ`` compresses - NFC IO. zlib and skipz are not implemented. + NFC IO. zlib and skipz are not implemented. Compression + flags are not supported with ``hotadd``. aio_buffer_size: NFC AIO extra size in bytes, advertised in OPEN_SESSION. Default 64 KiB. ESXi 8 accepts 2 MiB (``2097152``) and rejects 16 MiB and 32 MiB. This is an OpenVixDiskLib extension (VDDK uses - ``vixDiskLib.nfcAio.Session.BufSizeIn64KB``). + ``vixDiskLib.nfcAio.Session.BufSizeIn64KB``). Ignored + for HotAdd. aio_buffer_count: NFC AIO buffer pool count. Default 1. - VDDK's default is 4. + VDDK's default is 4. Ignored for HotAdd. """ LOG.debug("Openning VixDiskLib disk: %s", disk_path) compression = _nfc_compression(flags) @@ -300,6 +322,25 @@ def open( raise NotImplementedError("ConnectEx was read-only; cannot open for write") vm = vim.VirtualMachine(conn.vm_moref, conn.si._stub) + if conn.transport_mode == "hotadd": + if compression != nfc_open.NFC_COMPRESSION_NONE: + raise NotImplementedError( + "NBD compression open flags are not supported with hotadd" + ) + disk = hotadd.open_disk( + conn.si, + vm, + disk_path, + snapshot_ref=conn.snapshot_ref, + read_only=read_only, + ) + handle = _DiskHandle(disk, conn.transport_mode) + try: + yield handle + finally: + self.close(handle) + return + nfc_ssl = conn.transport_mode == "nbdssl" ticket = nfc_auth.get_nfc_ticket( conn.si, vm, read_only=read_only, disk_path=None if read_only else disk_path @@ -309,7 +350,7 @@ def open( ) session = nfc_auth.NfcAuthSession(conn.si, ticket, authd_sock, nfc_ssl=nfc_ssl) try: - disk = nfc_open.open_disk( + nfc_disk = nfc_open.open_disk( session, disk_path, read_only=read_only, @@ -320,7 +361,7 @@ def open( except Exception: authd_sock.close() raise - handle = _DiskHandle(disk, authd_sock, conn.transport_mode) + handle = _DiskHandle(nfc_disk, conn.transport_mode, authd_sock) try: yield handle finally: @@ -386,7 +427,7 @@ def write( disk_handle.disk.write(start_sector, num_sectors, data) def close(self, disk_handle: _DiskHandle) -> None: - """Close the VMDK and the authd socket used for NFC. + """Close the VMDK and, for NFC, the authd socket. Args: disk_handle: Handle from ``open``. @@ -395,10 +436,11 @@ def close(self, disk_handle: _DiskHandle) -> None: try: disk_handle.disk.close() finally: - try: - disk_handle.authd_sock.close() - except OSError: - pass + if disk_handle.authd_sock is not None: + try: + disk_handle.authd_sock.close() + except OSError: + pass def disconnect(self, conn: _Connection) -> None: """Logout of the VIM session. diff --git a/tests/integration/base.py b/tests/integration/base.py index 57157fc..d32efa7 100644 --- a/tests/integration/base.py +++ b/tests/integration/base.py @@ -144,6 +144,28 @@ def _load_test_config() -> dict[str, Any]: } +def load_hotadd_proxy_config() -> dict[str, str] | None: + """Return optional SSH settings for the Linux HotAdd proxy, if configured.""" + if not os.path.isfile(_CONFIG_PATH): + return None + with open(_CONFIG_PATH, encoding="utf-8") as config_file: + data = yaml.safe_load(config_file) or {} + proxy = data.get("hotadd_proxy") + if not isinstance(proxy, dict) or not proxy.get("host"): + return None + identity = proxy.get("identity_file") + if identity: + identity_file = os.path.expanduser(str(identity)) + else: + default_key = os.path.expanduser("~/.ssh/id_ed25519") + identity_file = default_key if os.path.isfile(default_key) else "" + return { + "host": str(proxy["host"]), + "user": str(proxy.get("user", "root")), + "identity_file": identity_file, + } + + def _connect_vim( host: str, username: str, @@ -199,7 +221,9 @@ def _find_datastore(datacenter: vim.Datacenter, datastore_name: str) -> vim.Data return matches[0] -def _vm_config_spec(vm_name: str, datastore_name: str) -> vim.vm.ConfigSpec: +def _vm_config_spec( + vm_name: str, datastore_name: str, disk_controller: str = "pvscsi" +) -> vim.vm.ConfigSpec: config = vim.vm.ConfigSpec() config.name = vm_name config.guestId = "otherGuest64" @@ -207,10 +231,21 @@ def _vm_config_spec(vm_name: str, datastore_name: str) -> vim.vm.ConfigSpec: config.numCPUs = 1 config.files = vim.vm.FileInfo(vmPathName=f"[{datastore_name}]") - controller = vim.vm.device.ParaVirtualSCSIController() - controller.key = 1000 - controller.busNumber = 0 - controller.sharedBus = vim.vm.device.VirtualSCSIController.Sharing.noSharing + if disk_controller == "nvme": + controller: vim.vm.device.VirtualController = ( + vim.vm.device.VirtualNVMEController() + ) + controller.key = 1000 + controller.busNumber = 0 + elif disk_controller == "pvscsi": + scsi = vim.vm.device.ParaVirtualSCSIController() + scsi.key = 1000 + scsi.busNumber = 0 + scsi.sharedBus = vim.vm.device.VirtualSCSIController.Sharing.noSharing + controller = scsi + else: + raise ValueError(f"unsupported disk_controller: {disk_controller}") + controller_spec = vim.vm.device.VirtualDeviceSpec() controller_spec.operation = vim.vm.device.VirtualDeviceSpec.Operation.add controller_spec.device = controller @@ -234,8 +269,12 @@ def _vm_config_spec(vm_name: str, datastore_name: str) -> vim.vm.ConfigSpec: return config -def create_lab_vm() -> LabEnv: - """Create an empty VM with a 10 GiB thin disk for I/O tests.""" +def create_lab_vm(*, disk_controller: str = "pvscsi") -> LabEnv: + """Create an empty VM with a 10 GiB thin disk for I/O tests. + + Args: + disk_controller: ``pvscsi`` (default) or ``nvme``. + """ cfg = _load_test_config() thumbprint = nfc_auth.get_ssl_cert_thumbprint(cfg["host"], cfg["port"]) si = _connect_vim( @@ -260,7 +299,11 @@ def create_lab_vm() -> LabEnv: vm_name = _LAB_VM_PREFIX + uuid.uuid4().hex[:12] vm = _wait_for_task( datacenter.vmFolder.CreateVM_Task( - config=_vm_config_spec(vm_name, datastore.name), pool=pool, host=host + config=_vm_config_spec( + vm_name, datastore.name, disk_controller=disk_controller + ), + pool=pool, + host=host, ) ) disks = [ diff --git a/tests/integration/hotadd_proxy.py b/tests/integration/hotadd_proxy.py new file mode 100644 index 0000000..82ba0ab --- /dev/null +++ b/tests/integration/hotadd_proxy.py @@ -0,0 +1,139 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""SSH helpers for running OpenVixDiskLib HotAdd on the Linux proxy.""" + +from __future__ import annotations + +import os +import subprocess + +from tests.integration.base import load_hotadd_proxy_config + +_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) +REMOTE_DIR = "/tmp/openvixdisklib-hotadd" +REMOTE_VENV = f"{REMOTE_DIR}/.venv" +REMOTE_PYTHON = f"{REMOTE_VENV}/bin/python" +SSH_CONNECT_TIMEOUT_S = 15 +DEFAULT_SSH_TIMEOUT_S = 300 + + +def ssh_base(proxy: dict[str, str]) -> list[str]: + """Return the ``ssh user@host`` prefix for ``proxy``.""" + cmd = [ + "ssh", + "-o", + "BatchMode=yes", + "-o", + "StrictHostKeyChecking=accept-new", + "-o", + f"ConnectTimeout={SSH_CONNECT_TIMEOUT_S}", + ] + if proxy.get("identity_file"): + cmd.extend(["-i", proxy["identity_file"]]) + cmd.append(f"{proxy['user']}@{proxy['host']}") + return cmd + + +def ssh_proxy( + proxy: dict[str, str], + remote: str, + *, + stdin: bytes | None = None, + timeout: int = DEFAULT_SSH_TIMEOUT_S, +) -> subprocess.CompletedProcess[bytes]: + """Run ``remote`` on the HotAdd proxy and return the completed process.""" + return subprocess.run( + [*ssh_base(proxy), remote], + input=stdin, + capture_output=True, + timeout=timeout, + check=False, + ) + + +def prepare_hotadd_proxy( + extra_files: dict[str, str] | None = None, +) -> dict[str, str]: + """Probe SSH, sync ``openvixdisklib``, and return proxy settings. + + ``extra_files`` maps a remote basename under ``REMOTE_DIR`` to a + local path that is copied after the package. Raises ``RuntimeError`` + when the proxy is missing or unreachable. + """ + proxy = load_hotadd_proxy_config() + if proxy is None: + raise RuntimeError("hotadd_proxy missing from .test_config.yaml") + probe = ssh_proxy(proxy, "echo ok") + if probe.returncode != 0: + raise RuntimeError( + f"cannot ssh to {proxy['user']}@{proxy['host']}: " + f"{probe.stderr.decode(errors='replace').strip()}" + ) + _sync_package(proxy, extra_files or {}) + return proxy + + +def _sync_package(proxy: dict[str, str], extra_files: dict[str, str]) -> None: + mkdir = ssh_proxy(proxy, f"mkdir -p {REMOTE_DIR}/openvixdisklib") + if mkdir.returncode != 0: + raise RuntimeError( + f"mkdir on proxy failed: {mkdir.stderr.decode(errors='replace')}" + ) + archive = subprocess.run( + [ + "tar", + "-C", + os.path.join(_REPO_ROOT, "openvixdisklib"), + "-czf", + "-", + ".", + ], + capture_output=True, + check=False, + ) + if archive.returncode != 0: + raise RuntimeError( + f"tar package failed: {archive.stderr.decode(errors='replace')}" + ) + unpack = ssh_proxy( + proxy, + f"rm -rf {REMOTE_DIR}/openvixdisklib && mkdir -p {REMOTE_DIR}/openvixdisklib " + f"&& tar -C {REMOTE_DIR}/openvixdisklib -xzf -", + stdin=archive.stdout, + ) + if unpack.returncode != 0: + raise RuntimeError( + f"copy package to proxy failed: {unpack.stderr.decode(errors='replace')}" + ) + for remote_name, local_path in extra_files.items(): + with open(local_path, "rb") as handle: + contents = handle.read() + copy = ssh_proxy(proxy, f"cat > {REMOTE_DIR}/{remote_name}", stdin=contents) + if copy.returncode != 0: + raise RuntimeError( + f"copy {remote_name} to proxy failed: " + f"{copy.stderr.decode(errors='replace')}" + ) + venv = ssh_proxy( + proxy, + f"test -x {REMOTE_PYTHON} || python3 -m venv {REMOTE_VENV}", + ) + if venv.returncode != 0: + raise RuntimeError( + f"could not create proxy venv: {venv.stderr.decode(errors='replace')}" + ) + deps = ssh_proxy( + proxy, + f"{REMOTE_PYTHON} -c 'import pyVmomi, pyVim, fastlz'", + ) + if deps.returncode != 0: + install = ssh_proxy( + proxy, + f"{REMOTE_PYTHON} -m pip install 'pyVmomi>=7.0' pyOpenSSL pyfastlz", + ) + if install.returncode != 0: + raise RuntimeError( + "proxy venv pip install failed: " + f"{install.stderr.decode(errors='replace')}" + ) diff --git a/tests/integration/hotadd_remote.py b/tests/integration/hotadd_remote.py new file mode 100644 index 0000000..3820923 --- /dev/null +++ b/tests/integration/hotadd_remote.py @@ -0,0 +1,84 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Run HotAdd I/O inside the proxy guest. Invoked over SSH by tests.""" + +from __future__ import annotations + +import json +import sys + +from pyVmomi import vim + +from openvixdisklib import openvixdisklib as vixdisklib +from openvixdisklib.hotadd import find_proxy_vm + + +def main() -> int: + """Read connect kwargs and sector patterns from stdin, HotAdd, write/read.""" + cfg = json.load(sys.stdin) + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + modes = handle.get_transport_modes() + if "hotadd" not in modes: + print(json.dumps({"ok": False, "error": f"hotadd not listed: {modes}"})) + return 1 + patterns = { + int(sector): bytes.fromhex(data) for sector, data in cfg["patterns"].items() + } + sector_size = cfg.get("sector_size", vixdisklib.VIXDISKLIB_SECTOR_SIZE) + connect_kwargs = { + "server_name": cfg["server_name"], + "thumbprint": cfg["thumbprint"], + "username": cfg["username"], + "password": cfg["password"], + "vmx_spec": cfg["vmx_spec"], + "read_only": False, + "transport_modes": "hotadd", + "port": cfg.get("port", 443), + "allow_untrusted": cfg.get("allow_untrusted", False), + } + write_buf = vixdisklib.get_buffer(sector_size) + read_buf = vixdisklib.get_buffer(sector_size) + extra_after = 0 + mode = "" + with handle.connect(**connect_kwargs) as conn: + with handle.open(conn, cfg["disk_path"], flags=0) as disk: + mode = handle.get_transport_mode(disk) + if mode != "hotadd": + print(json.dumps({"ok": False, "error": f"mode {mode!r}"})) + return 1 + for start, expected in patterns.items(): + write_buf[:sector_size] = expected + handle.write(disk, start, 1, write_buf) + read_buf[:sector_size] = b"\xa5" * sector_size + handle.read(disk, start, 1, read_buf) + if read_buf.raw[:sector_size] != expected: + print( + json.dumps( + { + "ok": False, + "error": f"mismatch at sector {start}", + } + ) + ) + return 1 + extra_after = _extra_disk_count(conn.si) + print( + json.dumps({"ok": True, "mode": mode, "extra_disks_after_close": extra_after}) + ) + return 0 + + +def _extra_disk_count(si) -> int: + """Return how many non-boot disks remain on the proxy VM.""" + proxy = find_proxy_vm(si) + disks = [ + device + for device in proxy.config.hardware.device + if isinstance(device, vim.vm.device.VirtualDisk) + ] + return max(0, len(disks) - 1) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/integration/test_hotadd.py b/tests/integration/test_hotadd.py new file mode 100644 index 0000000..7479ce1 --- /dev/null +++ b/tests/integration/test_hotadd.py @@ -0,0 +1,184 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Exercise HotAdd from the Linux proxy guest over SSH.""" + +from __future__ import annotations + +import json +import os +from collections.abc import Iterator +from typing import Any + +import pytest +from pyVmomi import vim + +from openvixdisklib import openvixdisklib as vixdisklib +from tests.integration.base import ( + SECTOR_AT_1GB, + SECTOR_SIZE, + LabEnv, + _connect_vim, + create_lab_vm, + destroy_lab_vm, + load_hotadd_proxy_config, + pattern_bytes, +) +from tests.integration.hotadd_proxy import ( + REMOTE_DIR, + REMOTE_PYTHON, + prepare_hotadd_proxy, + ssh_proxy, +) + +_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) + + +@pytest.fixture(scope="session") +def hotadd_proxy() -> dict[str, str]: + """SSH settings for the Linux HotAdd proxy, or skip.""" + remote_py = os.path.join(_REPO_ROOT, "tests", "integration", "hotadd_remote.py") + try: + return prepare_hotadd_proxy({"hotadd_remote.py": remote_py}) + except RuntimeError as exc: + pytest.skip(str(exc)) + + +@pytest.fixture +def nvme_lab() -> Iterator[LabEnv]: + """Powered-off lab VM whose disk is on an NVMe controller.""" + env = create_lab_vm(disk_controller="nvme") + try: + yield env + finally: + destroy_lab_vm(env) + + +def _proxy_extra_disk_count(lab: LabEnv) -> int: + """Count non-boot virtual disks on the HotAdd proxy VM.""" + proxy_cfg = load_hotadd_proxy_config() + if proxy_cfg is None: + return 0 + si = _connect_vim( + lab.host, + lab.username, + lab.password, + lab.port, + lab.thumbprint, + lab.allow_untrusted, + ) + try: + content = si.RetrieveContent() + container = content.viewManager.CreateContainerView( + content.rootFolder, [vim.VirtualMachine], True + ) + try: + for vm in container.view: + ips: list[str] = [] + if vm.guest and vm.guest.net: + for nic in vm.guest.net: + ips.extend(nic.ipAddress or []) + if proxy_cfg["host"] in ips: + disks = [ + device + for device in vm.config.hardware.device + if isinstance(device, vim.vm.device.VirtualDisk) + ] + return max(0, len(disks) - 1) + finally: + container.Destroy() + finally: + from pyVim.connect import Disconnect + + Disconnect(si) + return 0 + + +def _run_hotadd_remote(proxy: dict[str, str], lab: LabEnv) -> dict[str, Any]: + patterns = { + "0": pattern_bytes(SECTOR_SIZE, b"OVDL-HA0").hex(), + str(SECTOR_AT_1GB): pattern_bytes(SECTOR_SIZE, b"OVDL-HA1").hex(), + } + payload = json.dumps( + { + "server_name": lab.host, + "thumbprint": lab.thumbprint, + "username": lab.username, + "password": lab.password, + "port": lab.port, + "allow_untrusted": lab.allow_untrusted, + "vmx_spec": lab.vmx_spec, + "disk_path": lab.disk_path, + "sector_size": SECTOR_SIZE, + "patterns": patterns, + } + ).encode() + result = ssh_proxy( + proxy, + f"cd {REMOTE_DIR} && PYTHONPATH={REMOTE_DIR} {REMOTE_PYTHON} hotadd_remote.py", + stdin=payload, + ) + if result.returncode != 0: + raise AssertionError( + f"hotadd_remote failed rc={result.returncode} " + f"stdout={result.stdout.decode(errors='replace')!r} " + f"stderr={result.stderr.decode(errors='replace')!r}" + ) + report = json.loads(result.stdout.decode()) + assert report.get("ok") is True, report + return report + + +def _assert_nbdssl_matches(lab: LabEnv, patterns: dict[int, bytes]) -> None: + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + read_buf = vixdisklib.get_buffer(SECTOR_SIZE) + kwargs = lab.vixdisklib_connect_kwargs( + { + "allow_untrusted": lab.allow_untrusted, + "transport_modes": "nbdssl", + "read_only": True, + } + ) + with ( + handle.connect(**kwargs) as conn, + handle.open( + conn, lab.disk_path, flags=vixdisklib.VIXDISKLIB_FLAG_OPEN_READ_ONLY + ) as disk, + ): + for start, expected in patterns.items(): + read_buf[:SECTOR_SIZE] = b"\xa5" * SECTOR_SIZE + handle.read(disk, start, 1, read_buf) + assert read_buf.raw[:SECTOR_SIZE] == expected + + +class TestHotAdd: + def test_pvscsi_write_read(self, lab: LabEnv, hotadd_proxy: dict[str, str]) -> None: + """HotAdd a PVSCSI lab disk on the proxy and verify via nbdssl.""" + extra_before = _proxy_extra_disk_count(lab) + report = _run_hotadd_remote(hotadd_proxy, lab) + assert report["mode"] == "hotadd" + assert report["extra_disks_after_close"] == extra_before + assert _proxy_extra_disk_count(lab) == extra_before + _assert_nbdssl_matches( + lab, + { + 0: pattern_bytes(SECTOR_SIZE, b"OVDL-HA0"), + SECTOR_AT_1GB: pattern_bytes(SECTOR_SIZE, b"OVDL-HA1"), + }, + ) + + def test_nvme_source_write_read( + self, nvme_lab: LabEnv, hotadd_proxy: dict[str, str] + ) -> None: + """HotAdd an NVMe-backed VMDK onto the proxy's SCSI controller.""" + extra_before = _proxy_extra_disk_count(nvme_lab) + report = _run_hotadd_remote(hotadd_proxy, nvme_lab) + assert report["mode"] == "hotadd" + assert report["extra_disks_after_close"] == extra_before + _assert_nbdssl_matches( + nvme_lab, + { + 0: pattern_bytes(SECTOR_SIZE, b"OVDL-HA0"), + SECTOR_AT_1GB: pattern_bytes(SECTOR_SIZE, b"OVDL-HA1"), + }, + ) diff --git a/tests/perf/hotadd_remote.py b/tests/perf/hotadd_remote.py new file mode 100644 index 0000000..fa49a69 --- /dev/null +++ b/tests/perf/hotadd_remote.py @@ -0,0 +1,80 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Time HotAdd write/read inside the proxy guest. Invoked over SSH by perf.""" + +from __future__ import annotations + +import json +import sys +import time + +from openvixdisklib import openvixdisklib as vixdisklib + + +def _pattern_bytes(length: int, seed: bytes) -> bytes: + return (seed * ((length // len(seed)) + 1))[:length] + + +def main() -> int: + """Read connect kwargs and size from stdin, HotAdd, time write/read.""" + cfg = json.load(sys.stdin) + nbytes = int(cfg["nbytes"]) + if nbytes % vixdisklib.VIXDISKLIB_SECTOR_SIZE: + print(json.dumps({"ok": False, "error": f"unaligned size {nbytes}"})) + return 1 + n_sectors = nbytes // vixdisklib.VIXDISKLIB_SECTOR_SIZE + payload = _pattern_bytes(nbytes, f"PERF-{cfg['label']}-".encode()) + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + modes = handle.get_transport_modes() + if "hotadd" not in modes: + print(json.dumps({"ok": False, "error": f"hotadd not listed: {modes}"})) + return 1 + connect_kwargs = { + "server_name": cfg["server_name"], + "thumbprint": cfg["thumbprint"], + "username": cfg["username"], + "password": cfg["password"], + "vmx_spec": cfg["vmx_spec"], + "read_only": False, + "transport_modes": "hotadd", + "port": cfg.get("port", 443), + "allow_untrusted": cfg.get("allow_untrusted", False), + } + write_buf = vixdisklib.get_buffer(nbytes) + read_buf = vixdisklib.get_buffer(nbytes) + write_buf[:nbytes] = payload + mode = "" + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, cfg["disk_path"], flags=0) as disk, + ): + mode = handle.get_transport_mode(disk) + if mode != "hotadd": + print(json.dumps({"ok": False, "error": f"mode {mode!r}"})) + return 1 + started = time.perf_counter() + handle.write(disk, 0, n_sectors, write_buf) + write_s = time.perf_counter() - started + read_buf[:nbytes] = b"\xa5" * nbytes + started = time.perf_counter() + handle.read(disk, 0, n_sectors, read_buf) + read_s = time.perf_counter() - started + if read_buf.raw[:nbytes] != payload: + print(json.dumps({"ok": False, "error": "mismatch after read"})) + return 1 + print( + json.dumps( + { + "ok": True, + "mode": mode, + "write_s": write_s, + "read_s": read_s, + } + ) + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/perf/test_compare.py b/tests/perf/test_compare.py index 2b608d3..7ceab17 100644 --- a/tests/perf/test_compare.py +++ b/tests/perf/test_compare.py @@ -14,6 +14,8 @@ import time from typing import Any +import pytest + from openvixdisklib import nfc_open from openvixdisklib import openvixdisklib as open_vix from tests.integration import vixdisklib @@ -23,6 +25,12 @@ ensure_vddk_library_path, pattern_bytes, ) +from tests.integration.hotadd_proxy import ( + REMOTE_DIR, + REMOTE_PYTHON, + prepare_hotadd_proxy, + ssh_proxy, +) _REPO = os.path.abspath(os.path.join(os.path.dirname(__file__), "../..")) _SIZES = ( @@ -262,6 +270,102 @@ def _mib_per_s(nbytes: int, seconds: float) -> float: return (nbytes / (1024 * 1024)) / seconds +_HOTADD_REMOTE = os.path.join(os.path.dirname(__file__), "hotadd_remote.py") +_HOTADD_SSH_TIMEOUT_S = 600 +_PerfRow = tuple[str, str, str, str, str, str, float, float, float, float] + + +def _time_hotadd_remote( + lab: LabEnv, proxy: dict[str, str], label: str, nbytes: int +) -> tuple[float, float]: + """Time OpenVixDiskLib HotAdd write/read on the Linux proxy guest.""" + payload = json.dumps( + { + "server_name": lab.host, + "thumbprint": lab.thumbprint, + "username": lab.username, + "password": lab.password, + "port": lab.port, + "allow_untrusted": lab.allow_untrusted, + "vmx_spec": lab.vmx_spec, + "disk_path": lab.disk_path, + "label": label, + "nbytes": nbytes, + } + ).encode() + result = ssh_proxy( + proxy, + f"cd {REMOTE_DIR} && PYTHONPATH={REMOTE_DIR} {REMOTE_PYTHON} " + "hotadd_perf_remote.py", + stdin=payload, + timeout=_HOTADD_SSH_TIMEOUT_S, + ) + if result.returncode != 0: + raise RuntimeError( + "hotadd perf remote failed " + f"rc={result.returncode} " + f"stdout={result.stdout.decode(errors='replace')!r} " + f"stderr={result.stderr.decode(errors='replace')!r}" + ) + report = json.loads(result.stdout.decode()) + if not report.get("ok"): + raise RuntimeError(f"hotadd perf remote error: {report}") + return float(report["write_s"]), float(report["read_s"]) + + +def _hotadd_rows(lab: LabEnv) -> list[_PerfRow]: + """Time plain HotAdd I/O for each transfer size on the proxy guest.""" + proxy = prepare_hotadd_proxy({"hotadd_perf_remote.py": _HOTADD_REMOTE}) + print(f"hotadd timings run on {proxy['user']}@{proxy['host']}") + rows: list[_PerfRow] = [] + for label, nbytes in _SIZES: + write_s, read_s = _time_hotadd_remote(lab, proxy, label, nbytes) + rows.append( + ( + label, + "-", + "-", + "hotadd", + "plain", + "openvixdisklib", + write_s, + read_s, + _mib_per_s(nbytes, write_s), + _mib_per_s(nbytes, read_s), + ) + ) + return rows + + +def _print_perf_table(rows: list[_PerfRow]) -> None: + """Print throughput rows to stdout (``tox -e perf`` uses ``-s``).""" + print() + print( + f"{'size':<14} {'aio_size':<8} {'aio_count':>9} " + f"{'transport':<10} {'flags':<12} {'library':<16} " + f"{'write_s':>10} {'read_s':>10} " + f"{'write_MiB/s':>12} {'read_MiB/s':>12}" + ) + for ( + label, + aio_label, + aio_count, + transport_mode, + mode_name, + name, + write_s, + read_s, + write_r, + read_r, + ) in rows: + print( + f"{label:<14} {aio_label:<8} {aio_count:>9} " + f"{transport_mode:<10} {mode_name:<12} {name:<16} " + f"{write_s:10.3f} {read_s:10.3f} " + f"{write_r:12.1f} {read_r:12.1f}" + ) + + class TestCompare: def test_write_read_throughput(self, lab: LabEnv, vddk: None) -> None: """Time matching write/read sizes on VDDK and openvixdisklib. @@ -287,7 +391,7 @@ def test_write_read_throughput(self, lab: LabEnv, vddk: None) -> None: True, ), ) - rows: list[tuple[str, str, int, str, str, str, float, float, float, float]] = [] + rows: list[_PerfRow] = [] for label, nbytes in _SIZES: for aio_buffer_count, aio_buffer_size in _AIO_SESSIONS: aio_label = _aio_size_label(aio_buffer_size) @@ -311,7 +415,7 @@ def test_write_read_throughput(self, lab: LabEnv, vddk: None) -> None: ( label, aio_label, - aio_buffer_count, + str(aio_buffer_count), transport_mode, mode_name, name, @@ -321,31 +425,21 @@ def test_write_read_throughput(self, lab: LabEnv, vddk: None) -> None: _mib_per_s(nbytes, read_s), ) ) - print() - print( - f"{'size':<14} {'aio_size':<8} {'aio_count':>9} " - f"{'transport':<10} {'flags':<12} {'library':<16} " - f"{'write_s':>10} {'read_s':>10} " - f"{'write_MiB/s':>12} {'read_MiB/s':>12}" - ) - for ( - label, - aio_label, - aio_buffer_count, - transport_mode, - mode_name, - name, - write_s, - read_s, - write_r, - read_r, - ) in rows: - print( - f"{label:<14} {aio_label:<8} {aio_buffer_count:>9} " - f"{transport_mode:<10} {mode_name:<12} {name:<16} " - f"{write_s:10.3f} {read_s:10.3f} " - f"{write_r:12.1f} {read_r:12.1f}" - ) + _print_perf_table(rows) + + def test_hotadd_write_read_throughput(self, lab: LabEnv) -> None: + """Time OpenVixDiskLib HotAdd write/read on the Linux proxy guest. + + Uses the same transfer sizes as ``test_write_read_throughput``. + FastLZ and NFC AIO do not apply. Native VDDK HotAdd is not + compared (it would also have to run in the guest). Skips when + ``hotadd_proxy`` is missing or SSH fails. + """ + try: + rows = _hotadd_rows(lab) + except RuntimeError as exc: + pytest.skip(str(exc)) + _print_perf_table(rows) if __name__ == "__main__": diff --git a/tests/unit/test_hotadd.py b/tests/unit/test_hotadd.py new file mode 100644 index 0000000..1636335 --- /dev/null +++ b/tests/unit/test_hotadd.py @@ -0,0 +1,276 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Unit tests for HotAdd attach specs and local block I/O.""" + +from __future__ import annotations + +import os +from unittest import mock + +import pytest +from pyVmomi import vim + +from openvixdisklib.hotadd import ( + AttachPlan, + HotAddDisk, + _byteswap_uuid, + _offline_scsi_unit, + build_attach_spec, + build_detach_spec, + find_scsi_block_device, + find_source_disk, + pick_scsi_slot, + wait_scsi_device_gone, +) +from openvixdisklib.openvixdisklib import ( + _available_transports, + _select_transport, +) + + +def _scsi(key: int = 1000, bus: int = 0) -> vim.vm.device.ParaVirtualSCSIController: + controller = vim.vm.device.ParaVirtualSCSIController() + controller.key = key + controller.busNumber = bus + return controller + + +def _lsi(key: int = 1000, bus: int = 0) -> vim.vm.device.VirtualLsiLogicController: + controller = vim.vm.device.VirtualLsiLogicController() + controller.key = key + controller.busNumber = bus + return controller + + +def _disk( + key: int, + controller_key: int, + unit: int, + file_name: str, +) -> vim.vm.device.VirtualDisk: + disk = vim.vm.device.VirtualDisk() + disk.key = key + disk.controllerKey = controller_key + disk.unitNumber = unit + backing = vim.vm.device.VirtualDisk.FlatVer2BackingInfo() + backing.fileName = file_name + disk.backing = backing + disk.capacityInKB = 1024 + return disk + + +class TestFindSourceDisk: + def test_nvme_source_is_accepted(self) -> None: + """NVMe-backed VMDKs are valid HotAdd sources.""" + nvme = vim.vm.device.VirtualNVMEController() + nvme.key = 31000 + nvme.busNumber = 0 + path = "[datastore0] nvme/nvme.vmdk" + disk = _disk(32000, 31000, 0, path) + assert find_source_disk([nvme, disk], path) is disk + + def test_sata_source_is_accepted(self) -> None: + """SATA-backed VMDKs are valid HotAdd sources.""" + sata = vim.vm.device.VirtualAHCIController() + sata.key = 15000 + sata.busNumber = 0 + path = "[datastore0] sata/sata.vmdk" + disk = _disk(16000, 15000, 0, path) + assert find_source_disk([sata, disk], path) is disk + + def test_ide_source_is_rejected(self) -> None: + """IDE disks cannot be HotAdded.""" + ide = vim.vm.device.VirtualIDEController() + ide.key = 200 + ide.busNumber = 0 + path = "[datastore0] ide/ide.vmdk" + disk = _disk(201, 200, 0, path) + with pytest.raises(NotImplementedError, match="IDE"): + find_source_disk([ide, disk], path) + + def test_missing_path_raises(self) -> None: + """Unknown backing paths raise FileNotFoundError.""" + scsi = _scsi() + disk = _disk(2000, 1000, 0, "[datastore0] vm/vm.vmdk") + with pytest.raises(FileNotFoundError): + find_source_disk([scsi, disk], "[datastore0] other/other.vmdk") + + +class TestAttachSpec: + def test_uses_free_unit_one_on_existing_scsi(self) -> None: + """A proxy with a boot disk at unit 0 HotAdds at unit 1.""" + devices = [ + _lsi(), + _disk(2000, 1000, 0, "[datastore0] proxy/boot.vmdk"), + ] + source = "[datastore0] src/src.vmdk" + plan = build_attach_spec(devices, source, read_only=True, capacity_kb=2048) + assert isinstance(plan, AttachPlan) + assert plan.bus_number == 0 + assert plan.unit_number == 1 + assert len(plan.spec.deviceChange) == 1 + change = plan.spec.deviceChange[0] + assert change.operation == vim.vm.device.VirtualDeviceSpec.Operation.add + assert change.fileOperation is None + disk = change.device + assert isinstance(disk, vim.vm.device.VirtualDisk) + assert disk.controllerKey == 1000 + assert disk.unitNumber == 1 + assert disk.backing.fileName == source + assert disk.backing.diskMode == "independent_nonpersistent" + assert disk.capacityInKB == 2048 + + def test_writable_open_uses_persistent_mode(self) -> None: + """Restore / write HotAdd attaches the VMDK persistently.""" + devices = [_scsi(), _disk(2000, 1000, 0, "[datastore0] proxy/boot.vmdk")] + plan = build_attach_spec(devices, "[datastore0] src/src.vmdk", read_only=False) + disk = plan.spec.deviceChange[0].device + assert disk.backing.diskMode == "persistent" + + def test_nvme_source_still_targets_proxy_scsi(self) -> None: + """NVMe sources are attached onto the proxy SCSI controller.""" + proxy = [_lsi(), _disk(2000, 1000, 0, "[datastore0] proxy/boot.vmdk")] + plan = build_attach_spec(proxy, "[datastore0] nvme/nvme.vmdk", read_only=True) + disk = plan.spec.deviceChange[0].device + assert disk.controllerKey == 1000 + assert not isinstance(disk, vim.vm.device.VirtualNVMEController) + + def test_full_bus_adds_pvscsi_controller(self) -> None: + """A new PVSCSI controller is added when every SCSI unit is taken.""" + devices: list[vim.vm.device.VirtualDevice] = [_scsi()] + key = 2000 + for unit in range(16): + if unit == 7: + continue + devices.append(_disk(key, 1000, unit, f"[datastore0] proxy/d{unit}.vmdk")) + key += 1 + plan = build_attach_spec(devices, "[datastore0] src/src.vmdk", read_only=True) + assert plan.bus_number == 1 + assert plan.unit_number == 0 + assert len(plan.spec.deviceChange) == 2 + ctrl_change, disk_change = plan.spec.deviceChange + assert ctrl_change.fileOperation is None + assert isinstance(ctrl_change.device, vim.vm.device.ParaVirtualSCSIController) + assert ctrl_change.device.busNumber == 1 + assert disk_change.fileOperation is None + assert disk_change.device.controllerKey == ctrl_change.device.key + assert disk_change.device.unitNumber == 0 + + def test_pick_scsi_slot_skips_reserved_unit_seven(self) -> None: + """SCSI unit 7 stays unused.""" + devices = [ + _scsi(), + _disk(2000, 1000, 0, "[datastore0] proxy/boot.vmdk"), + ] + controller, bus, unit = pick_scsi_slot(devices) + assert controller is not None + assert bus == 0 + assert unit == 1 + assert unit != 7 + + def test_detach_does_not_set_file_operation(self) -> None: + """Detach must never delete the source VMDK.""" + disk = _disk(2001, 1000, 1, "[datastore0] src/src.vmdk") + spec = build_detach_spec(disk) + change = spec.deviceChange[0] + assert change.operation == vim.vm.device.VirtualDeviceSpec.Operation.remove + assert change.fileOperation is None + assert change.device is disk + + +class TestScsiSysfs: + def test_finds_device_when_linux_host_differs_from_vmware_bus( + self, tmp_path, monkeypatch: pytest.MonkeyPatch + ) -> None: + """VMware bus 0 can appear as Linux SCSI host 2.""" + block = tmp_path / "2:0:1:0" / "block" / "sdb" + block.mkdir(parents=True) + monkeypatch.setattr("openvixdisklib.hotadd.SCSI_DEVICE_DIR", str(tmp_path)) + assert find_scsi_block_device(0, 1) == "/dev/sdb" + + def test_offline_scsi_unit_writes_delete( + self, tmp_path, monkeypatch: pytest.MonkeyPatch + ) -> None: + """Linux keeps the LUN until sysfs delete is written.""" + device = tmp_path / "2:0:1:0" + (device / "block" / "sdb").mkdir(parents=True) + monkeypatch.setattr("openvixdisklib.hotadd.SCSI_DEVICE_DIR", str(tmp_path)) + _offline_scsi_unit(1) + assert (device / "delete").read_text(encoding="ascii") == "1\n" + + def test_wait_scsi_device_gone_returns_when_sysfs_empty( + self, tmp_path, monkeypatch: pytest.MonkeyPatch + ) -> None: + """Close succeeds after the SCSI sysfs node disappears.""" + monkeypatch.setattr("openvixdisklib.hotadd.SCSI_DEVICE_DIR", str(tmp_path)) + monkeypatch.setattr("openvixdisklib.hotadd.DEVICE_POLL_S", 0.01) + wait_scsi_device_gone(0, 1, timeout_s=1) + + def test_wait_scsi_device_gone_times_out( + self, tmp_path, monkeypatch: pytest.MonkeyPatch + ) -> None: + """Close fails if the LUN stays in sysfs after detach.""" + (tmp_path / "2:0:1:0" / "block" / "sdb").mkdir(parents=True) + monkeypatch.setattr("openvixdisklib.hotadd.SCSI_DEVICE_DIR", str(tmp_path)) + monkeypatch.setattr("openvixdisklib.hotadd.DEVICE_POLL_S", 0.01) + with pytest.raises(TimeoutError, match="still present"): + wait_scsi_device_gone(0, 1, timeout_s=0.05) + + +class TestHotAddDiskIO: + def test_read_write_sectors(self, tmp_path) -> None: + """pread/pwrite round-trip 512-byte sectors on a local file.""" + path = tmp_path / "disk.img" + path.write_bytes(b"\x00" * 1024) + fd = os.open(path, os.O_RDWR) + detached: list[bool] = [] + disk = HotAddDisk(fd, str(path), 0, 1, lambda: detached.append(True)) + pattern = b"OVDL-HA" * (512 // 7) + b"OVDL-HA"[: 512 % 7] + disk.write(1, 1, pattern) + buf = bytearray(512) + result = disk.readinto(1, 1, buf, skip_decompression=True) + assert bytes(buf) == pattern + assert result.uncompressed_length == 512 + assert result.fragments == () + disk.close() + assert detached == [True] + disk.close() + assert detached == [True] + + +class TestSelectTransport: + @mock.patch( + "openvixdisklib.openvixdisklib.hotadd.is_vmware_guest", return_value=False + ) + def test_colon_list_skips_hotadd_on_bare_metal( + self, mock_guest: mock.MagicMock + ) -> None: + """Bare metal skips hotadd and uses the next usable mode.""" + del mock_guest + assert _select_transport("file:san:hotadd:nbdssl:nbd") == "nbdssl" + assert _available_transports() == ["nbdssl", "nbd"] + with pytest.raises(NotImplementedError, match="hotadd"): + _select_transport("hotadd") + + @mock.patch( + "openvixdisklib.openvixdisklib.hotadd.is_vmware_guest", return_value=True + ) + def test_colon_list_selects_hotadd_in_guest( + self, mock_guest: mock.MagicMock + ) -> None: + """A VMware guest uses hotadd when it is first in the colon list.""" + del mock_guest + assert _select_transport("file:san:hotadd:nbdssl") == "hotadd" + assert _available_transports() == ["nbdssl", "nbd", "hotadd"] + + def test_default_is_nbdssl(self) -> None: + """None still defaults to nbdssl, even in a guest.""" + assert _select_transport(None) == "nbdssl" + + +def test_byteswap_uuid_matches_vmware_bios_uuid() -> None: + """Linux DMI UUID is byte-swapped relative to vim.vm.ConfigInfo.uuid.""" + dmi = "8ac43342-7478-0792-f6c6-131b895335ba" + bios = "4233c48a-7874-9207-f6c6-131b895335ba" + assert _byteswap_uuid(dmi).lower() == bios From 87b4f281d69de44810def82a07380ea6df4e7bdb Mon Sep 17 00:00:00 2001 From: Lucian Petrut Date: Mon, 21 Sep 2026 13:50:37 +0000 Subject: [PATCH 2/2] san transport support The SAN transport can be used when VMFS is stored on an iSCSI lun. Instead of talking to ESXi over NBD/NFC, the data is retrieved from the iSCSI lun directly, which will be temporarily mounted locally. This also means that we need to parse VMDKs residing on VMFS. Normally we'd use vmfs(6)-fuse to mount the filesystem and then parse the VMDK using libqemu. Unfortunately all the vmfs-fuse implementations seem to be abandoned, so we had to implement VMFS and VMDK parsing in OpenVixDiskLib. For this reason, we'll consider this implementation experimental only. FibreChannel is unsupported, we didn't have hardware to test it. --- README.md | 21 +- docs/probing_samples/vddk_san_trace.py | 156 ++++ docs/reverse_engineering_procedure.md | 43 +- docs/san.md | 76 ++ openvixdisklib/openvixdisklib.py | 66 +- openvixdisklib/san.py | 867 ++++++++++++++++++++++ tests/integration/base.py | 55 +- tests/integration/iscsi_lab.py | 898 +++++++++++++++++++++++ tests/integration/test_openvixdisklib.py | 4 +- tests/integration/test_san.py | 119 +++ tests/unit/test_hotadd.py | 12 +- tests/unit/test_san.py | 168 +++++ 12 files changed, 2436 insertions(+), 49 deletions(-) create mode 100644 docs/probing_samples/vddk_san_trace.py create mode 100644 docs/san.md create mode 100644 openvixdisklib/san.py create mode 100644 tests/integration/iscsi_lab.py create mode 100644 tests/integration/test_san.py create mode 100644 tests/unit/test_san.py diff --git a/README.md b/README.md index 8b9364f..e496045 100644 --- a/README.md +++ b/README.md @@ -14,12 +14,14 @@ Python naming). VIM login and inventory use [pyVmomi](https://github.com/vmware/pyvmomi). The NFC ticket, ESXi authd handshake, and disk I/O were reverse-engineered from VDDK 8 NBD traffic; see `docs/`. Linux HotAdd uses the public -vSphere `ReconfigureVM` API (see `docs/hotadd.md`). +vSphere `ReconfigureVM` API (see `docs/hotadd.md`). Linux SAN reads a +shared VMFS LUN locally (see `docs/san.md`). ## Status Implemented against vCenter 8 / ESXi 8. Default transport is `nbdssl` -(`nbd` is still available). Linux guests can also use `hotadd`: +(`nbd` is still available). Linux hosts that see the VMFS LUN can use +`san`. Linux guests can also use `hotadd`: - `VixDiskLib_ConnectEx` (UID credentials) - `VixDiskLib_Open` (datastore path, read-only or read-write) @@ -27,11 +29,14 @@ Implemented against vCenter 8 / ESXi 8. Default transport is `nbdssl` - `VixDiskLib_Write` - HotAdd on a Linux VMware guest (SCSI, NVMe, or SATA source disks, attached onto a proxy SCSI controller) +- SAN on a Linux host that sees the same VMFS LUN as ESXi (NAA match, + local `pread` / `pwrite` of the flat extent) Not implemented: compression open flags other than FastLZ, CBT / allocated-block queries, disk geometry (`DDB_GET`), encrypted disks, -direct ESXi `ha-nfc` without vCenter `vpxa-nfc`, SAN / file transports, -Windows HotAdd, and HotAdd onto a proxy NVMe controller. +direct ESXi `ha-nfc` without vCenter `vpxa-nfc`, file transport, +snapshot / SESPARSE SAN chains, Windows HotAdd, and HotAdd onto a +proxy NVMe controller. Requires Python 3.10 or later. @@ -78,6 +83,7 @@ VDDK-shaped handle. | `openvixdisklib/nfc_auth.py` | VIM login, NFC ticket, authd on 902 | | `openvixdisklib/nfc_open.py` | Classic NFC handshake, AIO open, sector read/write | | `openvixdisklib/hotadd.py` | Linux-guest SCSI HotAdd attach, local block I/O | +| `openvixdisklib/san.py` | Linux SAN: NAA match, VMFS map, local block I/O | | `openvixdisklib/fastlz.py` | FastLZ NFC adapter (pip `pyfastlz`) | | `tests/integration/` | Live pytest suite against a lab vCenter | | `tests/perf/` | Throughput comparison of OpenVixDiskLib vs VDDK | @@ -104,13 +110,17 @@ datastore: datastore0 hotadd_proxy: host: hotadd-proxy.example.com user: root +iscsi_san: + portal: 192.0.2.10 ``` A session-scoped pytest fixture creates an empty VM with a 10 GiB thin disk on that datastore and tears it down when the session ends. Tests write known patterns and read them back. HotAdd tests SSH into `hotadd_proxy` (a Linux guest on the same datastore) and skip if SSH -fails. +fails. SAN tests bring up a loop-backed iSCSI LUN, create a VMFS +datastore, and skip if LIO, `iscsiadm`, or software iSCSI cannot be +used. The session lab VM is not placed on that LUN. ```bash tox -e integration @@ -159,5 +169,6 @@ Lint and typecheck: `tox -e pep8`, `tox -e mypy`. | `docs/nfc_read.md` | AIO IO / `VixDiskLib_Read` | | `docs/nfc_write.md` | AIO IO / `VixDiskLib_Write` | | `docs/hotadd.md` | Linux-guest SCSI HotAdd | +| `docs/san.md` | Linux SAN / VMFS LUN I/O | | `docs/ssl_hook.md` | TLS intercept used for capture | | `docs/reverse_engineering_procedure.md` | How the protocol was recovered | diff --git a/docs/probing_samples/vddk_san_trace.py b/docs/probing_samples/vddk_san_trace.py new file mode 100644 index 0000000..9a05b31 --- /dev/null +++ b/docs/probing_samples/vddk_san_trace.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +"""Trace native VDDK SAN open/read (device matching + VMFS, not NFC). + +SAN is not a wire protocol. Capture with:: + + strace -f -e openat,pread64,pwrite64,ioctl -o /tmp/vddk-san.strace \\ + python3 docs/probing_samples/vddk_san_trace.py + +``InitEx`` must pass a libDir that contains ``lib64/libdiskLibPlugin.so`` +(the advanced transport plugin). VDDK then matches the VMFS LUN by NAA +and reads the flat extent through its VMFS driver. + +Replace the ```` fields with lab values. Do not commit +credentials or live IPs. +""" + +import ctypes +import os + +REPO = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) +VDDK_DIR = os.path.join(REPO, ".vddk") +lib = ctypes.CDLL(os.path.join(VDDK_DIR, "libvixDiskLib.so")) + +VIXDISKLIB_CRED_UID = 1 +VIXDISKLIB_FLAG_OPEN_READ_ONLY = 4 +SECTOR = 512 + + +class VixDiskLibUidPasswdCreds(ctypes.Structure): + _fields_ = [ + ("userName", ctypes.c_char_p), + ("password", ctypes.c_char_p), + ] + + +class VixDiskLibSessionIdCreds(ctypes.Structure): + _fields_ = [ + ("cookie", ctypes.c_char_p), + ("userName", ctypes.c_char_p), + ("key", ctypes.c_char_p), + ] + + +class VixDiskLibCreds(ctypes.Union): + _fields_ = [ + ("uid", VixDiskLibUidPasswdCreds), + ("sessionId", VixDiskLibSessionIdCreds), + ] + + +class VixDiskLibConnectParams(ctypes.Structure): + _fields_ = [ + ("vmxSpec", ctypes.c_char_p), + ("serverName", ctypes.c_char_p), + ("thumbPrint", ctypes.c_char_p), + ("privateUse", ctypes.c_longlong), + ("credType", ctypes.c_uint32), + ("creds", VixDiskLibCreds), + ("port", ctypes.c_uint32), + ("nfcHostPort", ctypes.c_uint32), + ("vimApiVer", ctypes.c_char_p), + ] + + +def check(err, what): + if err != 0: + lib.VixDiskLib_GetErrorText.restype = ctypes.c_void_p + lib.VixDiskLib_GetErrorText.argtypes = [ctypes.c_uint64, ctypes.c_char_p] + msg = lib.VixDiskLib_GetErrorText(err, None) + text = ctypes.cast(msg, ctypes.c_char_p).value + lib.VixDiskLib_FreeErrorText.argtypes = [ctypes.c_char_p] + lib.VixDiskLib_FreeErrorText(ctypes.cast(msg, ctypes.c_char_p)) + raise SystemExit(f"{what} failed: {err} {text}") + + +def main(): + os.makedirs("/tmp/vddk-san-trace", exist_ok=True) + config_path = "/tmp/vddk-san-trace/vddk.config" + with open(config_path, "w") as f: + f.write("tmpDirectory=/tmp/vddk-san-trace\n") + f.write("log.fileName=/tmp/vddk-san-trace/vddk.log\n") + f.write("log.fileLevel=verbose\n") + f.write("vixDiskLib.transport.LogLevel=4\n") + + plugin = os.path.join(VDDK_DIR, "lib64", "libdiskLibPlugin.so") + if not os.path.isfile(plugin): + raise SystemExit( + f"missing {plugin}; SAN needs libdiskLibPlugin under libDir/lib64" + ) + + lib.VixDiskLib_InitEx.argtypes = [ + ctypes.c_uint32, ctypes.c_uint32, ctypes.c_void_p, ctypes.c_void_p, + ctypes.c_void_p, ctypes.c_char_p, ctypes.c_char_p] + lib.VixDiskLib_InitEx.restype = ctypes.c_uint64 + check(lib.VixDiskLib_InitEx( + 8, 0, None, None, None, VDDK_DIR.encode(), config_path.encode()), + "InitEx") + + lib.VixDiskLib_ListTransportModes.restype = ctypes.c_char_p + print("ListTransportModes", lib.VixDiskLib_ListTransportModes(), flush=True) + + params = VixDiskLibConnectParams() + params.vmxSpec = b"" + params.serverName = b"" + params.thumbPrint = b"" + params.credType = VIXDISKLIB_CRED_UID + params.creds.uid.userName = b"" + params.creds.uid.password = b"" + params.port = 443 + + lib.VixDiskLib_ConnectEx.argtypes = [ + ctypes.POINTER(VixDiskLibConnectParams), ctypes.c_char, + ctypes.c_char_p, ctypes.c_char_p, ctypes.POINTER(ctypes.c_void_p)] + lib.VixDiskLib_ConnectEx.restype = ctypes.c_uint64 + conn = ctypes.c_void_p() + check(lib.VixDiskLib_ConnectEx( + params, True, None, b"san", ctypes.byref(conn)), + "ConnectEx") + print("ConnectEx ok", flush=True) + + lib.VixDiskLib_Open.argtypes = [ + ctypes.c_void_p, ctypes.c_char_p, ctypes.c_uint32, + ctypes.POINTER(ctypes.c_void_p)] + lib.VixDiskLib_Open.restype = ctypes.c_uint64 + disk = ctypes.c_void_p() + path = b"[ovdl-iscsi-] /.vmdk" + check(lib.VixDiskLib_Open(conn, path, VIXDISKLIB_FLAG_OPEN_READ_ONLY, ctypes.byref(disk)), + "Open") + print("Open ok", flush=True) + + lib.VixDiskLib_GetTransportMode.argtypes = [ctypes.c_void_p] + lib.VixDiskLib_GetTransportMode.restype = ctypes.c_char_p + print("transport", lib.VixDiskLib_GetTransportMode(disk), flush=True) + + lib.VixDiskLib_Read.argtypes = [ + ctypes.c_void_p, ctypes.c_uint64, ctypes.c_uint64, ctypes.c_char_p] + lib.VixDiskLib_Read.restype = ctypes.c_uint64 + for start, n in ((0, 1), ((1024 * 1024 * 1024) // SECTOR, 1)): + buf = ctypes.create_string_buffer(n * SECTOR) + check(lib.VixDiskLib_Read(disk, start, n, buf), f"Read {start}+{n}") + print( + f"Read start={start} n={n} first16={buf.raw[:16].hex()}", + flush=True, + ) + + lib.VixDiskLib_Close.argtypes = [ctypes.c_void_p] + lib.VixDiskLib_Close.restype = ctypes.c_uint64 + lib.VixDiskLib_Disconnect.argtypes = [ctypes.c_void_p] + lib.VixDiskLib_Disconnect.restype = ctypes.c_uint64 + lib.VixDiskLib_Close(disk) + lib.VixDiskLib_Disconnect(conn) + lib.VixDiskLib_Exit() + + +if __name__ == "__main__": + main() diff --git a/docs/reverse_engineering_procedure.md b/docs/reverse_engineering_procedure.md index c9cce38..db6e65f 100644 --- a/docs/reverse_engineering_procedure.md +++ b/docs/reverse_engineering_procedure.md @@ -9,11 +9,12 @@ NFC work can follow the same loop instead of rediscovering it. Scope so far: `VixDiskLib_ConnectEx` + `VixDiskLib_Open` + `VixDiskLib_Read` + `VixDiskLib_Write` against lab vCenter 8.0.1 / -ESXi 8, transports `nbd`, `nbdssl`, and Linux-guest `hotadd`. -Validation method: `tests/integration/` (the session-scoped `lab` +ESXi 8, transports `nbd`, `nbdssl`, Linux-guest `hotadd`, and Linux +`san`. Validation method: `tests/integration/` (the session-scoped `lab` fixture creates a temporary empty VM with a 10 GiB disk and destroys it when the pytest session ends). HotAdd live tests also SSH into a Linux -proxy guest; see `docs/hotadd.md`. +proxy guest; see `docs/hotadd.md`. SAN live tests present a loop-backed +iSCSI LUN; see `docs/san.md`. Rule from `AGENTS.md`: reuse pyVmomi for every public VIM operation. Only reimplement what pyVmomi does not expose. @@ -63,6 +64,8 @@ names is in this table. | `strace -f -x` on `write` / `send*` | First writable NFC capture without rebuilding the hook (Step 10) | Noisy; TLS still opaque; `-s` truncates large extras | | `pickle` of `LabEnv` | Create the temp VM unhooked, then load it under the hook | `/tmp` only; never commit pickles (lab host and credentials) | | ctypes drivers in `docs/probing_samples/` | Repeatable `ConnectEx` / `Open` / `Read` / `Write` under capture | Not library code | +| in-kernel LIO + `losetup` / `iscsiadm` | File-backed iSCSI LUN so ESXi and the runner share a NAA | Not a VDDK protocol; lab-only (`tests/integration/iscsi_lab.py`) | +| `strace -e openat,pread64,pwrite64,ioctl` | Which `/dev/sd*` / `by-id` VDDK SAN opens and at which offsets | No VMFS structure names; pair with `vixDiskLib.transport.LogLevel=4` | `ltrace` was considered for OpenSSL and libc `write`. It was not used: VDDK is stripped enough that `strace` on syscalls plus the `LD_PRELOAD` @@ -419,3 +422,37 @@ source VMDK to the proxy VM and opens a local whole disk. OpenVixDiskLib reuses pyVmomi `ReconfigureVM` for attach/detach and `pread`/`pwrite` on the Linux SCSI device. NVMe and SATA source disks are remapped onto a proxy SCSI controller. Details: `docs/hotadd.md`. + +## SAN (not NFC) + +SAN is also not a wire protocol. The backup host must see the **same +SCSI LUN** ESXi uses for the VMFS datastore, match it by NAA, then read +the VMDK data file through a VMFS driver. tcpdump of NFC is the wrong +tool. + +Lab: a 20 GiB sparse file, `losetup`, in-kernel LIO iblock + iSCSI +portal on the default-route IPv4, local `iscsiadm` login, pyVmomi +`AddInternetScsiSendTargets` / `CreateVmfsDatastore`. Do not format or +mount the LUN on Linux. + +Probe: `docs/probing_samples/vddk_san_trace.py` with +`vixDiskLib.transport.LogLevel=4`. Native VDDK SAN needs +`.vddk/lib64/libdiskLibPlugin.so`. Capture with +`strace -e openat,pread64,pwrite64,ioctl`. + +Findings used by `openvixdisklib/san.py`: + +- GPT VMFS type GUID `2ae031aa-0f40-db11-9590-000c2911d1b8`, partition + LBA 2048 (1 MiB) +- LVM magic `0xC001D00D` at partition + 1 MiB +- FS magic `0x2fabf15e` version 24 at partition + 2 MiB or + 19 MiB +- File descriptors (`fdmd`) hold 64-bit SFB/LFB pointers at the end of + a two-block descriptor. For large files the pointer array is in 64 KiB + sub-blocks. SFB + `((cluster * resourcesPerCluster) + resource) << fileBlockShift` + is relative to file-block 0, which sits after the LVM label and 16 × + 1 MiB heartbeats. Holes (address 0) read as zeros. Writes need an + allocated file block; SAN does not allocate. +- First cut: powered-off persistent FlatVer2, no snapshot chain. + +Details: `docs/san.md`. diff --git a/docs/san.md b/docs/san.md new file mode 100644 index 0000000..6364ebe --- /dev/null +++ b/docs/san.md @@ -0,0 +1,76 @@ +# SAN transport + +OpenVixDiskLib can read and write a VMDK from a locally visible VMFS +LUN: the same SCSI disk ESXi uses for the datastore. This is not an +NFC protocol. pyVmomi is used only for inventory (datastore NAA / VMFS +UUID). I/O is `pread` / `pwrite` on `/dev/sdX` after a minimal VMFS6 +lookup of the flat extent. + +NBD and NBDSSL remain the default. `transport_modes=None` is still +`nbdssl`. `san` is advertised when `/sys/class/scsi_disk` (or +`scsi_host`) exists, and selected from a colon list such as +`file:san:hotadd:nbdssl:nbd`. `"san"` alone on a host with no SCSI +sysfs raises `NotImplementedError`. + +## Mapping from VDDK + +| VDDK behaviour | OpenVixDiskLib | +| -------------- | -------------- | +| Physical proxy that sees the VMFS LUN | Same. Match by NAA (`VmfsDatastoreInfo.extent.diskName` → `/dev/disk/by-id/wwn-*`). | +| Advanced transport plugin (`libdiskLibPlugin.so`) | Not used. VMFS mapping is in `openvixdisklib/san.py`. | +| Open the LUN `O_DIRECT` | `pread` / `pwrite` on the whole disk (same surface as HotAdd). | +| VMFS driver maps guest sector → file block | GPT VMFS partition, LVM/FS magics, VMDK descriptor scan, file-descriptor pointer walk (SFB/LFB). | +| Snapshot / SESPARSE chains | Not implemented. Powered-off persistent FlatVer2 only. | +| NFS / missing LUN | `RuntimeError`. No silent fallback to NBD. | + +Colon lists pick the first **usable** mode. On this Linux SCSI host that +is `san`. Inside a VMware guest without a shared LUN it is `hotadd`. +Default `None` is still `nbdssl`. + +## Matching and I/O + +1. Resolve `[datastore] path.vmdk`. The datastore must be VMFS + (`VmfsDatastoreInfo`). NFS is rejected. +2. Read the first extent's `canonicalName` (NAA) and the VMFS UUID. +3. Open the local disk whose `/dev/disk/by-id/wwn-0x…` matches that NAA. +4. Parse GPT for the VMFS type GUID + (`2ae031aa-0f40-db11-9590-000c2911d1b8`; ESXi stores RFC UUID bytes + on disk). +5. Check LVM magic `0xC001D00D` at partition + 1 MiB and FS magic + `0x2fabf15e` (version 24) at partition + 2 MiB or + 19 MiB. +6. Scan allocated 1 MiB LUN chunks for `# Disk DescriptorFile` and the + `RW … VMFS "…-flat.vmdk"` extent. +7. Find the VMFS6 regular-file descriptor (`fdmd`) whose + `fileLength` matches that capacity. Pointers are 64-bit, aligned to + the end of the file descriptor (two metadata blocks). Direct SFB/LFB + or one pointer-block level is followed. SFB volume offset is + `((cluster * resourcesPerCluster) + resource) << fileBlockShift` + relative to file-block 0 (after the LVM label and 16 heartbeat + slots). Holes (address 0) read as zeros. Writes need an allocated file + block; SAN does not allocate. + +FastLZ open flags raise `NotImplementedError`. `readinto` returns a +`ReadResult` with empty `fragments`. + +ESXi keeps a VMFS cache of a mounted datastore. Writes from this +initiator are visible to a later SAN open (after the block-device +cache is dropped). They are not visible to `nbdssl` on the same +session. Cross-check nbdssl writes, then SAN reads. + +Implementation: `openvixdisklib.san`. + +## Lab + +Live tests present a 20 GiB sparse file through in-kernel LIO (iblock +on a loop device) so ESXi can create a unique VMFS datastore +`ovdl-iscsi-`. The pytest runner logs in with `iscsiadm` and sees +the same NAA. Configure an optional portal in `.test_config.yaml`: + +```yaml +iscsi_san: + portal: 192.0.2.10 # default: IPv4 of the default route +``` + +Tests skip when passwordless sudo, TCP 3260, or software iSCSI is +missing. They do not create VMkernel NICs from scratch. The +session-wide `lab` VM stays on the YAML `datastore`. diff --git a/openvixdisklib/openvixdisklib.py b/openvixdisklib/openvixdisklib.py index cb91948..a8eaf4a 100644 --- a/openvixdisklib/openvixdisklib.py +++ b/openvixdisklib/openvixdisklib.py @@ -11,7 +11,7 @@ ``VixDiskLibHandle.connect`` / ``open`` / ``read`` match the VDDK wrapper in ``tests/integration/vixdisklib.py``. VIM login uses pyVmomi; NFC ticket, authd, and disk I/O use ``nfc_auth`` and ``nfc_open``. Linux HotAdd uses -``hotadd``. +``hotadd``. Linux SAN uses ``san`` (local SCSI / VMFS I/O). """ from __future__ import annotations @@ -25,7 +25,7 @@ from pyVim.connect import Disconnect from pyVmomi import vim -from openvixdisklib import hotadd, nfc_auth, nfc_open +from openvixdisklib import hotadd, nfc_auth, nfc_open, san ReadResult = nfc_open.ReadResult ReadFragment = nfc_open.ReadFragment @@ -88,6 +88,8 @@ def _parse_vm_moref(vmx_spec: str | None) -> str: def _available_transports() -> list[str]: """Return transports this process can use, in advertisement order.""" modes = ["nbdssl", "nbd"] + if san.is_available(): + modes.append("san") if hotadd.is_vmware_guest(): modes.append("hotadd") return modes @@ -98,8 +100,9 @@ def _select_transport(transport_modes: str | None) -> str: ``None`` defaults to ``nbdssl``. A colon-separated list (VDDK style, for example ``file:san:hotadd:nbdssl:nbd``) picks the first - of ``nbdssl``, ``nbd``, and ``hotadd`` that is usable here. - ``hotadd`` is usable only inside a VMware guest. + of ``san``, ``hotadd``, ``nbdssl``, and ``nbd`` that is usable here. + ``san`` needs a local SCSI disk. ``hotadd`` is usable only inside a + VMware guest. """ if transport_modes is None: return "nbdssl" @@ -136,11 +139,11 @@ def __init__( class _DiskHandle: - """Opened disk (NFC or HotAdd) plus an optional authd TLS socket.""" + """Opened disk (NFC, SAN, or HotAdd) plus an optional authd TLS socket.""" def __init__( self, - disk: nfc_open.NfcDisk | hotadd.HotAddDisk, + disk: nfc_open.NfcDisk | hotadd.HotAddDisk | san.SanDisk, transport_mode: str, authd_sock=None, ) -> None: @@ -234,10 +237,10 @@ def connect( snapshot_ref: Snapshot moref. Unused on the NFC ticket. Required for HotAdd when the source VM is powered on. read_only: When False, the disk may be opened for write. - transport_modes: ``nbdssl``, ``nbd``, ``hotadd``, or a colon - list. The first usable mode is used; ``None`` defaults - to ``nbdssl``. ``hotadd`` is usable only in a VMware - guest. + transport_modes: ``nbdssl``, ``nbd``, ``san``, ``hotadd``, + or a colon list. The first usable mode is used; + ``None`` defaults to ``nbdssl``. ``san`` needs a local + SCSI disk. ``hotadd`` is usable only in a VMware guest. port: HTTPS port, usually 443. allow_untrusted: Skip management TLS verification when True. When False with no ``thumbprint``, the system CA store @@ -289,14 +292,15 @@ def open( aio_buffer_size: int = nfc_open.NFC_AIO_BUFFER_SIZE, aio_buffer_count: int = nfc_open.NFC_AIO_BUFFER_COUNT, ) -> Iterator[_DiskHandle]: - """Open ``disk_path`` over NFC or HotAdd. Matches ``VixDiskLib_Open``. + """Open ``disk_path`` over NFC, SAN, or HotAdd. Matches ``VixDiskLib_Open``. Read-only NFC opens request ``NfcGetVmFiles`` (VM only). The VMDK path, including a snapshot parent such as ``…-000007.vmdk``, is sent on NFC ``OPEN_FILE``. Writable NFC opens use ``NfcRandomAccessOpenDisk`` and resolve a device key from the - disk's backing chain. ``hotadd`` SCSI-attaches the VMDK to this - guest (Linux proxy) and opens the local block device. + disk's backing chain. ``san`` maps the VMFS LUN locally. + ``hotadd`` SCSI-attaches the VMDK to this guest (Linux proxy) + and opens the local block device. Args: conn: Connection from ``connect``. @@ -305,15 +309,15 @@ def open( the disk read-only; omit it for write. ``VIXDISKLIB_FLAG_OPEN_COMPRESSION_FASTLZ`` compresses NFC IO. zlib and skipz are not implemented. Compression - flags are not supported with ``hotadd``. + flags are not supported with ``hotadd`` or ``san``. aio_buffer_size: NFC AIO extra size in bytes, advertised in OPEN_SESSION. Default 64 KiB. ESXi 8 accepts 2 MiB (``2097152``) and rejects 16 MiB and 32 MiB. This is an OpenVixDiskLib extension (VDDK uses ``vixDiskLib.nfcAio.Session.BufSizeIn64KB``). Ignored - for HotAdd. + for HotAdd and SAN. aio_buffer_count: NFC AIO buffer pool count. Default 1. - VDDK's default is 4. Ignored for HotAdd. + VDDK's default is 4. Ignored for HotAdd and SAN. """ LOG.debug("Openning VixDiskLib disk: %s", disk_path) compression = _nfc_compression(flags) @@ -322,18 +326,30 @@ def open( raise NotImplementedError("ConnectEx was read-only; cannot open for write") vm = vim.VirtualMachine(conn.vm_moref, conn.si._stub) - if conn.transport_mode == "hotadd": + if conn.transport_mode in ("hotadd", "san"): if compression != nfc_open.NFC_COMPRESSION_NONE: raise NotImplementedError( - "NBD compression open flags are not supported with hotadd" + "NBD compression open flags are not supported with " + f"{conn.transport_mode}" + ) + if conn.transport_mode == "hotadd": + disk: nfc_open.NfcDisk | hotadd.HotAddDisk | san.SanDisk = ( + hotadd.open_disk( + conn.si, + vm, + disk_path, + snapshot_ref=conn.snapshot_ref, + read_only=read_only, + ) + ) + else: + disk = san.open_disk( + conn.si, + vm, + disk_path, + snapshot_ref=conn.snapshot_ref, + read_only=read_only, ) - disk = hotadd.open_disk( - conn.si, - vm, - disk_path, - snapshot_ref=conn.snapshot_ref, - read_only=read_only, - ) handle = _DiskHandle(disk, conn.transport_mode) try: yield handle diff --git a/openvixdisklib/san.py b/openvixdisklib/san.py new file mode 100644 index 0000000..b999743 --- /dev/null +++ b/openvixdisklib/san.py @@ -0,0 +1,867 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Linux SAN transport: open a VMDK from a locally visible VMFS LUN. + +The backup host must see the same SCSI LUN ESXi uses for the datastore +(matched by NAA). This is not NFC: inventory uses pyVmomi, I/O is +``pread`` / ``pwrite`` on the local disk after a minimal VMFS6 read of +the flat extent. Snapshot chains and VMFS allocation on write are not +implemented; holes read as zeros and writes need an existing file block. +""" + +from __future__ import annotations + +import fcntl +import logging +import os +import re +import struct +from collections.abc import Iterable +from dataclasses import dataclass + +from pyVmomi import vim + +from openvixdisklib.nfc_open import ReadResult + +LOG = logging.getLogger(__name__) + +SECTOR_SIZE = 512 +FILE_BLOCK_SIZE = 1024 * 1024 +SCSI_DISK_DIR = "/sys/class/scsi_disk" +DISK_BY_ID = "/dev/disk/by-id" +LVM_MAGIC = 0xC001D00D +FS_MAGIC = 0x2FABF15E +FS_MAGIC_L = 0x2FABF15F +LVM_OFFSET = 0x100000 +HEARTBEAT_COUNT = 16 +FS_HEADER_OFFSETS = (0x200000, 0x1300000) +FDMD_MAGIC = 0x66646D64 +RFMD_MAGIC = 0x72666D64 +GPT_SIG = b"EFI PART" +# GPT type GUID as stored by ESXi (RFC UUID bytes, not mixed-endian). +# Same GUID as 2ae031aa-0f40-db11-9590-000c2911d1b8 / AA31E02A-400F-11DB-... +VMFS_TYPE_GUIDS = ( + bytes.fromhex("2ae031aa0f40db119590000c2911d1b8"), + bytes.fromhex("aa31e02a400f11db9590000c2911d1b8"), +) +DESCRIPTOR_NEEDLE = b"# Disk DescriptorFile" +SCAN_LIMIT = 20 * 1024 * 1024 * 1024 +BLKFLSBUF = 0x1261 +ADDR_SFB = 0x1 +ADDR_SB = 0x2 +ADDR_LFB = 0x7 +ZLA_FILE_BLOCK = 0x1 +ZLA_SUB_BLOCK = 0x2 +ZLA_POINTER_BLOCK = 0x3 +ZLA_POINTER2_BLOCK = 0x5 +DESC_REGFILE = 3 + + +@dataclass(frozen=True) +class Extent: + """Map ``length`` bytes of a VMFS file onto a LUN offset.""" + + file_offset: int + lun_offset: int + length: int + + +@dataclass(frozen=True) +class GptPartition: + """A GPT partition used as a VMFS extent.""" + + start_lba: int + end_lba: int + + @property + def start_bytes(self) -> int: + """Byte offset of the partition on the LUN.""" + return self.start_lba * SECTOR_SIZE + + +@dataclass(frozen=True) +class FsInfo: + """Subset of the VMFS6 filesystem descriptor needed for SAN I/O.""" + + offset: int + uuid: str + file_block_size: int + md_alignment: int + sfb_to_lfb_shift: int + ptr_block_shift: int + major_version: int + + +@dataclass(frozen=True) +class FileMeta: + """VMFS file-descriptor metadata for a regular file.""" + + fd_offset: int + file_length: int + block_size: int + zla: int + num_blocks: int + pointers: tuple[int, ...] + + +def is_available() -> bool: + """Return True when this Linux host can see SCSI disks.""" + return os.path.isdir(SCSI_DISK_DIR) or os.path.isdir("/sys/class/scsi_host") + + +def parse_datastore_path(disk_path: str) -> tuple[str, str]: + """Split ``[datastore] rel/path.vmdk`` into name and relative path. + + Args: + disk_path: Datastore path of the VMDK descriptor. + """ + match = re.match(r"^\[([^\]]+)\]\s*(.*)$", disk_path.strip()) + if not match or not match.group(1) or not match.group(2): + raise ValueError(f"unsupported disk path: {disk_path!r}") + return match.group(1), match.group(2).replace("\\", "/") + + +def flat_extent_name(descriptor_relpath: str) -> str: + """Return the sibling ``-flat.vmdk`` basename for ``descriptor_relpath``.""" + base = os.path.basename(descriptor_relpath) + if base.endswith("-flat.vmdk"): + return base + if base.endswith(".vmdk"): + return base[:-5] + "-flat.vmdk" + return base + + +def normalize_naa(value: str) -> str: + """Return a canonical ``naa.`` string.""" + text = value.lower().strip() + if text.startswith("wwn-0x"): + text = text[6:] + elif text.startswith("0x"): + text = text[2:] + elif text.startswith("scsi-3"): + text = text[6:] + elif text.startswith("naa."): + text = text[4:] + hexpart = "".join(ch for ch in text if ch in "0123456789abcdef") + if not hexpart: + raise ValueError(f"cannot parse NAA from {value!r}") + return "naa." + hexpart + + +def find_local_device(naa: str) -> str: + """Return the local block device whose WWN matches ``naa``. + + Args: + naa: Canonical name such as ``naa.6001405...``. + """ + hexpart = normalize_naa(naa)[4:] + if os.path.isdir(DISK_BY_ID): + for name in sorted(os.listdir(DISK_BY_ID)): + if "-part" in name: + continue + lower = name.lower() + if lower.startswith("wwn-0x") and lower[6:] == hexpart: + return os.path.realpath(os.path.join(DISK_BY_ID, name)) + if lower.startswith("scsi-3") and lower[6:] == hexpart: + return os.path.realpath(os.path.join(DISK_BY_ID, name)) + raise RuntimeError(f"no local SCSI disk matching {naa}") + + +def vmfs_extent_naa(datastore: vim.Datastore) -> str: + """Return the NAA of the first VMFS extent of ``datastore``.""" + info = datastore.info + if not isinstance(info, vim.host.VmfsDatastoreInfo) or info.vmfs is None: + raise RuntimeError(f"datastore {datastore.name!r} is not VMFS") + extents = list(info.vmfs.extent or []) + if not extents: + raise RuntimeError(f"VMFS datastore {datastore.name!r} has no extents") + return normalize_naa(str(extents[0].diskName)) + + +def vmfs_uuid(datastore: vim.Datastore) -> str: + """Return the VMFS UUID string from pyVmomi.""" + info = datastore.info + if not isinstance(info, vim.host.VmfsDatastoreInfo) or info.vmfs is None: + raise RuntimeError(f"datastore {datastore.name!r} is not VMFS") + uuid = getattr(info.vmfs, "uuid", None) + if not uuid: + raise RuntimeError(f"VMFS datastore {datastore.name!r} has no UUID") + return str(uuid).lower() + + +def parse_gpt_vmfs(fd: int) -> GptPartition: + """Return the first GPT partition with the VMFS type GUID.""" + header = os.pread(fd, 512, 512) + if header[:8] != GPT_SIG: + raise RuntimeError("LUN has no GPT header") + (part_lba,) = struct.unpack_from(" int: + return address & 0x7 + + +def parse_sfb(address: int) -> tuple[int, int]: + """Return ``(cluster, resource)`` from a VMFS6 small-file-block address.""" + cluster = (address >> 15) & 0x7FFFFFFF + resource = (address >> 51) & 0x1FFF + return cluster, resource + + +def parse_lfb(address: int) -> int: + """Return the block number from a VMFS6 large-file-block address.""" + return (address >> 15) & 0x7FFFFFFF + + +def sfb_volume_offset( + address: int, resources_per_cluster: int, file_block_shift: int +) -> int: + """Return the volume byte offset of a VMFS6 SFB address.""" + cluster, resource = parse_sfb(address) + return ((cluster * resources_per_cluster) + resource) << file_block_shift + + +def lfb_volume_offset( + address: int, file_block_shift: int, sfb_to_lfb_shift: int +) -> int: + """Return the volume byte offset of a VMFS6 LFB address.""" + return parse_lfb(address) << (file_block_shift + sfb_to_lfb_shift) + + +def _file_block_shift(file_block_size: int) -> int: + if file_block_size <= 0 or file_block_size & (file_block_size - 1): + raise RuntimeError(f"fileBlockSize {file_block_size} is not a power of two") + return file_block_size.bit_length() - 1 + + +def _default_sfb_rpc(file_block_size: int) -> int: + return min(0x2000, 0x20000000 // max(file_block_size, 1)) + + +def parse_vmdk_descriptor(text: str) -> tuple[int, str]: + """Return ``(capacity_sectors, flat_basename)`` from a VMDK descriptor. + + Args: + text: Descriptor file contents. + """ + extent_re = re.compile( + r'^\s*RW\s+(\d+)\s+VMFS\s+"([^"]+)"\s*$', re.MULTILINE | re.IGNORECASE + ) + match = extent_re.search(text) + if not match: + raise RuntimeError("VMDK descriptor has no RW VMFS extent") + return int(match.group(1)), match.group(2) + + +def _flush_block_device(fd: int) -> None: + """Drop kernel buffer cache for ``fd`` so initiator reads see target writes.""" + try: + fcntl.ioctl(fd, BLKFLSBUF) + except OSError: + LOG.debug("BLKFLSBUF failed", exc_info=True) + try: + os.posix_fadvise(fd, 0, 0, os.POSIX_FADV_DONTNEED) + except OSError: + LOG.debug("posix_fadvise DONTNEED failed", exc_info=True) + + +def _iter_allocated_mbs(fd: int, limit: int) -> Iterable[int]: + size = os.lseek(fd, 0, os.SEEK_END) + end = min(size, limit) + off = 0 + while off < end: + for sample in range(0, FILE_BLOCK_SIZE, 64 * 1024): + head = os.pread(fd, 16, off + sample) + if head and max(head) != 0: + yield off + break + off += FILE_BLOCK_SIZE + + +def find_vmdk_descriptor( + fd: int, relpath: str, limit: int = SCAN_LIMIT +) -> tuple[int, str]: + """Scan allocated LUN blocks for the VMDK descriptor of ``relpath``.""" + base = os.path.basename(relpath) + flat = flat_extent_name(relpath) + found: tuple[int, str] | None = None + for off in _iter_allocated_mbs(fd, limit): + chunk = os.pread(fd, FILE_BLOCK_SIZE, off) + pos = 0 + while True: + idx = chunk.find(DESCRIPTOR_NEEDLE, pos) + if idx < 0: + break + text = ( + chunk[idx : idx + 4096].split(b"\x00", 1)[0].decode("utf-8", "replace") + ) + if base in text or flat in text: + return off + idx, text + if found is None: + found = (off + idx, text) + pos = idx + 1 + if found is not None: + return found + raise RuntimeError(f"VMDK descriptor for {relpath!r} not found on the LUN") + + +def _uuid_from_fs(raw: bytes) -> str: + time_lo, time_hi = struct.unpack_from(" FsInfo: + """Read the VMFS6 filesystem descriptor from the partition.""" + for rel in FS_HEADER_OFFSETS: + offset = part.start_bytes + rel + blob = os.pread(fd, 0x180, offset) + (magic,) = struct.unpack_from(" tuple[int, int, int]: + """Return ``(fd_size, data_addrs_offset, data_addrs_size)`` for VMFS6.""" + fd_size = 2 * md_alignment + if md_alignment <= 0x1000: + data_addrs_size = 2560 + else: + data_addrs_size = md_alignment >> 1 + return fd_size, fd_size - data_addrs_size, data_addrs_size + + +def _unpack_u64s(data: bytes) -> list[int]: + count = len(data) // 8 + return list(struct.unpack_from(f"<{count}Q", data, 0)) if count else [] + + +def _find_fbb_rpc(fd: int, part: GptPartition, fs: FsInfo, limit: int) -> int: + """Return ``resourcesPerCluster`` from an FBB RFMD header, if present.""" + del part + default = _default_sfb_rpc(fs.file_block_size) + needle = struct.pack("= 0x20: + hdr = chunk[pos - 0x20 : pos - 0x20 + 0x30] + if len(hdr) >= 0x24: + (rpc,) = struct.unpack_from(" FileMeta | None: + fd_size, addrs_off, addrs_size = _fd_layout(fs.md_alignment) + raw = os.pread(fd, fd_size, fd_offset) + if len(raw) < fd_size: + return None + meta = raw[fs.md_alignment :] + if len(meta) < 0x68: + return None + (magic,) = struct.unpack_from(" FileMeta: + """Find a regular-file descriptor whose length matches ``file_length``.""" + needle = struct.pack("= 0x64: + fd_off = off + pos - 0x64 - fs.md_alignment + if fd_off < 0: + idx = pos + 1 + continue + meta = _read_file_meta(fd, fd_off, fs) + if meta is not None and meta.file_length == file_length: + matches.append(meta) + idx = pos + 1 + if not matches: + raise RuntimeError( + f"no VMFS regular file of length {file_length} found on the LUN" + ) + if len(matches) > 1: + LOG.warning( + "multiple VMFS files of length %s; using offset %#x", + file_length, + matches[0].fd_offset, + ) + return matches[0] + + +def _file_block_base(part: GptPartition, file_block_size: int) -> int: + """Return the LUN offset of small-file-block 0. + + File blocks start after the GPT partition header, the 1 MiB LVM + label, and 16 × 1 MiB heartbeat slots. + """ + return part.start_bytes + LVM_OFFSET + HEARTBEAT_COUNT * file_block_size + + +def _block_lun_offset( + address: int, + part: GptPartition, + fs: FsInfo, + sfb_rpc: int, +) -> int | None: + kind = _addr_type(address) + shift = _file_block_shift(fs.file_block_size) + base = _file_block_base(part, fs.file_block_size) + if kind == ADDR_SFB: + return base + sfb_volume_offset(address, sfb_rpc, shift) + if kind == ADDR_LFB: + lfb_shift = fs.sfb_to_lfb_shift or 9 + return base + lfb_volume_offset(address, shift, lfb_shift) + return None + + +def _read_pointer_array(fd: int, lun_offset: int, count: int) -> list[int]: + data = os.pread(fd, count * 8, lun_offset) + return _unpack_u64s(data)[:count] + + +def _is_data_block_addr(address: int) -> bool: + return address != 0 and _addr_type(address) in (ADDR_SFB, ADDR_LFB) + + +def _scan_pointer_block_array(fd: int, nblocks: int, limit: int) -> list[int]: + """Find a uint64 SFB/LFB pointer array when inode pointers are sub-blocks.""" + slot = 65536 + slots: list[tuple[int, list[int]]] = [] + for off in _iter_allocated_mbs(fd, limit): + chunk = os.pread(fd, FILE_BLOCK_SIZE, off) + for start in range(0, FILE_BLOCK_SIZE, slot): + vals = _unpack_u64s(chunk[start : start + slot]) + score = sum(1 for val in vals if _is_data_block_addr(val)) + if score >= 512: + slots.append((off + start, vals)) + if not slots: + return [] + slots.sort(key=lambda item: item[0]) + best: list[int] = [] + run: list[int] = [] + run_off: int | None = None + for off, vals in slots: + if run_off is not None and off != run_off + slot: + if len(run) > len(best): + best = run + run = [] + run.extend(vals) + run_off = off + if len(run) > len(best): + best = run + while best and not _is_data_block_addr(best[0]): + best.pop(0) + if sum(1 for val in best[:nblocks] if _is_data_block_addr(val)) < min(nblocks, 2): + return [] + if len(best) < nblocks: + best.extend([0] * (nblocks - len(best))) + return best[:nblocks] + + +def _expand_pointers( + fd: int, + meta: FileMeta, + part: GptPartition, + fs: FsInfo, + sfb_rpc: int, +) -> list[int]: + """Return per-file-block addresses (SFB/LFB), following one pointer level.""" + block_size = meta.block_size or fs.file_block_size + nblocks = max(meta.num_blocks, (meta.file_length + block_size - 1) // block_size) + ptrs = list(meta.pointers) + if meta.zla in (ZLA_FILE_BLOCK, ZLA_SUB_BLOCK, 0): + return ptrs[:nblocks] + if meta.zla in (ZLA_POINTER_BLOCK, ZLA_POINTER2_BLOCK): + if any(_addr_type(ptr) == ADDR_SB for ptr in ptrs if ptr): + scanned = _scan_pointer_block_array(fd, nblocks, SCAN_LIMIT) + if scanned: + return scanned + expanded: list[int] = [] + remaining = nblocks + for ptr in ptrs: + if remaining <= 0: + break + if ptr == 0: + expanded.extend([0] * min(remaining, 8192)) + remaining -= min(remaining, 8192) + continue + lun = _block_lun_offset(ptr, part, fs, sfb_rpc) + if lun is None: + LOG.debug("SAN pointer %#x is not an SFB/LFB; stopping expand", ptr) + break + chunk = _read_pointer_array( + fd, lun, min(remaining, fs.file_block_size // 8) + ) + expanded.extend(chunk) + remaining -= len(chunk) + if sum(1 for addr in expanded if addr) >= min(nblocks, 2): + return expanded[:nblocks] + scanned = _scan_pointer_block_array(fd, nblocks, SCAN_LIMIT) + if scanned: + return scanned + return expanded[:nblocks] + return ptrs[:nblocks] + + +def extents_from_file_meta( + fd: int, + meta: FileMeta, + part: GptPartition, + fs: FsInfo, + sfb_rpc: int, +) -> list[Extent]: + """Build a file-offset map from a VMFS file descriptor.""" + block_size = meta.block_size or fs.file_block_size + addresses = _expand_pointers(fd, meta, part, fs, sfb_rpc) + extents: list[Extent] = [] + for index, address in enumerate(addresses): + if not address: + continue + lun = _block_lun_offset(address, part, fs, sfb_rpc) + if lun is None: + continue + file_off = index * block_size + if file_off >= meta.file_length: + break + length = min(block_size, meta.file_length - file_off) + extents.append(Extent(file_offset=file_off, lun_offset=lun, length=length)) + return _coalesce_extents(extents, meta.file_length) + + +def _coalesce_extents(extents: list[Extent], capacity_bytes: int) -> list[Extent]: + clipped: list[Extent] = [] + for extent in extents: + if extent.file_offset >= capacity_bytes: + continue + length = min(extent.length, capacity_bytes - extent.file_offset) + if not clipped: + clipped.append(Extent(extent.file_offset, extent.lun_offset, length)) + continue + prev = clipped[-1] + if ( + prev.file_offset + prev.length == extent.file_offset + and prev.lun_offset + prev.length == extent.lun_offset + ): + clipped[-1] = Extent( + prev.file_offset, prev.lun_offset, prev.length + length + ) + else: + clipped.append(Extent(extent.file_offset, extent.lun_offset, length)) + return clipped + + +def map_file_range( + extents: Iterable[Extent], file_offset: int, length: int +) -> list[Extent]: + """Clip ``extents`` to the file range ``[file_offset, file_offset+length)``.""" + end = file_offset + length + hits: list[Extent] = [] + for extent in extents: + ext_end = extent.file_offset + extent.length + if ext_end <= file_offset or extent.file_offset >= end: + continue + start = max(file_offset, extent.file_offset) + stop = min(end, ext_end) + shift = start - extent.file_offset + hits.append( + Extent( + file_offset=start, + lun_offset=extent.lun_offset + shift, + length=stop - start, + ) + ) + return hits + + +def _pread_all(fd: int, size: int, offset: int) -> bytes: + chunks = bytearray() + remaining = size + pos = offset + while remaining: + data = os.pread(fd, remaining, pos) + if not data: + raise OSError(f"short read at offset {pos}: got {len(chunks)} of {size}") + chunks.extend(data) + remaining -= len(data) + pos += len(data) + return bytes(chunks) + + +def _pwrite_all(fd: int, data: bytes, offset: int) -> None: + remaining = memoryview(data) + pos = offset + while remaining: + written = os.pwrite(fd, remaining, pos) + if written <= 0: + raise OSError(f"short write at offset {pos}") + remaining = remaining[written:] + pos += written + + +class SanDisk: + """A VMDK opened as VMFS file extents on a local SCSI disk.""" + + def __init__( + self, + fd: int, + dev_path: str, + extents: list[Extent], + capacity_bytes: int, + sector_size: int = SECTOR_SIZE, + ) -> None: + """Wrap an open LUN fd and a VMFS file extent map. + + Args: + fd: File descriptor for the whole SCSI disk. + dev_path: Local path such as ``/dev/sdb``. + extents: Allocated VMFS file-block map. + capacity_bytes: Virtual size of the VMDK. + sector_size: Sector size in bytes (VDDK uses 512). + """ + self._fd = fd + self.dev_path = dev_path + self.extents = extents + self.capacity_bytes = capacity_bytes + self.sector_size = sector_size + self._closed = False + + def readinto( + self, + start_sector: int, + num_sectors: int, + buf: bytearray | memoryview, + skip_decompression: bool = False, + ) -> ReadResult: + """Read ``num_sectors`` into ``buf`` starting at ``start_sector``. + + Unmapped VMFS file blocks are returned as zeros. ``fragments`` is + always empty. + + Args: + start_sector: Sector offset from the start of the virtual disk. + num_sectors: Number of sectors to read. + buf: Destination buffer. + skip_decompression: Ignored; accepted for API compatibility. + """ + del skip_decompression + if num_sectors < 1: + raise ValueError("num_sectors must be at least 1") + length = num_sectors * self.sector_size + view = buf if isinstance(buf, memoryview) else memoryview(buf) + if view.readonly: + raise TypeError("read buffer is read-only") + raw = view.cast("B") if view.format != "B" else view + if len(raw) < length: + raise RuntimeError(f"read buffer is {len(raw)} bytes, need {length}") + file_off = start_sector * self.sector_size + raw[:length] = b"\x00" * length + for extent in map_file_range(self.extents, file_off, length): + data = _pread_all(self._fd, extent.length, extent.lun_offset) + dest = extent.file_offset - file_off + raw[dest : dest + extent.length] = data + return ReadResult( + uncompressed_length=length, compressed_length=length, fragments=() + ) + + def write(self, start_sector: int, num_sectors: int, data: bytes) -> None: + """Write ``num_sectors`` starting at ``start_sector``. + + Args: + start_sector: Sector offset from the start of the virtual disk. + num_sectors: Number of sectors to write. + data: Bytes to write; length must be ``num_sectors * sector_size``. + """ + if num_sectors < 1: + raise ValueError("num_sectors must be at least 1") + length = num_sectors * self.sector_size + if len(data) != length: + raise ValueError(f"write data is {len(data)} bytes, need {length}") + file_off = start_sector * self.sector_size + mapped = map_file_range(self.extents, file_off, length) + covered = sum(extent.length for extent in mapped) + if covered != length: + raise RuntimeError( + "SAN write needs allocated VMFS file blocks for the whole " + f"range (mapped {covered} of {length} bytes)" + ) + payload = memoryview(data) + for extent in mapped: + src = extent.file_offset - file_off + _pwrite_all( + self._fd, + payload[src : src + extent.length].tobytes(), + extent.lun_offset, + ) + os.fsync(self._fd) + + def close(self) -> None: + """Close the LUN file descriptor.""" + if self._closed: + return + try: + os.close(self._fd) + except OSError: + pass + self._closed = True + + def __enter__(self) -> SanDisk: + return self + + def __exit__(self, exc_type, exc, tb) -> None: + self.close() + + +def _find_datastore(si: vim.ServiceInstance, name: str) -> vim.Datastore: + content = si.RetrieveContent() + container = content.viewManager.CreateContainerView( + content.rootFolder, [vim.Datastore], True + ) + try: + for datastore in container.view: + if datastore.name == name: + return datastore + finally: + container.Destroy() + raise RuntimeError(f"datastore {name!r} not found") + + +def _disk_on_vm(source_vm: vim.VirtualMachine, disk_path: str) -> None: + devices = source_vm.config.hardware.device if source_vm.config else [] + for device in devices: + backing = getattr(device, "backing", None) + if backing is not None and getattr(backing, "fileName", None) == disk_path: + return + raise RuntimeError(f"{disk_path} is not a virtual disk of {source_vm._moId}") + + +def open_disk( + si: vim.ServiceInstance, + source_vm: vim.VirtualMachine, + disk_path: str, + snapshot_ref: str | None = None, + read_only: bool = True, +) -> SanDisk: + """Open ``disk_path`` from a locally visible VMFS LUN. + + Args: + si: Logged-in VIM session. + source_vm: VM that owns ``disk_path``. + disk_path: Datastore path of the VMDK descriptor. + snapshot_ref: Unused; snapshot chains are not implemented. + read_only: Open the LUN read-only when True. + """ + del snapshot_ref + if not is_available(): + raise RuntimeError("SAN transport requires a Linux host with SCSI disks") + ds_name, relpath = parse_datastore_path(disk_path) + datastore = _find_datastore(si, ds_name) + naa = vmfs_extent_naa(datastore) + expected_uuid = vmfs_uuid(datastore) + _disk_on_vm(source_vm, disk_path) + dev_path = find_local_device(naa) + flags = os.O_RDONLY if read_only else os.O_RDWR + fd = os.open(dev_path, flags) + try: + _flush_block_device(fd) + part = parse_gpt_vmfs(fd) + lvm = os.pread(fd, 4, part.start_bytes + LVM_OFFSET) + (lvm_magic,) = struct.unpack_from(" dict[str, str] | None: } +def load_iscsi_san_config() -> dict[str, str] | None: + """Return optional iSCSI SAN lab settings from ``.test_config.yaml``. + + The ``iscsi_san`` section is optional. When present, ``portal`` overrides + the default-route IPv4 used as the LIO listen address. + """ + if not os.path.isfile(_CONFIG_PATH): + return None + with open(_CONFIG_PATH, encoding="utf-8") as config_file: + data = yaml.safe_load(config_file) or {} + section = data.get("iscsi_san") + if not isinstance(section, dict): + return {} + result: dict[str, str] = {} + if section.get("portal"): + result["portal"] = str(section["portal"]) + return result + + def _connect_vim( host: str, username: str, @@ -222,7 +241,10 @@ def _find_datastore(datacenter: vim.Datacenter, datastore_name: str) -> vim.Data def _vm_config_spec( - vm_name: str, datastore_name: str, disk_controller: str = "pvscsi" + vm_name: str, + datastore_name: str, + disk_controller: str = "pvscsi", + thin_provisioned: bool = True, ) -> vim.vm.ConfigSpec: config = vim.vm.ConfigSpec() config.name = vm_name @@ -252,7 +274,8 @@ def _vm_config_spec( backing = vim.vm.device.VirtualDisk.FlatVer2BackingInfo() backing.diskMode = "persistent" - backing.thinProvisioned = True + backing.thinProvisioned = thin_provisioned + backing.eagerlyScrub = not thin_provisioned backing.fileName = f"[{datastore_name}]" disk = vim.vm.device.VirtualDisk() disk.key = 2000 @@ -269,13 +292,22 @@ def _vm_config_spec( return config -def create_lab_vm(*, disk_controller: str = "pvscsi") -> LabEnv: - """Create an empty VM with a 10 GiB thin disk for I/O tests. +def create_lab_vm( + *, + disk_controller: str = "pvscsi", + datastore: str | None = None, + thin_provisioned: bool = True, +) -> LabEnv: + """Create an empty VM with a 10 GiB disk for I/O tests. Args: disk_controller: ``pvscsi`` (default) or ``nvme``. + datastore: Datastore name. Defaults to ``.test_config.yaml``. + thin_provisioned: Thin VMDK when True. SAN tests use False so the + flat extent is preallocated on VMFS. """ cfg = _load_test_config() + datastore_name = datastore or cfg["datastore"] thumbprint = nfc_auth.get_ssl_cert_thumbprint(cfg["host"], cfg["port"]) si = _connect_vim( cfg["host"], @@ -289,18 +321,21 @@ def create_lab_vm(*, disk_controller: str = "pvscsi") -> LabEnv: try: content = si.RetrieveContent() datacenter = _find_datacenter(content, cfg["datacenter"]) - datastore = _find_datastore(datacenter, cfg["datastore"]) - if not datastore.host: + datastore_obj = _find_datastore(datacenter, datastore_name) + if not datastore_obj.host: raise RuntimeError( - f"datastore {cfg['datastore']!r} is not mounted on any host" + f"datastore {datastore_name!r} is not mounted on any host" ) - host = datastore.host[0].key + host = datastore_obj.host[0].key pool = host.parent.resourcePool vm_name = _LAB_VM_PREFIX + uuid.uuid4().hex[:12] vm = _wait_for_task( datacenter.vmFolder.CreateVM_Task( config=_vm_config_spec( - vm_name, datastore.name, disk_controller=disk_controller + vm_name, + datastore_obj.name, + disk_controller=disk_controller, + thin_provisioned=thin_provisioned, ), pool=pool, host=host, @@ -320,7 +355,7 @@ def create_lab_vm(*, disk_controller: str = "pvscsi") -> LabEnv: password=cfg["password"], allow_untrusted=cfg["allow_untrusted"], datacenter=cfg["datacenter"], - datastore=cfg["datastore"], + datastore=datastore_obj.name, thumbprint=thumbprint, vm_moref=vm._moId, vmx_spec=f"moref={vm._moId}", diff --git a/tests/integration/iscsi_lab.py b/tests/integration/iscsi_lab.py new file mode 100644 index 0000000..55698d6 --- /dev/null +++ b/tests/integration/iscsi_lab.py @@ -0,0 +1,898 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""File-backed iSCSI LUN for SAN transport tests. + +Presents a loop device through in-kernel LIO so lab ESXi can create a +temporary VMFS datastore. The pytest runner also logs in as an initiator +so the same NAA is visible locally for SAN I/O. Not part of the library. +""" + +from __future__ import annotations + +import contextlib +import json +import logging +import os +import socket +import subprocess +import time +import uuid +from collections.abc import Iterator +from dataclasses import dataclass +from typing import Any + +from pyVim.connect import Disconnect +from pyVmomi import vim + +from openvixdisklib import nfc_auth +from tests.integration.base import ( + _connect_vim, + _find_datacenter, + _find_datastore, + _load_test_config, + _wait_for_task, + load_iscsi_san_config, +) + +LOG = logging.getLogger(__name__) + +IQN_PREFIX = "iqn.2026-09.io.openvixdisklib:" +DATASTORE_PREFIX = "ovdl-iscsi-" +STATE_DIR = "/var/tmp" +CONFIGFS_TARGET = "/sys/kernel/config/target" +ISCSI_PORT = 3260 +LUN_SIZE_BYTES = 20 * 1024 * 1024 * 1024 +HBA_WAIT_S = 180 +HBA_POLL_S = 2 +DEVICE_WAIT_S = 60 +DEVICE_POLL_S = 0.5 +LIO_HBA = "iblock_0" + + +@dataclass +class IscsiSanLab: + """Runtime state for one file-backed iSCSI VMFS datastore.""" + + lab_id: str + portal: str + port: int + iqn: str + img_path: str + loop_dev: str + naa: str + local_dev: str + datastore_name: str + host_moref: str + backstore_name: str + iptables_added: bool = False + enabled_software_iscsi: bool = False + bound_vnic: str = "" + + +def default_portal_ip() -> str: + """Return the IPv4 address used for the default route.""" + sock = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) + try: + sock.connect(("8.8.8.8", 80)) + ip = sock.getsockname()[0] + finally: + sock.close() + if not ip or ip.startswith("127."): + raise RuntimeError("could not determine a non-loopback portal IPv4 address") + return str(ip) + + +def _need_sudo() -> bool: + return os.geteuid() != 0 + + +def _priv_args(args: list[str]) -> list[str]: + if _need_sudo(): + return ["sudo", "-n", *args] + return args + + +def _run( + args: list[str], + check: bool = True, + privileged: bool = False, + input_text: str | None = None, +) -> subprocess.CompletedProcess[str]: + cmd = _priv_args(args) if privileged else args + LOG.debug("run %s", cmd) + result = subprocess.run( + cmd, capture_output=True, text=True, check=False, input=input_text + ) + if check and result.returncode != 0: + raise RuntimeError( + f"{cmd[0]} failed rc={result.returncode}: " + f"{result.stderr.strip() or result.stdout.strip()}" + ) + return result + + +def _write_attr(path: str, value: str) -> None: + payload = value if value.endswith("\n") else value + "\n" + if not _need_sudo(): + with open(path, "w", encoding="ascii") as handle: + handle.write(payload) + return + result = _run(["tee", path], privileged=True, input_text=payload, check=False) + if result.returncode != 0: + raise RuntimeError(f"cannot write {path}: {result.stderr.strip()}") + + +def _read_attr(path: str) -> str: + if not _need_sudo(): + with open(path, encoding="ascii") as handle: + return handle.read().strip() + result = _run(["cat", path], privileged=True) + return result.stdout.strip() + + +def _mkdir(path: str) -> None: + if not _need_sudo(): + os.makedirs(path, exist_ok=True) + return + _run(["mkdir", "-p", path], privileged=True) + + +def _rmdir(path: str) -> None: + if not _need_sudo(): + os.rmdir(path) + return + _run(["rmdir", path], privileged=True, check=False) + + +def _unlink(path: str) -> None: + if not _need_sudo(): + os.unlink(path) + return + _run(["rm", "-f", path], privileged=True, check=False) + + +def _symlink(target: str, path: str) -> None: + if not _need_sudo(): + os.symlink(target, path) + return + _run(["ln", "-s", target, path], privileged=True) + + +def _listdir(path: str) -> list[str]: + try: + return os.listdir(path) + except FileNotFoundError: + return [] + + +def _load_lio_modules() -> None: + for module in ("target_core_mod", "target_core_iblock", "iscsi_target_mod"): + result = _run(["modprobe", module], check=False, privileged=True) + if result.returncode != 0: + raise RuntimeError( + f"modprobe {module} failed: {result.stderr.strip() or result.stdout.strip()}" + ) + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + if os.path.isdir(CONFIGFS_TARGET): + _mkdir(os.path.join(CONFIGFS_TARGET, "core")) + _mkdir(os.path.join(CONFIGFS_TARGET, "iscsi")) + return + time.sleep(0.1) + raise RuntimeError(f"{CONFIGFS_TARGET} did not appear after loading LIO modules") + + +def _create_sparse_file(path: str, size: int) -> None: + fd = os.open(path, os.O_RDWR | os.O_CREAT | os.O_EXCL, 0o600) + try: + os.ftruncate(fd, size) + finally: + os.close(fd) + + +def _attach_loop(img_path: str) -> str: + result = _run( + ["losetup", "--find", "--show", "--direct-io=on", img_path], + check=False, + privileged=True, + ) + if result.returncode != 0: + result = _run(["losetup", "--find", "--show", img_path], privileged=True) + loop_dev = result.stdout.strip() + if not loop_dev.startswith("/dev/loop"): + raise RuntimeError(f"losetup returned unexpected device {loop_dev!r}") + return loop_dev + + +def _setup_lio_target( + lab_id: str, loop_dev: str, portal: str, port: int, iqn: str, backstore: str +) -> None: + _load_lio_modules() + hba_dir = os.path.join(CONFIGFS_TARGET, "core", LIO_HBA) + _mkdir(hba_dir) + bs_dir = os.path.join(hba_dir, backstore) + _mkdir(bs_dir) + _write_attr(os.path.join(bs_dir, "control"), f"udev_path={loop_dev}") + serial = f"ovdl{lab_id}"[:36] + _write_attr(os.path.join(bs_dir, "wwn", "vpd_unit_serial"), serial) + _write_attr(os.path.join(bs_dir, "enable"), "1") + + tgt_dir = os.path.join(CONFIGFS_TARGET, "iscsi", iqn) + _mkdir(tgt_dir) + tpgt_dir = os.path.join(tgt_dir, "tpgt_1") + _mkdir(tpgt_dir) + lun_dir = os.path.join(tpgt_dir, "lun", "lun_0") + _mkdir(lun_dir) + link_path = os.path.join(lun_dir, backstore) + if not os.path.lexists(link_path): + _symlink(bs_dir, link_path) + _mkdir(os.path.join(tpgt_dir, "np", f"{portal}:{port}")) + attrib = os.path.join(tpgt_dir, "attrib") + _write_attr(os.path.join(attrib, "authentication"), "0") + _write_attr(os.path.join(attrib, "generate_node_acls"), "1") + _write_attr(os.path.join(attrib, "cache_dynamic_acls"), "1") + _write_attr(os.path.join(attrib, "demo_mode_write_protect"), "0") + _write_attr(os.path.join(tpgt_dir, "enable"), "1") + + +def _teardown_lio_target(iqn: str, backstore: str) -> None: + tgt_dir = os.path.join(CONFIGFS_TARGET, "iscsi", iqn) + tpgt_dir = os.path.join(tgt_dir, "tpgt_1") + enable = os.path.join(tpgt_dir, "enable") + if os.path.isfile(enable): + with contextlib.suppress(Exception): + _write_attr(enable, "0") + np_dir = os.path.join(tpgt_dir, "np") + for portal in _listdir(np_dir): + _rmdir(os.path.join(np_dir, portal)) + lun0 = os.path.join(tpgt_dir, "lun", "lun_0") + link_path = os.path.join(lun0, backstore) + if os.path.lexists(link_path): + _unlink(link_path) + _rmdir(lun0) + _rmdir(os.path.join(tpgt_dir, "lun")) + _rmdir(tpgt_dir) + _rmdir(tgt_dir) + + bs_dir = os.path.join(CONFIGFS_TARGET, "core", LIO_HBA, backstore) + bs_enable = os.path.join(bs_dir, "enable") + if os.path.isfile(bs_enable): + with contextlib.suppress(Exception): + _write_attr(bs_enable, "0") + _rmdir(bs_dir) + + +def _block_by_id() -> set[str]: + by_id = "/dev/disk/by-id" + if not os.path.isdir(by_id): + return set() + return {os.path.join(by_id, name) for name in os.listdir(by_id)} + + +def _iscsi_login(iqn: str, portal: str, port: int) -> None: + target = f"{portal}:{port}" + _run( + ["iscsiadm", "-m", "discovery", "-t", "sendtargets", "-p", target], + privileged=True, + ) + result = _run( + ["iscsiadm", "-m", "node", "-T", iqn, "-p", target, "--login"], + check=False, + privileged=True, + ) + if ( + result.returncode != 0 + and "already" not in (result.stderr + result.stdout).lower() + ): + raise RuntimeError( + f"iscsiadm login failed: {result.stderr.strip() or result.stdout.strip()}" + ) + + +def _iscsi_logout(iqn: str, portal: str, port: int) -> None: + target = f"{portal}:{port}" + _run( + ["iscsiadm", "-m", "node", "-T", iqn, "-p", target, "--logout"], + check=False, + privileged=True, + ) + _run( + ["iscsiadm", "-m", "node", "-T", iqn, "-p", target, "-o", "delete"], + check=False, + privileged=True, + ) + + +def _wait_new_wwn_dev(before: set[str], timeout_s: float = DEVICE_WAIT_S) -> str: + deadline = time.monotonic() + timeout_s + while time.monotonic() < deadline: + after = _block_by_id() + new_wwn = sorted( + path + for path in after - before + if os.path.basename(path).startswith("wwn-") + and not os.path.basename(path).endswith("-part1") + and "-part" not in os.path.basename(path) + ) + if new_wwn: + return os.path.realpath(new_wwn[0]) + time.sleep(DEVICE_POLL_S) + raise RuntimeError("local iSCSI LUN did not appear under /dev/disk/by-id") + + +def _naa_from_dev(dev_path: str) -> str: + name = os.path.basename(os.path.realpath(dev_path)) + wwid_path = f"/sys/block/{name}/device/wwid" + wwid = "" + if os.path.isfile(wwid_path): + wwid = _read_attr(wwid_path).lower() + if not wwid: + by_id = "/dev/disk/by-id" + real = os.path.realpath(dev_path) + for entry in _listdir(by_id): + path = os.path.join(by_id, entry) + if os.path.realpath(path) == real and entry.startswith("wwn-0x"): + wwid = "naa." + entry[len("wwn-0x") :] + break + if wwid.startswith("naa."): + return wwid + if wwid.startswith("wwn-0x"): + return "naa." + wwid[6:] + if wwid.startswith("0x"): + return "naa." + wwid[2:] + digits = "".join(ch for ch in wwid if ch in "0123456789abcdef") + if len(digits) >= 16: + return "naa." + digits + raise RuntimeError(f"could not read NAA for {dev_path}: wwid={wwid!r}") + + +def _ensure_iscsi_port(portal: str, port: int) -> bool: + result = _run( + [ + "iptables", + "-C", + "INPUT", + "-p", + "tcp", + "-d", + portal, + "--dport", + str(port), + "-j", + "ACCEPT", + ], + check=False, + privileged=True, + ) + if result.returncode == 0: + return False + _run( + [ + "iptables", + "-I", + "INPUT", + "1", + "-p", + "tcp", + "-d", + portal, + "--dport", + str(port), + "-j", + "ACCEPT", + ], + check=False, + privileged=True, + ) + return True + + +def _drop_iscsi_port(portal: str, port: int) -> None: + _run( + [ + "iptables", + "-D", + "INPUT", + "-p", + "tcp", + "-d", + portal, + "--dport", + str(port), + "-j", + "ACCEPT", + ], + check=False, + privileged=True, + ) + + +def _state_path(lab_id: str) -> str: + return os.path.join(STATE_DIR, f"ovdl-iscsi-{lab_id}.json") + + +def _write_state(lab: IscsiSanLab) -> None: + payload = { + "lab_id": lab.lab_id, + "portal": lab.portal, + "port": lab.port, + "iqn": lab.iqn, + "img_path": lab.img_path, + "loop_dev": lab.loop_dev, + "naa": lab.naa, + "local_dev": lab.local_dev, + "datastore_name": lab.datastore_name, + "host_moref": lab.host_moref, + "backstore_name": lab.backstore_name, + "iptables_added": lab.iptables_added, + "enabled_software_iscsi": lab.enabled_software_iscsi, + "bound_vnic": lab.bound_vnic, + } + with open(_state_path(lab.lab_id), "w", encoding="utf-8") as handle: + json.dump(payload, handle) + + +def _read_state(path: str) -> IscsiSanLab | None: + try: + with open(path, encoding="utf-8") as handle: + data = json.load(handle) + except (OSError, json.JSONDecodeError): + return None + try: + return IscsiSanLab( + lab_id=str(data["lab_id"]), + portal=str(data["portal"]), + port=int(data["port"]), + iqn=str(data["iqn"]), + img_path=str(data["img_path"]), + loop_dev=str(data["loop_dev"]), + naa=str(data["naa"]), + local_dev=str(data.get("local_dev", "")), + datastore_name=str(data["datastore_name"]), + host_moref=str(data.get("host_moref", "")), + backstore_name=str(data["backstore_name"]), + iptables_added=bool(data.get("iptables_added", False)), + enabled_software_iscsi=bool(data.get("enabled_software_iscsi", False)), + bound_vnic=str(data.get("bound_vnic", "")), + ) + except (KeyError, TypeError, ValueError): + return None + + +def _software_iscsi_enabled(host: vim.HostSystem) -> bool: + return bool( + getattr(host.config.storageDevice, "softwareInternetScsiEnabled", False) + ) + + +def _ensure_software_iscsi(host: vim.HostSystem) -> bool: + """Enable the software iSCSI adapter when it is missing. Return True if we enabled it.""" + if _software_iscsi_enabled(host) or _find_software_iscsi_hba_or_none(host): + return False + LOG.info("enabling software iSCSI on %s", host.name) + host.configManager.storageSystem.UpdateSoftwareInternetScsiEnabled(True) + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + fresh = vim.HostSystem(host._moId, host._stub) + if _find_software_iscsi_hba_or_none(fresh): + return True + time.sleep(1) + raise RuntimeError(f"software iSCSI adapter did not appear on {host.name}") + + +def _find_software_iscsi_hba_or_none( + host: vim.HostSystem, +) -> vim.host.InternetScsiHba | None: + for hba in host.config.storageDevice.hostBusAdapter: + if not isinstance(hba, vim.host.InternetScsiHba): + continue + driver = (hba.driver or "").lower() + model = (hba.model or "").lower() + if driver == "iscsi_vmk" or "software" in model: + return hba + return None + + +def _bind_management_vmk(host: vim.HostSystem, hba: vim.host.InternetScsiHba) -> str: + """Bind vmk0 when the software adapter has no VMkernel NIC.""" + storage = host.configManager.storageSystem + try: + bound = storage.QueryBoundVnics(iScsiHbaDevice=hba.device) or [] + except Exception: + bound = [] + if bound: + return "" + vmk = None + for vnic in host.config.network.vnic or []: + if vnic.device == "vmk0": + vmk = vnic.device + break + if vmk is None and host.config.network.vnic: + vmk = host.config.network.vnic[0].device + if not vmk: + return "" + LOG.info("binding %s to %s on %s", vmk, hba.device, host.name) + try: + storage.BindVnic(iScsiHbaDevice=hba.device, vnicDevice=vmk) + except Exception as exc: + LOG.warning("BindVnic %s failed: %s", vmk, exc) + return "" + return vmk + + +def _find_software_iscsi_hba(host: vim.HostSystem) -> vim.host.InternetScsiHba: + hba = _find_software_iscsi_hba_or_none(host) + if hba is None: + raise RuntimeError( + f"host {host.name} has no software iSCSI adapter; " + "enable vmhba iscsi_vmk before running SAN tests" + ) + return hba + + +def _host_from_datastore(datastore: vim.Datastore) -> vim.HostSystem: + if not datastore.host: + raise RuntimeError(f"datastore {datastore.name!r} is not mounted on any host") + return datastore.host[0].key + + +def _refresh_host(si: vim.ServiceInstance, moref: str) -> vim.HostSystem: + return vim.HostSystem(moref, si._stub) + + +def _find_scsi_disk(host: vim.HostSystem, naa: str) -> vim.host.ScsiDisk | None: + want = naa.lower() + if not want.startswith("naa."): + want = "naa." + want + bare = want[4:] + for lun in host.config.storageDevice.scsiLun or []: + if not isinstance(lun, vim.host.ScsiDisk): + continue + canonical = (lun.canonicalName or "").lower() + if canonical == want or canonical.replace("naa.", "") == bare: + return lun + uuid = (lun.uuid or "").lower().replace("-", "") + if bare in uuid: + return lun + return None + + +def _add_send_target( + host: vim.HostSystem, hba: vim.host.InternetScsiHba, portal: str, port: int +) -> None: + for existing in hba.configuredSendTarget or []: + if existing.address == portal and int(existing.port or ISCSI_PORT) == port: + return + target = vim.host.InternetScsiHba.SendTarget() + target.address = portal + target.port = port + host.configManager.storageSystem.AddInternetScsiSendTargets( + iScsiHbaDevice=hba.device, targets=[target] + ) + + +def _remove_send_target(host: vim.HostSystem, portal: str, port: int) -> None: + try: + hba = _find_software_iscsi_hba(host) + except RuntimeError: + return + match = None + for existing in hba.configuredSendTarget or []: + if existing.address == portal and int(existing.port or ISCSI_PORT) == port: + match = existing + break + if match is None: + return + with contextlib.suppress(Exception): + host.configManager.storageSystem.RemoveInternetScsiSendTargets( + iScsiHbaDevice=hba.device, targets=[match] + ) + + +def _wait_for_esxi_disk( + si: vim.ServiceInstance, host_moref: str, naa: str +) -> vim.host.ScsiDisk: + deadline = time.monotonic() + HBA_WAIT_S + last_names: list[str] = [] + while time.monotonic() < deadline: + host = _refresh_host(si, host_moref) + with contextlib.suppress(Exception): + host.configManager.storageSystem.RescanHba( + hbaDevice=_find_software_iscsi_hba(host).device + ) + with contextlib.suppress(Exception): + host.configManager.storageSystem.RescanVmfs() + host = _refresh_host(si, host_moref) + disk = _find_scsi_disk(host, naa) + if disk is not None: + return disk + last_names = [ + str(lun.canonicalName) + for lun in host.config.storageDevice.scsiLun or [] + if isinstance(lun, vim.host.ScsiDisk) + ] + time.sleep(HBA_POLL_S) + raise RuntimeError( + f"ESXi host did not discover iSCSI LUN {naa}; scsi disks={last_names[:12]}" + ) + + +def _create_vmfs_datastore( + host: vim.HostSystem, disk: vim.host.ScsiDisk, name: str +) -> vim.Datastore: + ds_sys = host.configManager.datastoreSystem + options = ds_sys.QueryVmfsDatastoreCreateOptions(devicePath=disk.devicePath) + if not options: + options = ds_sys.QueryVmfsDatastoreCreateOptions( + devicePath=disk.devicePath, vmfsMajorVersion=6 + ) + if not options: + raise RuntimeError( + f"QueryVmfsDatastoreCreateOptions returned no layout for {disk.canonicalName}" + ) + spec = options[0].spec + spec.vmfs.volumeName = name + if getattr(spec.vmfs, "majorVersion", None) in (None, 0): + spec.vmfs.majorVersion = 6 + return ds_sys.CreateVmfsDatastore(spec=spec) + + +def _destroy_vms_on_datastore( + si: vim.ServiceInstance, datacenter: vim.Datacenter, name: str +) -> None: + content = si.RetrieveContent() + container = content.viewManager.CreateContainerView( + datacenter, [vim.VirtualMachine], True + ) + try: + vms = list(container.view) + finally: + container.Destroy() + for vm in vms: + try: + datastores = [ds.name for ds in (vm.datastore or [])] + except Exception: + LOG.debug("skipping VM while listing datastores", exc_info=True) + continue + if name not in datastores: + continue + LOG.info("destroying leftover SAN lab VM %s on %s", vm.name, name) + try: + if vm.runtime.powerState == vim.VirtualMachinePowerState.poweredOn: + _wait_for_task(vm.PowerOffVM_Task()) + _wait_for_task(vm.Destroy_Task()) + except Exception: + LOG.exception("failed to destroy leftover VM %s", vm.name) + + +def _remove_vmfs_datastore( + si: vim.ServiceInstance, datacenter: vim.Datacenter, name: str +) -> None: + _destroy_vms_on_datastore(si, datacenter, name) + try: + datastore = _find_datastore(datacenter, name) + except RuntimeError: + return + hosts = [mount.key for mount in datastore.host or []] + for host in hosts: + with contextlib.suppress(Exception): + host.configManager.datastoreSystem.RemoveDatastore(datastore) + return + with contextlib.suppress(Exception): + _wait_for_task(datastore.Destroy_Task()) + + +def _connect_lab_vim() -> tuple[vim.ServiceInstance, dict[str, Any]]: + cfg = _load_test_config() + thumbprint = nfc_auth.get_ssl_cert_thumbprint(cfg["host"], cfg["port"]) + si = _connect_vim( + cfg["host"], + cfg["username"], + cfg["password"], + cfg["port"], + thumbprint, + cfg["allow_untrusted"], + ) + return si, cfg + + +def cleanup_leftover_iscsi_labs( + si: vim.ServiceInstance | None = None, + datacenter: vim.Datacenter | None = None, +) -> None: + """Tear down leftover ovdl-iscsi LIO targets, files, and VMFS volumes.""" + for name in _listdir(STATE_DIR): + if not name.startswith("ovdl-iscsi-") or not name.endswith(".json"): + continue + lab = _read_state(os.path.join(STATE_DIR, name)) + if lab is None: + continue + LOG.warning("cleaning leftover iSCSI SAN lab %s", lab.lab_id) + if si is not None and datacenter is not None: + with contextlib.suppress(Exception): + _remove_vmfs_datastore(si, datacenter, lab.datastore_name) + if lab.host_moref: + host = _refresh_host(si, lab.host_moref) + with contextlib.suppress(Exception): + _remove_send_target(host, lab.portal, lab.port) + with contextlib.suppress(Exception): + _restore_esxi_iscsi(host, lab) + _teardown_local(lab) + + +def _require_privileged() -> None: + result = _run(["true"], check=False, privileged=True) + if result.returncode != 0: + raise RuntimeError( + "iSCSI SAN lab needs root or passwordless sudo for LIO, " + f"losetup, and iscsiadm: {result.stderr.strip()}" + ) + + +def _chmod_dev(path: str, mode: str = "0666") -> None: + _run(["chmod", mode, path], privileged=True, check=False) + + +def _teardown_local(lab: IscsiSanLab) -> None: + _iscsi_logout(lab.iqn, lab.portal, lab.port) + _teardown_lio_target(lab.iqn, lab.backstore_name) + if lab.loop_dev: + _run(["losetup", "-d", lab.loop_dev], check=False, privileged=True) + if lab.img_path: + with contextlib.suppress(OSError): + os.unlink(lab.img_path) + if lab.iptables_added: + _drop_iscsi_port(lab.portal, lab.port) + with contextlib.suppress(OSError): + os.unlink(_state_path(lab.lab_id)) + + +def setup_iscsi_san_lab() -> IscsiSanLab: + """Create a loop-backed LIO LUN, attach it to lab ESXi, and format VMFS. + + Raises: + RuntimeError: when the runner or ESXi cannot host the LUN. Callers + should skip the SAN tests. + """ + _require_privileged() + + san_cfg = load_iscsi_san_config() or {} + portal = san_cfg.get("portal") or default_portal_ip() + lab_id = uuid.uuid4().hex[:12] + iqn = f"{IQN_PREFIX}ovdl-{lab_id}" + backstore = f"ovdl_{lab_id}" + img_path = os.path.join(STATE_DIR, f"ovdl-iscsi-{lab_id}.img") + datastore_name = f"{DATASTORE_PREFIX}{lab_id}" + loop_dev = "" + iptables_added = False + local_dev = "" + naa = "" + si: vim.ServiceInstance | None = None + lab = IscsiSanLab( + lab_id=lab_id, + portal=portal, + port=ISCSI_PORT, + iqn=iqn, + img_path=img_path, + loop_dev=loop_dev, + naa=naa, + local_dev=local_dev, + datastore_name=datastore_name, + host_moref="", + backstore_name=backstore, + iptables_added=False, + ) + try: + si, cfg = _connect_lab_vim() + content = si.RetrieveContent() + datacenter = _find_datacenter(content, cfg["datacenter"]) + cleanup_leftover_iscsi_labs(si, datacenter) + lab_ds = _find_datastore(datacenter, cfg["datastore"]) + host = _host_from_datastore(lab_ds) + lab.host_moref = host._moId + lab.enabled_software_iscsi = _ensure_software_iscsi(host) + host = _refresh_host(si, lab.host_moref) + hba = _find_software_iscsi_hba(host) + lab.bound_vnic = _bind_management_vmk(host, hba) + host = _refresh_host(si, lab.host_moref) + hba = _find_software_iscsi_hba(host) + + _create_sparse_file(img_path, LUN_SIZE_BYTES) + loop_dev = _attach_loop(img_path) + lab.loop_dev = loop_dev + _setup_lio_target(lab_id, loop_dev, portal, ISCSI_PORT, iqn, backstore) + iptables_added = _ensure_iscsi_port(portal, ISCSI_PORT) + lab.iptables_added = iptables_added + + before = _block_by_id() + _iscsi_login(iqn, portal, ISCSI_PORT) + local_dev = _wait_new_wwn_dev(before) + _chmod_dev(local_dev) + naa = _naa_from_dev(local_dev) + lab.local_dev = local_dev + lab.naa = naa + _write_state(lab) + + host = _refresh_host(si, lab.host_moref) + hba = _find_software_iscsi_hba(host) + _add_send_target(host, hba, portal, ISCSI_PORT) + disk = _wait_for_esxi_disk(si, lab.host_moref, naa) + host = _refresh_host(si, lab.host_moref) + _create_vmfs_datastore(host, disk, datastore_name) + _write_state(lab) + LOG.info( + "iSCSI SAN lab %s portal=%s:%s naa=%s datastore=%s local=%s", + lab_id, + portal, + ISCSI_PORT, + naa, + datastore_name, + local_dev, + ) + return lab + except Exception: + if si is not None: + with contextlib.suppress(Exception): + content = si.RetrieveContent() + cfg = _load_test_config() + datacenter = _find_datacenter(content, cfg["datacenter"]) + _remove_vmfs_datastore(si, datacenter, datastore_name) + if lab.host_moref: + with contextlib.suppress(Exception): + _remove_send_target( + _refresh_host(si, lab.host_moref), portal, ISCSI_PORT + ) + with contextlib.suppress(Exception): + _restore_esxi_iscsi(_refresh_host(si, lab.host_moref), lab) + _teardown_local(lab) + raise + finally: + if si is not None: + Disconnect(si) + + +def _restore_esxi_iscsi(host: vim.HostSystem, lab: IscsiSanLab) -> None: + storage = host.configManager.storageSystem + if lab.bound_vnic: + with contextlib.suppress(Exception): + hba = _find_software_iscsi_hba(host) + storage.UnbindVnic( + iScsiHbaDevice=hba.device, vnicDevice=lab.bound_vnic, force=True + ) + if lab.enabled_software_iscsi: + with contextlib.suppress(Exception): + storage.UpdateSoftwareInternetScsiEnabled(False) + + +def teardown_iscsi_san_lab(lab: IscsiSanLab) -> None: + """Remove the VMFS datastore, ESXi send target, and local LIO LUN.""" + si: vim.ServiceInstance | None = None + try: + si, cfg = _connect_lab_vim() + content = si.RetrieveContent() + datacenter = _find_datacenter(content, cfg["datacenter"]) + _remove_vmfs_datastore(si, datacenter, lab.datastore_name) + if lab.host_moref: + host = _refresh_host(si, lab.host_moref) + _remove_send_target(host, lab.portal, lab.port) + with contextlib.suppress(Exception): + hba = _find_software_iscsi_hba(host) + host.configManager.storageSystem.RescanHba(hbaDevice=hba.device) + _restore_esxi_iscsi(host, lab) + except Exception: + LOG.exception("ESXi teardown for iSCSI SAN lab %s failed", lab.lab_id) + finally: + if si is not None: + Disconnect(si) + _teardown_local(lab) + + +@contextlib.contextmanager +def iscsi_san_lab() -> Iterator[IscsiSanLab]: + """Context manager around :func:`setup_iscsi_san_lab`.""" + lab = setup_iscsi_san_lab() + try: + yield lab + finally: + teardown_iscsi_san_lab(lab) diff --git a/tests/integration/test_openvixdisklib.py b/tests/integration/test_openvixdisklib.py index 6c853d6..de5111d 100644 --- a/tests/integration/test_openvixdisklib.py +++ b/tests/integration/test_openvixdisklib.py @@ -81,7 +81,9 @@ def test_write_and_read_sector_zero_and_one_gib( 0: pattern_bytes(SECTOR_SIZE, b"OVDL-S0"), SECTOR_AT_1GB: pattern_bytes(SECTOR_SIZE, b"OVDL-1GB"), } - assert handle.get_transport_modes() == ["nbdssl", "nbd"] + modes = handle.get_transport_modes() + assert modes[0:2] == ["nbdssl", "nbd"] + assert set(modes) <= {"nbdssl", "nbd", "san", "hotadd"} with ( handle.connect(**connect_kwargs) as conn, handle.open(conn, lab.disk_path, flags=open_flags) as disk, diff --git a/tests/integration/test_san.py b/tests/integration/test_san.py new file mode 100644 index 0000000..c2c22a6 --- /dev/null +++ b/tests/integration/test_san.py @@ -0,0 +1,119 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""SAN transport against a file-backed iSCSI VMFS LUN. + +Skipped when the runner cannot present a LIO target or ESXi cannot +mount it. The session-wide ``lab`` VM stays on the YAML datastore. +""" + +from __future__ import annotations + +from collections.abc import Iterator + +import pytest + +from openvixdisklib import openvixdisklib as vixdisklib +from tests.integration.base import ( + SECTOR_AT_1GB, + SECTOR_SIZE, + LabEnv, + create_lab_vm, + destroy_lab_vm, + pattern_bytes, +) +from tests.integration.iscsi_lab import ( + IscsiSanLab, + setup_iscsi_san_lab, + teardown_iscsi_san_lab, +) + + +@pytest.fixture(scope="module") +def iscsi_lab() -> Iterator[IscsiSanLab]: + """Bring up a loop-backed iSCSI VMFS datastore, or skip.""" + try: + lab = setup_iscsi_san_lab() + except (RuntimeError, TimeoutError, OSError) as exc: + pytest.skip(f"iSCSI SAN lab unavailable: {exc}") + try: + yield lab + finally: + teardown_iscsi_san_lab(lab) + + +@pytest.fixture +def san_lab(iscsi_lab: IscsiSanLab) -> Iterator[LabEnv]: + """Powered-off 10 GiB VM on the iSCSI datastore (not the session lab).""" + env = create_lab_vm(datastore=iscsi_lab.datastore_name, thin_provisioned=False) + try: + yield env + finally: + destroy_lab_vm(env) + + +class TestSanTransport: + def test_write_and_read_sector_zero_and_one_gib(self, san_lab: LabEnv) -> None: + """SAN write/read of sector 0 and 1 GiB; nbdssl write is visible to SAN. + + ESXi keeps a VMFS cache of the mounted datastore, so nbdssl does not + observe SAN writes from this initiator. The reverse does: nbdssl + writes hit the LUN, SAN open flushes the initiator cache, then reads. + """ + handle = vixdisklib.VixDiskLibHandle( + vixdisklib_compatibility_version="8.0", config_path=None + ) + modes = handle.get_transport_modes() + if "san" not in modes: + pytest.skip(f"san not advertised: {modes}") + write_buf = vixdisklib.get_buffer(SECTOR_SIZE) + read_buf = vixdisklib.get_buffer(SECTOR_SIZE) + patterns = { + 0: pattern_bytes(SECTOR_SIZE, b"SAN-S0"), + SECTOR_AT_1GB: pattern_bytes(SECTOR_SIZE, b"SAN-1GB"), + } + connect_kwargs = san_lab.vixdisklib_connect_kwargs( + { + "allow_untrusted": san_lab.allow_untrusted, + "transport_modes": "san", + } + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, san_lab.disk_path, flags=0) as disk, + ): + assert handle.get_transport_mode(disk) == "san" + for start, expected in patterns.items(): + write_buf[:SECTOR_SIZE] = expected + handle.write(disk, start, 1, write_buf) + read_buf[:SECTOR_SIZE] = b"\xa5" * SECTOR_SIZE + handle.read(disk, start, 1, read_buf) + assert read_buf.raw[:SECTOR_SIZE] == expected + + nbd_patterns = { + 0: pattern_bytes(SECTOR_SIZE, b"NBD-S0"), + SECTOR_AT_1GB: pattern_bytes(SECTOR_SIZE, b"NBD-1GB"), + } + nbd_kwargs = san_lab.vixdisklib_connect_kwargs( + { + "allow_untrusted": san_lab.allow_untrusted, + "transport_modes": "nbdssl", + } + ) + with ( + handle.connect(**nbd_kwargs) as conn, + handle.open(conn, san_lab.disk_path, flags=0) as disk, + ): + assert handle.get_transport_mode(disk) == "nbdssl" + for start, expected in nbd_patterns.items(): + write_buf[:SECTOR_SIZE] = expected + handle.write(disk, start, 1, write_buf) + + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, san_lab.disk_path, flags=0) as disk, + ): + for start, expected in nbd_patterns.items(): + read_buf[:SECTOR_SIZE] = b"\xa5" * SECTOR_SIZE + handle.read(disk, start, 1, read_buf) + assert read_buf.raw[:SECTOR_SIZE] == expected diff --git a/tests/unit/test_hotadd.py b/tests/unit/test_hotadd.py index 1636335..ff7c0bb 100644 --- a/tests/unit/test_hotadd.py +++ b/tests/unit/test_hotadd.py @@ -240,27 +240,29 @@ def test_read_write_sectors(self, tmp_path) -> None: class TestSelectTransport: + @mock.patch("openvixdisklib.openvixdisklib.san.is_available", return_value=False) @mock.patch( "openvixdisklib.openvixdisklib.hotadd.is_vmware_guest", return_value=False ) def test_colon_list_skips_hotadd_on_bare_metal( - self, mock_guest: mock.MagicMock + self, mock_guest: mock.MagicMock, mock_san: mock.MagicMock ) -> None: - """Bare metal skips hotadd and uses the next usable mode.""" - del mock_guest + """Bare metal without SAN skips hotadd and uses the next usable mode.""" + del mock_guest, mock_san assert _select_transport("file:san:hotadd:nbdssl:nbd") == "nbdssl" assert _available_transports() == ["nbdssl", "nbd"] with pytest.raises(NotImplementedError, match="hotadd"): _select_transport("hotadd") + @mock.patch("openvixdisklib.openvixdisklib.san.is_available", return_value=False) @mock.patch( "openvixdisklib.openvixdisklib.hotadd.is_vmware_guest", return_value=True ) def test_colon_list_selects_hotadd_in_guest( - self, mock_guest: mock.MagicMock + self, mock_guest: mock.MagicMock, mock_san: mock.MagicMock ) -> None: """A VMware guest uses hotadd when it is first in the colon list.""" - del mock_guest + del mock_guest, mock_san assert _select_transport("file:san:hotadd:nbdssl") == "hotadd" assert _available_transports() == ["nbdssl", "nbd", "hotadd"] diff --git a/tests/unit/test_san.py b/tests/unit/test_san.py new file mode 100644 index 0000000..fbc4f5b --- /dev/null +++ b/tests/unit/test_san.py @@ -0,0 +1,168 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Unit tests for SAN LUN matching, VMDK naming, and extent-map I/O.""" + +from __future__ import annotations + +import os +from pathlib import Path +from unittest import mock + +import pytest + +from openvixdisklib.openvixdisklib import ( + VIXDISKLIB_FLAG_OPEN_COMPRESSION_FASTLZ, + VixDiskLibHandle, + _available_transports, + _select_transport, +) +from openvixdisklib.san import ( + FILE_BLOCK_SIZE, + SECTOR_SIZE, + Extent, + SanDisk, + flat_extent_name, + lfb_volume_offset, + map_file_range, + normalize_naa, + parse_datastore_path, + parse_lfb, + parse_sfb, + parse_vmdk_descriptor, + sfb_volume_offset, +) + +SECTOR_AT_1GB = (1024 * 1024 * 1024) // SECTOR_SIZE + + +def test_normalize_naa_from_wwn() -> None: + """WWN and SCSI-3 names collapse to ``naa.``.""" + assert normalize_naa("wwn-0x6001405abc") == "naa.6001405abc" + assert normalize_naa("scsi-36001405abc") == "naa.6001405abc" + assert normalize_naa("NAA.6001405ABC") == "naa.6001405abc" + + +def test_parse_datastore_path() -> None: + """Split a datastore path into name and relative path.""" + assert parse_datastore_path("[ds0] vm/vm.vmdk") == ("ds0", "vm/vm.vmdk") + with pytest.raises(ValueError, match="unsupported disk path"): + parse_datastore_path("vm/vm.vmdk") + + +def test_flat_extent_name() -> None: + """Descriptor basename maps to the sibling ``-flat.vmdk``.""" + assert flat_extent_name("vm/disk.vmdk") == "disk-flat.vmdk" + assert flat_extent_name("disk-flat.vmdk") == "disk-flat.vmdk" + + +def test_parse_vmdk_descriptor() -> None: + """Read capacity and the VMFS extent name from a descriptor.""" + text = ( + '# Disk DescriptorFile\ncreateType="vmfs"\nRW 20971520 VMFS "disk-flat.vmdk"\n' + ) + sectors, name = parse_vmdk_descriptor(text) + assert sectors == 20971520 + assert name == "disk-flat.vmdk" + + +def test_file_block_base_after_lvm_and_heartbeats() -> None: + """SFB 0 starts after the GPT partition, LVM label, and 16 heartbeats.""" + from openvixdisklib.san import GptPartition, _file_block_base + + part = GptPartition(start_lba=2048, end_lba=1000) + assert _file_block_base(part, FILE_BLOCK_SIZE) == 18 * FILE_BLOCK_SIZE + + +def test_sfb_and_lfb_offsets() -> None: + """VMFS6 SFB/LFB addresses decode to volume offsets.""" + cluster, resource = 33, 210 + addr = 0x1 | (cluster << 15) | (resource << 51) + assert parse_sfb(addr) == (cluster, resource) + assert sfb_volume_offset(addr, 512, 20) == ((33 * 512) + 210) << 20 + lfb = 0x7 | (4 << 15) + assert parse_lfb(lfb) == 4 + assert lfb_volume_offset(lfb, 20, 9) == 4 << (20 + 9) + + +def test_map_file_range_clips_extents() -> None: + """Only the overlapping part of an extent is returned.""" + extents = [ + Extent(0, 2 * FILE_BLOCK_SIZE, FILE_BLOCK_SIZE), + Extent(FILE_BLOCK_SIZE, 5 * FILE_BLOCK_SIZE, FILE_BLOCK_SIZE), + ] + hits = map_file_range(extents, 512, 1024) + assert hits == [Extent(512, 2 * FILE_BLOCK_SIZE + 512, 1024)] + + +def test_sandisk_pread_pwrite(tmp_path: Path) -> None: + """SanDisk I/O uses the extent map; holes read as zeros.""" + lun_path = tmp_path / "lun.bin" + lun_path.write_bytes(b"\x00" * (4 * FILE_BLOCK_SIZE)) + fd = os.open(lun_path, os.O_RDWR) + extents = [Extent(0, 2 * FILE_BLOCK_SIZE, FILE_BLOCK_SIZE)] + disk = SanDisk(fd, str(lun_path), extents, 10 * 1024 * 1024 * 1024) + pattern = b"A" * SECTOR_SIZE + disk.write(0, 1, pattern) + buf = bytearray(SECTOR_SIZE) + result = disk.readinto(0, 1, buf, skip_decompression=True) + assert bytes(buf) == pattern + assert result.uncompressed_length == SECTOR_SIZE + assert result.fragments == () + hole = bytearray(b"\xa5" * SECTOR_SIZE) + disk.readinto(SECTOR_AT_1GB, 1, hole) + assert bytes(hole) == b"\x00" * SECTOR_SIZE + with pytest.raises(RuntimeError, match="allocated VMFS"): + disk.write(SECTOR_AT_1GB, 1, b"B" * SECTOR_SIZE) + disk.close() + disk.close() + + +class TestSelectTransportSan: + @mock.patch( + "openvixdisklib.openvixdisklib.hotadd.is_vmware_guest", return_value=False + ) + @mock.patch("openvixdisklib.openvixdisklib.san.is_available", return_value=True) + def test_colon_list_selects_san( + self, mock_san: mock.MagicMock, mock_guest: mock.MagicMock + ) -> None: + """A host with SCSI disks uses san when it is first in the colon list.""" + del mock_san, mock_guest + assert _select_transport("file:san:hotadd:nbdssl:nbd") == "san" + assert _available_transports() == ["nbdssl", "nbd", "san"] + + @mock.patch( + "openvixdisklib.openvixdisklib.hotadd.is_vmware_guest", return_value=False + ) + @mock.patch("openvixdisklib.openvixdisklib.san.is_available", return_value=False) + def test_san_alone_raises_when_unavailable( + self, mock_san: mock.MagicMock, mock_guest: mock.MagicMock + ) -> None: + """``san`` alone raises when the host has no SCSI sysfs.""" + del mock_san, mock_guest + with pytest.raises(NotImplementedError, match="san"): + _select_transport("san") + + def test_default_is_nbdssl(self) -> None: + """None still defaults to nbdssl when SAN is available.""" + assert _select_transport(None) == "nbdssl" + + +@mock.patch("openvixdisklib.openvixdisklib.vim.VirtualMachine") +def test_fastlz_open_flag_rejected_for_san(mock_vm: mock.MagicMock) -> None: + """FastLZ open flags are rejected the same way as HotAdd.""" + del mock_vm + handle = VixDiskLibHandle() + conn = mock.Mock() + conn.transport_mode = "san" + conn.read_only = False + conn.vm_moref = "vm-1" + conn.si = mock.Mock() + conn.snapshot_ref = None + with ( + pytest.raises(NotImplementedError, match="san"), + handle.open( + conn, "[ds] vm/vm.vmdk", flags=VIXDISKLIB_FLAG_OPEN_COMPRESSION_FASTLZ + ), + ): + pass