diff --git a/.gitignore b/.gitignore index 7a6162c..1d9a8c6 100644 --- a/.gitignore +++ b/.gitignore @@ -50,3 +50,7 @@ coverage.html # Lab credentials and VM/snapshot names for integration tests. .test_config.yaml + +# Lab setup scripts/configs with hardcoded lab credentials (vCenter, ESXi). +# See docs/encryption_lab_setup.md for the sanitized procedure. +.lab diff --git a/README.md b/README.md index 4866ece..7b301c0 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,8 @@ from VDDK 8 NBD traffic; see `docs/`. ## Status -Supported and tested on **vCenter 8 / ESXi 8** (lab: 8.0.1). The +Supported and tested on **vCenter 8 / ESXi 8** (lab: 8.0.1), including a +standalone ESXi host with no vCenter. The VixDiskLib compatibility mode is `8.0` only. VIM login requests pyVmomi's vim25 **8.x** versions, so a newer host such as vSphere 9 stays on 8.x SOAP instead of 9.x types. @@ -27,14 +28,32 @@ moment. Default transport is `nbdssl` (`nbd` is still available): -- `VixDiskLib_ConnectEx` (UID credentials) +- `VixDiskLib_ConnectEx` (UID credentials; vCenter or direct ESXi) - `VixDiskLib_Open` (datastore path, read-only or read-write) - `VixDiskLib_Read` (optional ``skip_decompression`` packs FastLZ extras) - `VixDiskLib_Write` - -Not implemented: compression open flags other than FastLZ, CBT / -allocated-block queries, disk geometry (`DDB_GET`), encrypted disks, -and direct ESXi `ha-nfc` without vCenter `vpxa-nfc`. +- `VixDiskLib_GetInfo` (capacity and physical geometry from the `Open` + reply; `biosGeo`/`adapterType`/`uuid` from `DDB_GET`, matching real + VDDK's cost and behavior) +- `VixDiskLib_QueryAllocatedBlocks` (allocated-block bitmap; see + `docs/nfc_read.md`) +- Changed Block Tracking: `openvixdisklib.nfc_auth.enable_change_tracking` + / `disk_change_id` / `query_changed_disk_areas` (public VIM API, not + part of VixDiskLib itself; see `docs/cbt.md`) +- Encrypted VM disks — read/write transparently with no + encryption-specific code (see `docs/encryption.md`) + +Reading/writing a snapshot delta file directly (and running +`query_allocated_blocks` against it) already works — `NFC_DELTA_DISK` +turned out to be an optional VMFS-only VDDK client optimization, not a +correctness requirement (see `docs/reverse_engineering_procedure.md`). + +Reading/writing an encrypted VM disk also already works, with no +encryption-specific code — ESXi handles it transparently below NFC +whenever the serving host already holds the disk's key (see +`docs/encryption.md`). + +Not implemented: compression open flags other than FastLZ. Requires Python 3.10 or later. diff --git a/docs/cbt.md b/docs/cbt.md new file mode 100644 index 0000000..64b1a23 --- /dev/null +++ b/docs/cbt.md @@ -0,0 +1,110 @@ +# Changed Block Tracking (CBT) + +This is not a reverse-engineered NFC feature. VixDiskLib does not expose +CBT itself: `VixDiskLib_QueryAllocatedBlocks` (implemented separately; +see `docs/nfc_read.md`) reports which blocks are *allocated* +(non-sparse) within a single NFC-opened disk, not which byte ranges +*changed* between two points in time. Real backup tools +get changed-range information from vSphere's public +`VirtualMachine.QueryChangedDiskAreas` VIM call instead, used alongside +VDDK/NFC reads for the actual bytes. `openvixdisklib.nfc_auth` wraps +that public pyVmomi call directly — no capture, no wire format to +document, per the project rule to reuse pyVmomi for anything it already +exposes. + +## Workflow + +1. `nfc_auth.enable_change_tracking(vm)` — sets + `VirtualMachineConfigSpec.changeTrackingEnabled = True` via + `ReconfigVM_Task`. Takes effect for writes from that point forward; + it does not retroactively track earlier changes. +2. Take a snapshot (or power-cycle the VM). A disk's `changeId` is + empty until this happens. +3. `nfc_auth.disk_change_id(vm, device_key)` — reads the current + `changeId` off `VirtualDisk.backing.changeId` (for example + `"52 f3 b6 37 30 8d ea 3e-58 70 c0 fd 61 44 26 62/2"`). +4. Do backup work (VDDK/NFC reads of the disk at that point). +5. Later, take another snapshot. +6. `nfc_auth.query_changed_disk_areas(vm, new_snapshot, device_key, + change_id_from_step_3)` — returns the byte ranges written between + the two snapshots. +7. Read only those ranges via VDDK/NFC on the new snapshot's disk + chain for an incremental backup. + +For an initial full backup, pass `change_id="*"` in step 6 without a +prior snapshot. **Correction from an earlier draft of this doc:** +this does *not* report the entire disk as one changed extent — see +"Wildcard `changeId='*'` reports allocated regions, not the whole +disk" below. + +## Validated in this lab + +Confirmed end-to-end against a temporary VM on the standalone ESXi +8.0.3 lab host (no vCenter): enabled CBT, snapshotted, wrote one +sector via `openvixdisklib.openvixdisklib` at a known offset, +snapshotted again, and called `query_changed_disk_areas` with the +first snapshot's `changeId`. The single reported extent +(`start=3932160, length=65536`, i.e. sectors 7680–7807) correctly +covered the written sector (7777). Extents were 64 KiB-aligned in this +lab's observations; that granularity is server-defined, not part of +the function's contract. + +### Wildcard `changeId="*"` reports allocated regions, not the whole disk + +Tested `query_changed_disk_areas(vm, snapshot, device_key, "*")` (the +initial-full-backup path, no prior snapshot needed) against a fresh +10 GiB thin-provisioned temp-VM disk. `result.length` correctly +reports the full declared virtual capacity (10737418240 bytes), but +`result.changed_areas` only covered **1 MiB** total — not the whole +disk. For a thin-provisioned disk, `"*"` reports the regions that are +actually *allocated* (backed by real data on the datastore), not the +full sparse virtual capacity; unwritten/unallocated regions have +nothing to back up regardless. A backup tool doing an initial full +backup with `"*"` should read exactly the reported extents, not assume +it needs to read `result.length` bytes. + +### One large contiguous write is one extent; scattered writes are not + +Wrote a single 4 MiB contiguous region plus three separate one-sector +writes at scattered offsets (same disk, one CBT interval), then +queried changed areas: + +``` +4 extents reported: + start= 196608 length= 65536 (64 KiB) + start= 51183616 length= 4259840 (4160 KiB) <-- covers the whole 4 MiB write as ONE extent + start= 460783616 length= 65536 (64 KiB) + start= 921567232 length= 65536 (64 KiB) +``` + +The 4 MiB write came back as a single extent (padded slightly beyond +4 MiB — 4259840 bytes vs. the exact 4194304 written — to the 64 KiB +tail-end granularity). Each scattered single-sector write produced its +own separate 64 KiB extent. **The extent list scales with the number +of discontiguous changed regions, not with the total volume of changed +data.** A multi-hundred-GB sequential write is still one small extent +record; thousands of scattered small writes (e.g. a busy database VM +doing random I/O across a large disk) produce thousands of extent +records in one `QueryChangedDiskAreas` response, since the API has no +pagination. Real backup tools facing that scenario typically chunk the +query with `start_offset` over fixed-size windows rather than querying +the whole disk in one call — `query_changed_disk_areas`'s +`start_offset` parameter exists for this, but nothing in this module +does the chunking loop itself; that is caller responsibility. + +Disk-size scaling itself (e.g., whether extent granularity increases +for very large disks) was not tested — only reasoned about above as an +open question, not verified against a large ESXi 8 disk. + +## What this does not cover + +- `VixDiskLib_QueryAllocatedBlocks` (NFC-level allocated-block bitmap + within a single disk, useful for skipping sparse regions inside a + delta disk) — implemented separately, see `docs/nfc_read.md`. Pairs + naturally with CBT: `query_changed_disk_areas` says which byte + ranges changed, `query_allocated_blocks` says which parts of a + snapshot's delta disk are actually worth reading. Note its "same + still-open write handle" staleness gotcha in `docs/nfc_read.md` if + chaining a CBT-driven write with an allocation check. +- `DDB_GET` fields (`biosGeo`, `adapterType`, `uuid`) — also + implemented, see `docs/nfc_open.md`. diff --git a/docs/encryption.md b/docs/encryption.md new file mode 100644 index 0000000..48990e8 --- /dev/null +++ b/docs/encryption.md @@ -0,0 +1,141 @@ +# Encrypted VM disks + +This is not a reverse-engineered NFC feature. When the ESXi host +serving an NFC session already holds the target disk's encryption key +— the common case, and the only one testable without a second ESXi +host (see "What this does not cover" below) — VM/VMDK encryption is +handled entirely by ESXi's storage stack, **below and invisible to the +NFC protocol**. `openvixdisklib` already reads and writes encrypted +disks correctly with no encryption-specific code at all; this was +verified, not implemented. + +## Investigation + +The backlog item was "encrypted disks" (VDDK supports VM/VMDK +encryption; OpenVixDiskLib had no code path for it and no lab evidence +either way). `strings` on `libvddkVimAccess.so` found suggestive +tokens — `"cannot cast the crypto manager to CryptoManagerHostKMS"`, +`"Cannot push crypto key to host"` — implying VDDK does some kind of +key-provisioning VIM call before NFC I/O on an encrypted disk. Testing +this needed an actual encrypted VM, which needed vCenter (VM +encryption / Key Management Server configuration is a vCenter-only +concept — a bare standalone ESXi host's `cryptoManager` is the base +`vim.encryption.CryptoManagerHost`, not `CryptoManagerHostKMS`, and +`AddKey` fails with `InvalidState`/"crypto state incapable": no KMS +trust relationship is possible without vCenter). Building that lab +(vCenter Server Appliance + a vSphere Native Key Provider + an +encrypted test VM) is documented separately in +`docs/encryption_lab_setup.md`. + +With a real encrypted disk in hand, the same SSL-hook technique as +every other NFC feature (`docs/ssl_hook.md`) captured native VDDK +8.0.2 opening it, compared against an identical capture against an +unencrypted disk on the same host. Findings: + +- VDDK's own log makes the branch explicit. Unencrypted: + `VddkVimAccess_HandleDiskCryptoKey: Handle the key of disk ...` + immediately followed by + `HandleDiskCryptoKey: Disk '...' is not encrypted. No need to handle + disk crypto key.` Encrypted: the same `HandleDiskCryptoKey` line + fires, but that second message is absent — it goes straight to + `VddkVimAccess_FreeNfcTicket` with no further narration. +- The encrypted-disk capture (844 SSL/socket records covering both the + vCenter SOAP session and the ESXi NFC session) contains **no** + `AddKey`, `ConfigureCryptoKey`, `EnableCrypto`, `PrepareCrypto`, or + `QueryCryptoKeyStatus` calls — the VIM methods that looked like + candidates for "push key to host" from the binary strings. The only + crypto-related SOAP content anywhere in the capture is routine + `RetrieveServiceContent` boilerplate (`cryptoManager + type="CryptoManagerKmip"` at vCenter, `type="CryptoManagerHostKMS"` + at the host) that appears on every connection, encrypted disk or not. + +That's negative evidence (no extra call *seen*), which only proves the +happy path. The decisive test was positive: used `openvixdisklib` +itself — no crypto-specific code — to write a known byte pattern to +an encrypted disk's sector 0 through normal `VixDiskLib_Write`-equivalent +NFC I/O, read it back through a **separate** `open()` (avoiding the +same-handle staleness class of bug noted in `docs/nfc_read.md`), and +got an exact match: a completely ordinary, transparent NFC round-trip. +Then fetched the same byte range directly from the disk's +`-flat.vmdk` file over ESXi's datastore-browser HTTP API +(`Range: bytes=0-511`) and confirmed the raw bytes on disk are genuine +ciphertext — the plaintext pattern does not appear. Real +encryption-at-rest is happening (confirmed independently at the wire +level too: the `.vmdk` descriptor carries `keyID=`, an `encryptionKeys=` +wrapped-key blob, and `ddb.iofilters = "vmwarevmcrypt"`), and it is +completely invisible to any NFC client, VDDK or otherwise. + +## Why this is the expected result + +VDDK's "push crypto key to host" logic exists for *provisioning*: the +case where the ESXi host about to serve NFC I/O does **not** already +have the disk's key cached locally — for example, right after a +cross-host vMotion, clone, or restore places the encrypted VM on a +host that has never served it before. In that case VDDK (running with +appropriate vCenter privileges) fetches the key and pushes it to the +target host via `CryptoManagerHostKMS.AddKey` before NFC can succeed. +Once the host has the key — which it always will for a VM that has +simply been encrypted and left in place, the scenario this lab's +single ESXi host can produce — ESXi's storage stack decrypts/encrypts +transparently at the IOFilter layer for every client, with nothing +encryption-specific in the NFC wire protocol at all. + +### Transports not covered + +Only the NBD/NFC path was tested. The `hotadd` transport most likely +behaves the same way, since the disk is still read through the ESXi +storage stack. The `san` transport is an open question: it reads the +underlying LUNs (iSCSI, Fibre Channel) directly, mounts VMFS and parses +the VMDK itself, bypassing the IOFilter layer where decryption happens, +so it would presumably see ciphertext. Not verified. + +## Validated in this lab + +Single-ESXi-host lab (`docs/encryption_lab_setup.md`), VM +`ovdl-crypto-test` with both its home directory and its virtual disk +encrypted via a vSphere Native Key Provider: + +- Read/write round-trip via `openvixdisklib.openvixdisklib` over the + vCenter-managed path (`server_name=`, `vmx_spec="moref=vm-N"`) + — matches the pattern written, confirmed as genuine ciphertext at + rest as described above. +- Native VDDK 8.0.2 (`ConnectEx` + `Open` + `Read`) succeeds against + the same disk with no special handling required, decrypting + correctly (a freshly-created 1 GiB disk reads back as zeros, as + expected for an unwritten region). + +## Cross-host key provisioning (resolved 2026-09-20) + +The gap flagged below — a host that doesn't already have the disk's +key cached — was resolved once a second ESXi host became available +(built for the host-switch investigation, `docs/host_switch_lab_setup.md`). +Relocated the encrypted test VM (compute **and** its disk, cold — +powered off) to a host that had never held its key at all, then opened +the disk on that new host: it just worked, both via `openvixdisklib` +and via cross-checked native VDDK, decrypting correctly with zero key +management on the client's part. + +vCenter pushes the key to the destination host automatically as part +of any relocation of an encrypted VM (`RelocateVM_Task`, cold or live) +— entirely transparent to any NFC client. The `CryptoManagerHostKMS.AddKey` +("push crypto key to host") mechanism the binary strings pointed to +does exist and does get exercised, but it's driven by vCenter's own +migration workflow, not by VDDK or any NFC client. There is nothing +for OpenVixDiskLib to implement here either: the "host doesn't have +the key yet" case simply cannot arise for a client reading a disk +through a normal, already-completed migration, because vCenter +resolves it before the migration finishes. + +## What this does not cover + +- Standard/KMIP external Key Management Servers were not tested — only + vSphere Native Key Provider. The NFC-transparency conclusion should + hold regardless of which KMS variant provisioned the key (the + key-caching-on-host mechanism is the same either way), but this + wasn't independently verified. +- Direct-ESXi (no vCenter) access to an encrypted disk was attempted + but not completed — hit an unrelated, reproducible pyVmomi issue + (a bare `vim.VirtualMachine("vm-N", stub)` construction raising + `ManagedObjectNotFound` against this specific host, independent of + encryption) that wasn't investigated further. Worth retrying if this + recurs elsewhere. diff --git a/docs/encryption_lab_setup.md b/docs/encryption_lab_setup.md new file mode 100644 index 0000000..e0f507a --- /dev/null +++ b/docs/encryption_lab_setup.md @@ -0,0 +1,212 @@ +# Building a VM-encryption test lab from a bare ESXi host + +This project's lab may be a standalone ESXi host (no vCenter — see +`docs/reverse_engineering_procedure.md`'s "Lab and artifacts"). VM/VMDK +encryption is a vCenter-only feature: a bare ESXi host's `cryptoManager` +is the base `vim.encryption.CryptoManagerHost`, not +`CryptoManagerHostKMS`, and pushing a key to it fails with +`InvalidState`/"crypto state incapable" — there is no way to test +encryption without standing up vCenter first. This document is that +procedure, distilled from doing it once. It assumes only what the base +lab already has: one ESXi host with spare RAM/CPU/datastore headroom +and a VCSA installer ISO (Broadcom-account-gated; not something this +procedure can supply). + +Runnable, credential-parameterized versions of every script referenced +here live in `.lab/` (gitignored — see `.lab/README.md`), not in this +doc. + +## 0. Size the host + +vCenter Server Appliance's smallest deployment size ("tiny") needs +**2 vCPUs and ~14 GiB RAM** on its own. If the ESXi host is itself a +nested VM (as in this lab — a KubeVirt VM on a Harvester cluster), its +own memory allocation may need to grow first. Check headroom on +whatever hosts the nested ESXi VM before resizing; growing a shared +VM's memory allocation is a shared-infrastructure change worth +confirming before doing it, not something to script unattended. + +## 1. Deploy VCSA to the ESXi host + +Use the CLI installer (`vcsa-cli-installer/lin64/vcsa-deploy`) from the +VCSA ISO, with a JSON template based on +`vcsa-cli-installer/templates/install/embedded_vCSA_on_ESXi.json`. +Two template gotchas that aren't obvious from VMware's own docs: + +- Use `"os": {"time_tools_sync": true}` instead of `"ntp_servers"` + pointed at a LAN router. Home-lab routers usually aren't real NTP + servers, and VCSA's firstboot hard-fails deployment entirely on an + NTP sync error (`err_ntp_sync_failed`). `time_tools_sync` syncs off + the ESXi host's own clock via VMware Tools instead — no NTP + dependency at all. +- `deployment_option: "tiny"` is enough for a throwaway test VM; verify + the target datastore has at least the ~25 GB free space VMware's own + docs require (thin-provisioned, so actual usage is much less + initially). + +**Environment gotcha**: the bundled `ovftool`/`vcsa-deploy.bin` are +old enough to need `libnsl.so.1` and `libcrypt.so.1`, which current +openSUSE (and likely other rolling-release distros) no longer ship by +default. Rather than hunt for compatible system packages, run the +whole CLI installer inside a throwaway `rockylinux:9` podman container, +which has the exact sonames needed: + +```bash +podman run --rm --network=host \ + -v /path/to/mounted/vcsa-iso:/mnt/vcsa-iso:ro \ + -v "$(pwd)":/work:rw \ + rockylinux:9 bash -c " + dnf install -y libnsl which ncurses libxcrypt-compat glibc-langpack-en + /mnt/vcsa-iso/vcsa-cli-installer/lin64/vcsa-deploy install \ + --accept-eula --no-ssl-certificate-verification --acknowledge-ceip \ + /work/vcsa-deploy-config.json + " +``` + +`--network=host` is needed so the container can reach the ESXi host +and the appliance's own IP directly on the LAN. + +## 2. Add the host to vCenter — and into a cluster + +Adding the host as a standalone host under a new datacenter is the +obvious first step (`Folder.CreateDatacenter` + +`HostFolder.AddStandaloneHost`), but **it is not enough**: Native Key +Provider setup fails at the point of actually encrypting a VM with + +``` +vim.vpxd.encryption.NativeKeyProviderNotSupported.NotInCluster +``` + +if the host is standalone rather than in a cluster, even though +earlier checks (`CryptoManagerKmip.IsKmsClusterActive`) may already +report the provider as active. Create an empty cluster +(`HostFolder.CreateClusterEx` with a bare `vim.cluster.ConfigSpecEx()` +— no DRS/HA needed) and move the host into it +(`ClusterComputeResource.MoveInto_Task`) before going further. + +**Side effect worth knowing about**: creating a cluster makes vCenter +automatically try to deploy its own vSphere Cluster Services (vCLS) +agent VMs onto it — unconditionally, even with DRS/HA both off. On a +nested-virtualization host that doesn't expose certain CPU features at +that nesting depth (this lab saw `Feature 'MWAIT' was 0, but must be 1` +and `Extended APIC register space was absent`), these VMs fail to +power on and vCenter retries indefinitely (~every 30s), cluttering +Recent Tasks. Harmless for testing purposes; can be silenced via the +cluster's "Retreat Mode" advanced setting if the noise becomes a +problem. + +## 3. Create a Native Key Provider + +No external KMIP server needed — vSphere's built-in Native Key +Provider (NKP) manages keys itself. The catch: **NKP creation is not +exposed over classic VIM SOAP at all.** +`CryptoManagerKmip.RegisterKmsCluster`'s `managementType` parameter +looks like it should do this, but it's read-only/reporting-only — +passing it raises `InvalidArgument`. NKP can only be *created* through +vCenter's newer REST/vAPI (`/api/vcenter/crypto-manager/kms/providers`). + +The official Python client for that API +(`vsphere-automation-sdk-python` on GitHub, or `vsphere-automation-sdk` +on PyPI) may not be installable in every environment — the PyPI sdist +is broken (missing `LICENSE.txt`) as of this writing, and installing +the GitHub version's own dependencies (`vapi-runtime`, +`vapi-common-client`) may trip a sandboxed environment's +package-installation safeguards. The 3 REST calls actually needed are +simple enough to hand-roll with plain `requests`: + +1. `POST /api/session` with HTTP Basic auth → a session-id string, sent + back as the `vmware-api-session-id` header on every later call. +2. `POST /api/vcenter/crypto-manager/kms/providers` with + `{"provider": "", "constraints": {"tpm_required": false}}` — + creates the provider, but it comes back `"health": "ERROR"` + ("requires backup") until step 3. +3. `POST /api/vcenter/crypto-manager/kms/providers?action=export` + (**the query param must be on the providers *collection* URL, not + the per-provider URL** — `.../providers/?action=export` + 404s) with `{"provider": "", "password": ""}`. The + response's `location.url` may carry vCenter's own (possibly stale) + reverse-DNS hostname instead of its real address — substitute + before using it. `POST` that URL with + `Authorization: Bearer ` to actually download + the backup blob. This step is what flips `health` to `OK` — for a + Native Key Provider, "back it up" is not an optional safety step, + it's what activates the provider in the first place. + +Mark it default afterward with plain VIM SOAP (this part *does* work +classically): `CryptoManagerKmip.SetDefaultKmsCluster(clusterId=...)`. + +## 4. Create and encrypt a test VM + +Two separate operations, in order: + +1. **Encrypt the VM's home directory** — a `ReconfigVM_Task` with + `VirtualMachineConfigSpec.crypto = CryptoSpecEncrypt(cryptoKeyId=...)`. + This alone is not enough on a VM with no vTPM or existing encryption + storage profile — it fails with + `encryptForbiddenWithoutEncryptedProfile`. Add a + `vim.vm.device.VirtualTPM()` device in the **same** `ReconfigVM_Task` + call as the crypto spec to satisfy this. + For a Native Key Provider specifically, leave + `CryptoKeyId.keyId = ""` and let vCenter auto-generate the actual + key — `CryptoManagerKmip.GenerateKey()` explicitly rejects native + providers ("Key provider ... is managed by ... or is a native key + provider"). +2. **Encrypt the disk itself** — step 1 does *not* encrypt attached + virtual disks; `backing.keyId` stays `None` and the `.vmdk` + descriptor is unchanged. A raw device-edit attempt + (`VirtualDeviceSpec.backing = BackingSpec(crypto=CryptoSpecEncrypt(...))`, + `operation=edit`) fails with `badPolicy` / + "Invalid storage policy for encryption operation" — disk encryption + is only reachable through SPBM (Storage Policy-Based Management), + which is yet another separate SOAP endpoint (`/pbm/sdk`). + Driving SPBM programmatically hit an unresolved + `vmodl.fault.SecurityError` on `PbmQueryProfile` even after + correctly copying the vCenter session cookie to the PBM stub (the + standard pyvmomi-community-samples pattern); not worth further + debugging time for a one-off lab setup. **Working fallback: the + vSphere Client UI** — right-click the VM → **VM Policies → Edit VM + Storage Policies** → set the storage policy to the built-in + **"VM Encryption Policy"** (a system-created profile every vCenter + ships; don't confuse it with "Management Storage Policy - + Encryption", a different profile for infrastructure objects) → OK. + This silently relocates the disk file (e.g. `foo.vmdk` → + `foo_1.vmdk`) — re-read the VM's device list afterward. + +## 5. Verify it's actually encrypted + +Don't trust `backing.keyId` alone — confirm at the wire level by +reading the raw `.vmdk` descriptor over ESXi's datastore-browser HTTP +API: + +``` +GET https:///folder//.vmdk?dcPath=ha-datacenter&dsName= +``` + +**Gotcha**: this legacy per-host API still uses `ha-datacenter` (the +standalone-host pseudo-datacenter name) even after the host becomes +vCenter-managed under a real datacenter name — not the vCenter +datacenter name. A genuinely encrypted disk's descriptor has +`keyID=...`, an `encryptionKeys="vmware:key/list/..."` line, and +`ddb.iofilters = "vmwarevmcrypt"`. For the strongest possible proof, +write a known byte pattern through normal NFC I/O and fetch the same +byte range from the `-flat.vmdk` file directly (`Range: +bytes=-`) — it should not contain the plaintext +pattern. + +## What this lab cannot test + +Nothing here provisions a *second* ESXi host, so the "host doesn't +already have this disk's key" scenario — cross-host vMotion/clone/ +restore of an encrypted VM — cannot be exercised. See +`docs/encryption.md`'s "What this does not cover" for why that matters. + +## Running the integration test + +`tests/integration/test_encryption.py` reuses the vCenter settings from +`.test_config.yaml` (see `README.md`) and needs two extra keys pointing +at the already-encrypted VM and disk (it is skipped without them): + +```yaml +encrypted_vm_moref: vm-123 +encrypted_disk_path: "[datastore0] vm/vm_1.vmdk" +``` diff --git a/docs/nfc_auth.md b/docs/nfc_auth.md index 36e89ed..0941b90 100644 --- a/docs/nfc_auth.md +++ b/docs/nfc_auth.md @@ -59,9 +59,9 @@ VDDK logs this as `Connected to VIM Server` / `Authenticating user` / `Logged in!`. OpenVixDiskLib keeps that `ServiceInstance` and its stub for the ticket call. -Direct ESXi login is the same SOAP login against hostd, but the NFC -moref and service name differ (`ha-nfc` instead of `nfcService` / -`vpxa-nfc`). The lab path is vCenter-mediated. +Direct ESXi login is the same SOAP login, this time against hostd +instead of vCenter. The NFC moref and service/PROXY name differ; see +"Direct ESXi (no vCenter)" below for the verified values. ## Stage 2: NFC ticket @@ -255,6 +255,90 @@ are for local ESXi credentials. With a vCenter ticket: argument is not what VDDK sends. The SHA-1 value is for verifying the TLS certificate, not for the `THUMBPRINT_SHA2` command. +## Direct ESXi (no vCenter) + +Captured against a standalone ESXi 8.0.3 host (`apiType: HostAgent`, +no vCenter in the picture at all) with the same SSL-hook technique +from `docs/ssl_hook.md`, using real VDDK 8.0.3 pointed straight at the +host (`vmxSpec=moref=`, `serverName=`). This corrects an +earlier guess in this file that assumed the moref would be `ha-nfc`. + +Differences from the vCenter-mediated path above: + +| Item | vCenter-mediated | Direct ESXi (verified) | +| ------------------------------- | ---------------------- | -------------------------- | +| `NfcService` moref | `nfcService` | `ha-nfc-service` | +| `NfcGetVmFilesResponse.service` | `vpxa-nfc` | `nfc` | +| `NfcGetVmFilesResponse.host` | present (ESXi address) | **absent** (omitted field) | +| authd `PROXY` line | `PROXY vpxa-nfc` | `PROXY nfc` | +| authd success line | `200 Connect ha-nfc` | `200 Connect ha-nfc` | + +The `NfcGetVmFiles` SOAP call itself is unchanged (`vm` argument only); +only the `_this` moref and the response fields differ: + +```xml + + <_this type="NfcService">ha-nfc-service + 1 + +``` + +```xml + + + 902 + ... + nfc + 1.1 + ... + + +``` + +Since `host` is absent, the client must already know where to dial +authd: the same ESXi host it just logged into over VIM. A vCenter +ticket always fills `host` because that ESXi address is not otherwise +known to the client. + +### Finding the `ha-nfc-service` moref + +VDDK does not hardcode this moref either. Before the `NfcGetVmFiles` +call, it issues an undocumented `RetrieveInternalContent` call on the +same `ServiceInstance` moref used for the public +`RetrieveServiceContent`: + +```xml + + <_this type="ServiceInstance">ServiceInstance + +``` + +The response carries ~20 undocumented managed-object refs +(`agentManager`, `llProvisioningManager`, `diskManager`, +`nfcService`, `proxyService`, ...); only `nfcService` matters here. +Its value was `nfcService` in the earlier vCenter capture and +`ha-nfc-service` on this bare ESXi host — VDDK reads it from this +response rather than assuming either name. + +### OpenVixDiskLib fix + +`openvixdisklib/nfc_auth.py` previously hardcoded +`NFC_SERVICE_MOID = "nfcService"`, which fails outright against a bare +ESXi host with `vmodl.fault.ManagedObjectNotFound`. It now resolves +the moref the same way VDDK does: `_nfc_service_moid()` issues the +`RetrieveInternalContent` SOAP call as raw XML over the existing +authenticated stub connection (registering pyVmomi types for the full +undocumented response schema wasn't worth it for one field) and +regex-extracts `nfcService` from the reply. + +`connect_authd()` also gained a `fallback_host` parameter: when +`ticket.host` is unset (the direct-ESXi case above), it dials the VIM +connection's own host instead. `openvixdisklib.py` passes +`conn.si._stub.host` for this. + +Validated end-to-end (`ConnectEx` + `Open` + `Read`, both `nbd` and +`nbdssl` transports) against a live standalone ESXi 8.0.3 host. + ## OpenVixDiskLib | Piece | Module | Reuses pyVmomi? | diff --git a/docs/nfc_open.md b/docs/nfc_open.md index bf61550..9ad7601 100644 --- a/docs/nfc_open.md +++ b/docs/nfc_open.md @@ -143,7 +143,7 @@ AIO types used for Open / Read / Close, correlated with the consecutive | 9 | `SET_SOCK_OPTS` | 12 | | | 22 | `SET_RES_POOL` | 4 | | | 4 | `OPEN_FILE` | 60 | path string | -| 11 | `DDB_GET` | 16 | key name (VDDK only) | +| 11 | `DDB_GET` | 16 | key name | | 7 | `IO` | 44 | sector bytes (read reply / write request) | | 5 | `CLOSE_FILE` | 8 | | | 3 | `CLOSE_SESSION` | 4 | | @@ -156,6 +156,65 @@ VDDK Open also issues several `DDB_GET` queries (`resumeConsolidateSector`, (16 zero bytes) on this unencrypted disk. They are not required to obtain a file handle or to read sector 0. +`VixDiskLib_GetInfo` (Step 14) triggers ~20 more `DDB_GET` calls right +after `OPEN_FILE`, for these keys (captured in request-order, key name +is the extra string after the 16-byte payload, no `ddb.` prefix on the +wire): `resumeConsolidateSector`, `isDigest` (×3), `iofilters` (×2), +`logicalSectorSize` (×2), `physicalSectorSize` (×2), +`isNativeLinkedClone` (×2), `KMFilters`, `sidecars`, `adapterType`, +`uuid`, `geometry.cylinders`, `geometry.heads`, `geometry.sectors`, +`geometry.biosCylinders`, `geometry.biosHeads`, `geometry.biosSectors`. +On this lab disk, `biosGeo` came back all zeros (key not found) and +`logicalSectorSize`/`physicalSectorSize`/the non-bios `geometry.*` keys +duplicate what OPEN_FILE already returned — only `adapterType` and +`uuid` are genuinely new information from this burst. + +### `DDB_GET` request/reply layout + +Decoded from the same capture (request/reply pairs matched by `opId` +across all ~28 calls seen in one `GetInfo`). + +Request: 16-byte fixed payload plus the key name as a raw ASCII extra +(no NUL terminator, not counted in `size` — same convention as +`OPEN_FILE`'s path): + +| Offset | Type | Meaning | +| ------ | -------- | --------------------------------------- | +| 0 | `uint64` | File handle (same value as `OPEN_FILE`) | +| 8 | `uint32` | Key name length in bytes | +| 12 | `uint32` | 0 | +| 16 | — | Key name (ASCII, no `ddb.` prefix) | + +Reply: 16 bytes plus a value extra, **not** padded (unlike +`QueryAllocatedBlocks`'s bitmap — verified by decoding all 28 replies +in sequence with no desync): + +| Offset | Type | Meaning | +| ------ | -------- | --------------------------------------- | +| 0-11 | — | Zero/unused in every capture | +| 12 | `uint32` | Value length in bytes (`0` = not found) | +| 16 | — | Value (ASCII **text**, not binary) | + +Values are ASCII text even for keys that sound numeric — +`geometry.cylinders` comes back as the literal bytes `b"2088"`, not a +binary `uint32`. This matches how a VMDK descriptor file's DDB (disk +database) section stores keys as plain-text `ddb. = ""` +lines; `adapterType` comes back as `b"lsilogic"` (a string), not +VDDK's numeric `VIXDISKLIB_ADAPTER_SCSI_LSILOGIC` enum value — VDDK's +own client does that string-to-enum mapping internally, which +OpenVixDiskLib does not reproduce (`DiskInfo.adapter_type` is the raw +DDB string). + +Implemented as `openvixdisklib.nfc_open.NfcDisk.ddb_get(key) -> str | +None` and `NfcDisk.query_full_info() -> DiskInfo` (5 round trips: +`geometry.biosCylinders`/`biosHeads`/`biosSectors`, `adapterType`, +`uuid`), wired into `VixDiskLibHandle.get_info`, which now matches +real VDDK's `VixDiskLib_GetInfo` exactly — capacity/physGeo free from +`OPEN_FILE`, the rest costing the same 5 round trips VDDK itself pays. +Validated against the live ESXi lab: matches native VDDK's `GetInfo` +output on the same disk (`adapterType=3` ↔ `"lsilogic"`, same `uuid` +string, same zeroed `biosGeo`). + ### OPEN_SESSION / sockopts / resource pool `OPEN_SESSION` payload is 16 bytes, little-endian: @@ -205,9 +264,19 @@ Reply payload (60 bytes), fields that matter: | 8 | `uint64` | File handle (opaque, per open) | | 16 | `uint32` | File type (`2` = `NFC_DISK`) | | 20 | `uint32` | Flags echoed (`0x1e` or `0x1a`) | +| 28 | `uint64` | Disk capacity in **bytes** | | 36 | `uint32` | Sector size (`512` on this VM) | - -Later AIO messages pass that handle as a `uint64`. +| 40 | `uint32` | Physical geometry cylinders | +| 44 | `uint32` | Physical geometry heads | +| 48 | `uint32` | Physical geometry sectors | + +Later AIO messages pass that handle as a `uint64`. Offset 28 was found +by capturing `VixDiskLib_GetInfo` (Step 14, +`docs/reverse_engineering_procedure.md`): it matches +`VixDiskLibInfo.capacity` converted to bytes, and offsets 40/44/48 +match `VixDiskLibInfo.physGeo` exactly — both already arrive with this +reply, no separate `GetInfo` wire call exists. `biosGeo`, `adapterType`, +and `uuid` are **not** here; VDDK gets those from `DDB_GET` (below). ### IO (read / write) @@ -232,6 +301,8 @@ classic type 4 `NFC_SESSION_COMPLETE`. | Handshake + AIO + OPEN_FILE | `openvixdisklib.nfc_open.open_disk` | | AIO extra size / pool count | `open_disk(..., aio_buffer_size=, aio_buffer_count=)` | | Sector read / write / close | `openvixdisklib.nfc_open.NfcDisk` | +| Full disk info (`GetInfo`) | `openvixdisklib.openvixdisklib.VixDiskLibHandle.get_info` | +| VMDK descriptor DDB lookup | `openvixdisklib.nfc_open.NfcDisk.ddb_get` | Run: @@ -246,10 +317,18 @@ I/O: `docs/nfc_read.md`, `docs/nfc_write.md`, and ## What is still VDDK-only -- `DDB_GET` / geometry / zlib and skipz compression / encryption keys -- `NFC_DELTA_DISK`, change-block tracking +- zlib and skipz compression - Host-switch (`NFC_AIO_SWITCH_HOST_*`) -- Direct ESXi `ha-nfc` without vCenter `vpxa-nfc` + +Reading/writing a snapshot delta file directly, and running +`query_allocated_blocks` against it, both already work with the +existing implementation — `NFC_DELTA_DISK` turned out to be an +optional VMFS-only VDDK client optimization, not a correctness +requirement; see `docs/reverse_engineering_procedure.md`. + +Reading/writing an encrypted disk also already works with the existing +implementation — ESXi handles encryption transparently below NFC +whenever the host already holds the key; see `docs/encryption.md`. Reads after open are in `docs/nfc_read.md`. Writes are in `docs/nfc_write.md`. diff --git a/docs/nfc_read.md b/docs/nfc_read.md index ee67652..ceafe22 100644 --- a/docs/nfc_read.md +++ b/docs/nfc_read.md @@ -189,9 +189,82 @@ uncompressed request (offsets in this read, not on disk) buf when skip_decompression=True: extras packed densely from offset 0 ``` +## `VixDiskLib_QueryAllocatedBlocks` (AIO type 13) + +Reverse-engineered by extending the SSL/write-hook capture (Step 13/14 +technique, `docs/reverse_engineering_procedure.md`) to a ctypes call to +`VixDiskLib_QueryAllocatedBlocks` after `Open`, first with +`startSector=0` then — after the first capture's field guesses turned +out wrong — again with a non-zero `startSector` against a known +already-allocated region, to disambiguate fields that are 0 in the +degenerate zero-start case. + +No SOAP or authd traffic; it is one more AIO message type in the +already-open NFC/AIO session (like `DDB_GET`). + +Request (48 bytes):: + + uint64 handle (from OPEN_FILE) + uint64 reserved (0) + uint64 chunk_size_bytes (chunk_size_sectors * sector_size) + uint64 start_offset_bytes (start_sector * sector_size) + uint64 chunk_count (num_sectors // chunk_size_sectors) + uint64 reserved (0) + +**Field-order pitfall:** `start_offset_bytes` is at byte offset 24, not +8 — offset 8 is a reserved/always-zero field. A capture with +`startSector=0` can't tell these two apart (both read 0); only a +capture with a non-zero start distinguishes them. A first +implementation attempt put `start_offset_bytes` at offset 8 and got +`chunk_count`-many all-zero bits back for every non-zero-start query, +even for byte ranges known (from a zero-start, full-range query) to be +allocated — the server was silently ignoring the offset the client +thought it was requesting and returning an artifact of a different +misread field. + +Reply: a 48-byte body (offset 32 echoes `chunk_count`) followed by a +bitmap extra, one bit per chunk (LSB-first, `1` = chunk has allocated +data), **padded up to a 4-byte boundary** — `ceil(chunk_count / 8)` +alone is correct only when that value is already a multiple of 4 +(true for the `chunk_count=16384` case tested first, which is why the +padding bug wasn't caught immediately; a `chunk_count=16` query +exposed it, since `ceil(16/8)=2` bytes under-reads the real 4-byte +reply and desyncs the connection — the *next* AIO reply's header then +reads as garbage). + +Both `start_sector` and `num_sectors` must be exact multiples of +`chunk_size_sectors`; the server returns an `NFC_AIO_MSG_ERROR` (type +1) reply otherwise (hit by accident during validation with a +non-aligned `start_sector`). + +`openvixdisklib.nfc_open.NfcDisk.query_allocated_blocks` implements +this and run-length-merges contiguous set bits into +`AllocatedBlock(offset, length)` tuples (sectors, matching VDDK's +`VixDiskLibBlock`), exposed as +`VixDiskLibHandle.query_allocated_blocks`. Validated against the live +ESXi lab: a full-disk query, a query of a known-allocated sub-range, +and an aligned empty range all match native VDDK's own +`VixDiskLib_QueryAllocatedBlocks` output on the same disk. + +### Gotcha: query on the same still-open write handle can see stale data + +Writing a sector and then immediately calling +`query_allocated_blocks` **on that same open handle, without closing +it first**, can report the just-written region as *not* allocated — +the allocation metadata this call reads apparently isn't guaranteed +current until the write handle is closed. Closing after the write and +reopening (or querying from a separate handle opened after the write +completed) reports it correctly. Confirmed on both native VDDK and +this implementation — same-session-no-close showed the write as +unallocated on both, a fresh handle after close showed it correctly +on both — so this is a real server/VMFS behavior, not a bug in either +client. Real backup tools reading allocation before a read pass +naturally do this anyway (open read-only after the writer's handle +already closed), so it's unlikely to bite in practice, but do not +call `query_allocated_blocks` right after a write on the same handle +and expect it to reflect that write. + ## What is still VDDK-only - zlib and skipz NBD compression flags - `VixDiskLib_ReadAsync` (same IO messages, different client threading) -- `VixDiskLib_QueryAllocatedBlocks` / allocation bitmaps -- `VixDiskLib_GetInfo` capacity (not required to read a known range) diff --git a/docs/reverse_engineering_procedure.md b/docs/reverse_engineering_procedure.md index b988f5c..25ec574 100644 --- a/docs/reverse_engineering_procedure.md +++ b/docs/reverse_engineering_procedure.md @@ -8,8 +8,10 @@ This file is the **sequence of steps**, including dead ends, so later NFC work can follow the same loop instead of rediscovering it. Scope so far: `VixDiskLib_ConnectEx` + `VixDiskLib_Open` + -`VixDiskLib_Read` + `VixDiskLib_Write` against lab vCenter 8.0.1 / -ESXi 8, transports `nbd` and `nbdssl`. Validation method: +`VixDiskLib_Read` + `VixDiskLib_Write` + `VixDiskLib_GetInfo` + +`VixDiskLib_QueryAllocatedBlocks` against lab vCenter 8.0.1 / ESXi 8, +transports `nbd` and `nbdssl`, plus a standalone ESXi 8.0.3 host with +no vCenter (Step 13). Validation method: `tests/integration/` (the session-scoped `lab` fixture creates a temporary empty VM with a 10 GiB disk and destroys it when the pytest session ends). @@ -384,17 +386,193 @@ compression on each IO. Proof: `tests/integration/test_nfc_read_write.py` (`fastlz`) and `tests/perf/test_compare.py`. +## Step 13 — Direct ESXi (`ha-nfc`) without vCenter + +Same SSL-hook technique (Step 4), this time pointing VDDK 8.0.3 +straight at a standalone ESXi 8.0.3 host (`vmxSpec=moref=`, +`serverName=`, no vCenter in the topology). Confirmed +OpenVixDiskLib's hardcoded `NFC_SERVICE_MOID = "nfcService"` fails on +this host with `vmodl.fault.ManagedObjectNotFound` *before* touching +the capture — reproduced with plain `openvixdisklib` calls, no hook +needed to see that failure. + +The capture showed VDDK does not hardcode the moref either: it calls +an undocumented `RetrieveInternalContent` on the `ServiceInstance` +moref first, and reads `nfcService` from the reply (`ha-nfc-service` +on this host, vs. `nfcService` in the earlier vCenter capture). +`NfcGetVmFilesResponse.service` was `nfc` (not `vpxa-nfc`), and its +`host` field was **absent** — the authd endpoint is implicitly the +same host already logged into. Full detail in `docs/nfc_auth.md` +("Direct ESXi (no vCenter)"). + +Fix: `_nfc_service_moid()` in `openvixdisklib/nfc_auth.py` issues the +`RetrieveInternalContent` call as raw SOAP over the existing stub +connection (its response schema has ~20 other undocumented morefs not +worth registering with pyVmomi's type system for one field), and +`connect_authd()` takes a `fallback_host` used when `ticket.host` is +unset. Validated end-to-end (`ConnectEx`/`Open`/`Read`, `nbd` and +`nbdssl`) against the live host. + +## Step 14 — `VixDiskLib_GetInfo` capacity + +Extended the ctypes probe from Step 13 to call `VixDiskLib_GetInfo` +after `Open`, under the SSL hook plus a `write`/`read` interceptor on +fd 902 (Step 7), to see what wire traffic `GetInfo` adds. + +Result: **no new SOAP or authd traffic** — the same `RetrieveContent` ++ `Login` + `NfcGetVmFiles` + authd sequence as a plain `Open`. All the +extra traffic is inside the already-open NFC/AIO session: ~20 more +`DDB_GET` (type 11) requests right after `OPEN_FILE`, for keys like +`adapterType`, `uuid`, `geometry.cylinders`, `geometry.biosCylinders`, +etc. (full list in `docs/nfc_open.md`). + +Dumping every byte of the `OPEN_FILE` reply (not just the fields the +earlier Open-only capture had labeled) found `capacity` (offset 28, +`uint64` bytes) and `physGeo` (offsets 40/44/48) already present — +verified they match `VixDiskLibInfo.capacity`/`physGeo` from the same +`GetInfo` call exactly. Only `biosGeo`, `adapterType`, and `uuid` are +genuinely `DDB_GET`-only; `biosGeo` came back "key not found" (zeros) +on this unencrypted lab disk. + +Fix: extended `_parse_open_reply` in `openvixdisklib/nfc_open.py` to +also read those offsets, added `nfc_open.DiskInfo`/`DiskGeometry`, and +exposed `VixDiskLibHandle.get_info()`. No new NFC message type was +needed — `DDB_GET` (`adapterType`/`uuid`/`biosGeo`) is still open work. +Validated against the live host: `capacity_sectors=33554432` +(16 GiB), `phys_geo=(2088, 255, 63)`, matching native VDDK's +`GetInfo` on the same disk. + +## Note — CBT needed no reverse engineering + +Investigated change-block tracking (backlog item "CBT / +`QueryAllocatedBlocks`") expecting an NFC capture like the steps above. +It turned out `VirtualMachine.QueryChangedDiskAreas` — the actual +changed-byte-range query backup tools use — is public pyVmomi API with +no VDDK/NFC involvement at all; only `VixDiskLib_QueryAllocatedBlocks` +(disk-internal allocated-block bitmap, a different and lesser feature) +needed NFC work (done separately, see Step 15 below). Implemented as +`openvixdisklib.nfc_auth.enable_change_tracking` / +`disk_change_id` / `query_changed_disk_areas`; validated end-to-end +against a temp VM on the lab (enable CBT, snapshot, write a known +sector, snapshot, query — the written sector fell inside the reported +extent). Full workflow and lab evidence: `docs/cbt.md`. + +## Step 15 — `VixDiskLib_QueryAllocatedBlocks` (AIO type 13) + +Same SSL/write-hook technique, extended to `VixDiskLib_QueryAllocatedBlocks` +after `Open`. A first capture with `startSector=0` produced a request +where two 0-valued 8-byte fields were ambiguous — either could plausibly +be `start_offset_bytes`. Implementing from that guess alone put +`start_offset_bytes` at the wrong offset (8 instead of 24) and passed +every zero-start test while silently returning wrong (all-empty) +results for any non-zero start. A second capture with a **non-zero** +`startSector` against a region already known (from the first capture) +to be allocated broke the tie and found the real field order. + +A second, independent bug (bitmap reply length) was caught the same +way: `ceil(chunk_count/8)` matched the observed reply length for +`chunk_count=16384` (already a multiple of 4) but under-read and +desynced the connection for `chunk_count=16` — the reply pads the +bitmap to a 4-byte boundary. Diagnosed by testing candidate `extra_recv` +byte counts against whether the *next* AIO round-trip (a normal +`CLOSE_FILE`) completed cleanly, rather than guessing from a single +capture. + +Full protocol detail, the exact request/reply layout, and both bugs: +`docs/nfc_read.md`. Implemented as +`openvixdisklib.nfc_open.NfcDisk.query_allocated_blocks` +(`AllocatedBlock` dataclass) and +`VixDiskLibHandle.query_allocated_blocks`. Validated against the live +ESXi lab: full-disk query, a known-allocated sub-range, and an aligned +empty range all match native VDDK's own output on the same disk. + +## Step 16 — `DDB_GET` (AIO type 11) + +Already partly captured as a side effect of Step 14 (`VixDiskLib_GetInfo` +triggers ~28 `DDB_GET` calls); no new capture was needed, just decoding +the request/reply pairs from that saved log by matching `opId` across +both directions. Confirmed the request's first 8 bytes equal the +`OPEN_FILE` handle from the same capture, and that a "found" reply's +extra is plain ASCII text (`b"lsilogic"`, `b"2088"`, ...), not binary — +matching how a VMDK descriptor's DDB section stores key/value pairs as +text. No padding on the reply extra (unlike Step 15's bitmap), +confirmed by decoding all 28 request/reply pairs from one capture in +sequence without desync. + +Implemented as `NfcDisk.ddb_get(key) -> str | None` and +`NfcDisk.query_full_info() -> DiskInfo` (the 5 keys needed for +`bios_geo`/`adapter_type`/`uuid`), wired into +`VixDiskLibHandle.get_info` in place of the OPEN_FILE-only version +from Step 14 — `get_info` now matches real VDDK's `VixDiskLib_GetInfo` +completely, including paying the same round-trip cost. Full layout: +`docs/nfc_open.md`. Validated against the live ESXi lab: matches +native VDDK's `GetInfo` output on the same disk exactly. + +## Note — `NFC_DELTA_DISK` needed no protocol work either + +Investigated the backlog item "`NFC_DELTA_DISK` (reading directly +from a snapshot chain)" expecting a distinct wire message or OPEN_FILE +variant, similar to the CBT/`QueryAllocatedBlocks` split earlier. +`strings` on `libvixDiskLib.so` found `NFC_DELTA_DISK` is a **file-type +value** (like `NFC_DISK`), used by an internal VDDK client-side +heuristic ("`"%s" would probably benefit from bitmap copying, so +overriding file type to NFC_DELTA_DISK`") — a VMFS-only optimization +for very sparse redo logs, skipped entirely on NFS per an adjacent +string, and never observed to trigger in this lab's captures (no such +log line, `strings`-confirmed heuristic notwithstanding). + +Verified end-to-end that reading, writing, and `query_allocated_blocks` +against an actual post-snapshot delta file all already work correctly +with the existing NFC_DISK-only implementation — no code change +needed. The one real finding from this investigation was a gotcha, not +a gap: an initial test that wrote a sector and immediately queried +allocated blocks *on the same still-open write handle* reported the +write as unallocated; closing the handle first (or opening a separate +one) reported it correctly. Reproduced identically against **native +VDDK** on the same delta file (two-process capture, since loading +native VDDK in the same process as pyVmomi segfaults on this host's +OpenSSL — see `docs/ssl_hook.md`'s Limits section), so this is real +server/VMFS behavior, not specific to either client. Documented as a +`query_allocated_blocks` caveat in `docs/nfc_read.md`. + +## Note — encryption needed no NFC work either + +Investigated the backlog item "encrypted disks" expecting an extra +key-provisioning VIM call, per `"Cannot push crypto key to host"` and +similar strings in `libvddkVimAccess.so`. Building a lab to test this +needed vCenter (a bare ESXi host cannot do VM encryption at all — +`docs/encryption_lab_setup.md`). With a real encrypted disk available, +the same SSL-hook capture used throughout this doc showed **no** +`AddKey`/`ConfigureCryptoKey`/`EnableCrypto`/`PrepareCrypto` calls +anywhere in the wire capture, and `openvixdisklib` — no +encryption-specific code — round-tripped a known pattern through +completely ordinary NFC read/write while the bytes on disk (confirmed +via the raw `.vmdk` descriptor and the flat file's bytes) were genuine +ciphertext. ESXi's storage stack handles disk encryption transparently +below the NFC layer whenever the serving host already holds the key — +initially the only case this single-host lab could produce. Once a +second host existed (built for the host-switch investigation below), +went back and closed the one remaining gap: relocated the encrypted VM +(cold, compute+disk) to a host that had never held its key at all, and +reading its disk from there just worked, decrypting correctly, with +vCenter having pushed the key automatically as part of the migration +— nothing for OpenVixDiskLib to implement. Full investigation: +`docs/encryption.md`. + ## What to write down After a stage works: -| Document | Contents | -| --------------------------------------- | --------------------------------------------- | -| `docs/nfc_auth.md` | Ticket SOAP + authd wire format | -| `docs/nfc_open.md` | Classic NFC + AIO open | -| `docs/nfc_read.md` | AIO IO / `VixDiskLib_Read` | -| `docs/nfc_write.md` | AIO IO / `VixDiskLib_Write` | -| `docs/ssl_hook.md` | Capture tool only | +| Document | Contents | +| --------------------------------------- | ----------------------------------------------- | +| `docs/nfc_auth.md` | Ticket SOAP + authd wire format | +| `docs/nfc_open.md` | Classic NFC + AIO open | +| `docs/nfc_read.md` | AIO IO / `VixDiskLib_Read` | +| `docs/nfc_write.md` | AIO IO / `VixDiskLib_Write` | +| `docs/ssl_hook.md` | Capture tool only | +| `docs/cbt.md` | Changed Block Tracking (VIM API, no NFC) | +| `docs/encryption.md` | Encrypted disks (transparent below NFC) | +| `docs/encryption_lab_setup.md` | Building a vCenter/NKP lab from bare ESXi | | `docs/reverse_engineering_procedure.md` | This procedure (update when the method changes) | Keep the hook and ctypes driver under `/tmp`. They are not part of @@ -404,8 +582,5 @@ OpenVixDiskLib. Not yet reversed, same loop as above: -- `DDB_GET` / disk geometry, zlib/skipz compression, encrypted disks -- `NFC_DELTA_DISK`, CBT / `QueryAllocatedBlocks` -- `VixDiskLib_GetInfo` capacity +- zlib/skipz compression - Host-switch AIO messages -- Direct ESXi `ha-nfc` without vCenter `vpxa-nfc` diff --git a/docs/ssl_hook.md b/docs/ssl_hook.md index cca3a43..6d52ceb 100644 --- a/docs/ssl_hook.md +++ b/docs/ssl_hook.md @@ -143,3 +143,15 @@ skipped. that is done offline on the hex log. - It must not ship in OpenVixDiskLib. Keep it out of the library path used by `openvixdisklib/nfc_auth.py`. +- `ctypes.CDLL` on `libvixDiskLib.so` **in the same process as + pyVmomi** can segfault, at least on this lab's Python/glibc build: + VDDK's bundled OpenSSL and the system OpenSSL pyVmomi already loaded + (for its own HTTPS) collide. Symptom: `Segmentation fault (core + dumped)`, no Python traceback. Split into two separate processes + instead — one doing pyVmomi/setup work, one doing only + `ctypes.CDLL`/native VDDK calls, handing data between them via a + file (see `docs/reverse_engineering_procedure.md`'s note on + `NFC_DELTA_DISK` for an example). This is the same underlying + conflict as `tests/integration/test_vddk.py` / + `test_crosscheck.py` needing `tox -e integration`'s isolated + subprocess env rather than running inside the main pytest process. diff --git a/openvixdisklib/nfc_auth.py b/openvixdisklib/nfc_auth.py index d50744a..184942e 100644 --- a/openvixdisklib/nfc_auth.py +++ b/openvixdisklib/nfc_auth.py @@ -19,8 +19,12 @@ import contextlib import hashlib +import re import socket import ssl +import time +import types +from dataclasses import dataclass from pyVim.connect import Disconnect, SmartConnect from pyVmomi import vim @@ -31,9 +35,11 @@ GetVmodlType, ) -NFC_SERVICE_MOID = "nfcService" AUTHD_DEFAULT_PORT = 902 _NFC_TYPES_REGISTERED = False +_NFC_SERVICE_MOID_RE = re.compile(r"]*>([^<]+)") +_TASK_POLL_S = 0.5 +_TASK_TIMEOUT_S = 300 def _ssl_client_context(verify: bool = True) -> ssl.SSLContext: @@ -126,6 +132,43 @@ def _register_nfc_types() -> None: _NFC_TYPES_REGISTERED = True +def _nfc_service_moid(si: vim.ServiceInstance) -> str: + """Return the NfcService moref via the internal ``RetrieveInternalContent`` call. + + vCenter and a bare ESXi host disagree on this moref (``nfcService`` vs. + ``ha-nfc-service``); VDDK resolves it dynamically instead of assuming + vCenter's name, which is why OpenVixDiskLib must too. The response also + carries ~20 other undocumented managed-object refs (agent manager, disk + manager, and so on) that aren't worth registering with pyVmomi's type + system just to read one field, so the call is issued as raw SOAP over + the existing authenticated connection and only ``nfcService`` is pulled + out of the XML. + """ + stub = si._stub + info = types.SimpleNamespace( + wsdlName="RetrieveInternalContent", version=stub.version, params=() + ) + request = stub.SerializeRequest(si, info, ()) + headers = { + "Cookie": stub.cookie, + "SOAPAction": stub.versionId, + "Content-Type": "text/xml; charset=utf-8", + } + conn = stub.GetConnection() + try: + conn.request("POST", stub.path, request, headers) + response = conn.getresponse() + body = response.read().decode("utf-8") + finally: + stub.ReturnConnection(conn) + match = _NFC_SERVICE_MOID_RE.search(body) + if response.status != 200 or not match: + raise RuntimeError( + f"RetrieveInternalContent (status {response.status}) had no nfcService moref" + ) + return match.group(1) + + def nfc_service(si: vim.ServiceInstance) -> vim.NfcService: """Return the vCenter/ESXi NfcService managed object on ``si``'s SOAP stub. @@ -134,7 +177,7 @@ def nfc_service(si: vim.ServiceInstance) -> vim.NfcService: """ _register_nfc_types() nfc_cls = GetVmodlType("vim.NfcService") - return nfc_cls(NFC_SERVICE_MOID, si._stub) + return nfc_cls(_nfc_service_moid(si), si._stub) def _vim_preferred_api_versions() -> list[str]: @@ -339,6 +382,7 @@ def connect_authd( allow_untrusted: bool = False, timeout: float = 30.0, nfc_ssl: bool = True, + fallback_host: str | None = None, ) -> ssl.SSLSocket: """Complete the ESXi authd handshake using an NFC HostServiceTicket. @@ -361,8 +405,12 @@ def connect_authd( timeout: Socket timeout in seconds. nfc_ssl: When True (the default), PROXY to the NFCSSL service used by nbdssl. Pass False for plaintext NFC (nbd). + fallback_host: Host to dial when ``ticket.host`` is unset. A ticket + issued directly by a bare ESXi host (no vCenter) omits ``host`` + entirely, since the authd endpoint is that same host; pass the + VIM connection's host in that case. """ - host = ticket.host + host = ticket.host or fallback_host port = ticket.port or AUTHD_DEFAULT_PORT raw = socket.create_connection((host, port), timeout=timeout) try: @@ -487,9 +535,112 @@ def authenticate( read_only=read_only, ) authd_sock = connect_authd( - ticket, allow_untrusted=allow_untrusted, nfc_ssl=nfc_ssl + ticket, allow_untrusted=allow_untrusted, nfc_ssl=nfc_ssl, fallback_host=host ) except Exception: Disconnect(si) raise return NfcAuthSession(si, ticket, authd_sock, nfc_ssl=nfc_ssl) + + +# --- Changed Block Tracking (CBT) --- +# +# VixDiskLib does not expose CBT itself: ``VixDiskLib_QueryAllocatedBlocks`` +# reports which blocks are allocated (non-sparse) within a single +# NFC-opened disk, not which byte ranges changed between two points in +# time. Real backup tools get that from vSphere's public +# ``VirtualMachine.QueryChangedDiskAreas`` VIM call instead, used +# alongside VDDK/NFC reads. The helpers below are thin wrappers around +# that public pyVmomi call — no NFC reverse engineering was needed for +# them. See ``docs/cbt.md``. + + +@dataclass(frozen=True, slots=True) +class ChangedExtent: + """One changed byte range, as returned by ``QueryChangedDiskAreas``.""" + + start: int + length: int + + +@dataclass(frozen=True, slots=True) +class ChangedDiskAreas: + """Result of ``QueryChangedDiskAreas``, converted to plain dataclasses.""" + + start_offset: int + length: int + changed_areas: tuple[ChangedExtent, ...] + + +def _wait_for_cbt_task(task: vim.Task): + deadline = time.monotonic() + _TASK_TIMEOUT_S + while task.info.state in (vim.TaskInfo.State.running, vim.TaskInfo.State.queued): + if time.monotonic() > deadline: + raise TimeoutError(f"timed out waiting for vSphere task {task}") + time.sleep(_TASK_POLL_S) + if task.info.state != vim.TaskInfo.State.success: + raise RuntimeError(f"vSphere task failed: {task.info.error}") + return task.info.result + + +def enable_change_tracking(vm: vim.VirtualMachine) -> None: + """Enable CBT on ``vm``. + + Takes effect for writes from this point forward; it does not + retroactively track earlier changes. A ``changeId`` for a disk only + becomes available after the next snapshot or power cycle once this + is set. + """ + spec = vim.vm.ConfigSpec(changeTrackingEnabled=True) + _wait_for_cbt_task(vm.ReconfigVM_Task(spec=spec)) + + +def disk_change_id(vm: vim.VirtualMachine, device_key: int) -> str: + """Return the current ``changeId`` for the disk with ``device_key``. + + Requires CBT to be enabled and at least one snapshot (or power + cycle) to have happened since. Raises ``ValueError`` if the disk + isn't found or has no ``changeId`` yet (CBT not active for it). + """ + for device in vm.config.hardware.device: + if isinstance(device, vim.vm.device.VirtualDisk) and device.key == device_key: + change_id = getattr(device.backing, "changeId", None) + if not change_id: + raise ValueError( + f"disk {device_key} on {vm._moId} has no changeId yet " + "(enable CBT and take a snapshot first)" + ) + return change_id + raise ValueError(f"no VirtualDisk with device key {device_key} on {vm._moId}") + + +def query_changed_disk_areas( + vm: vim.VirtualMachine, + snapshot: vim.vm.Snapshot, + device_key: int, + change_id: str, + start_offset: int = 0, +) -> ChangedDiskAreas: + """Return byte ranges changed since ``change_id``, up to ``snapshot``. + + Thin wrapper around the public + ``VirtualMachine.QueryChangedDiskAreas`` VIM call. ``change_id`` is + the value from an earlier ``disk_change_id()`` call (or ``"*"`` for + the entire disk, e.g. for an initial full backup). Extents are + 64 KiB-aligned in this lab's observations, but that granularity is + server-defined and not part of this function's contract. + """ + result = vm.QueryChangedDiskAreas( + snapshot=snapshot, + deviceKey=device_key, + startOffset=start_offset, + changeId=change_id, + ) + return ChangedDiskAreas( + start_offset=result.startOffset, + length=result.length, + changed_areas=tuple( + ChangedExtent(start=extent.start, length=extent.length) + for extent in result.changedArea + ), + ) diff --git a/openvixdisklib/nfc_open.py b/openvixdisklib/nfc_open.py index 1b25bd5..e567e7a 100644 --- a/openvixdisklib/nfc_open.py +++ b/openvixdisklib/nfc_open.py @@ -64,8 +64,14 @@ NFC_AIO_MSG_IO = 7 NFC_AIO_MSG_SET_SOCK_OPTS = 9 NFC_AIO_MSG_DDB_GET = 11 +NFC_AIO_MSG_QUERY_ALLOCATED_BLOCKS = 13 NFC_AIO_MSG_SET_RES_POOL = 22 +# Chunk size used in this project's capture/validation of +# query_allocated_blocks (128 sectors = 64 KiB); not a documented VDDK +# default, just a convenient granularity that worked in this lab. +NFC_QUERY_ALLOCATED_BLOCKS_CHUNK_SECTORS = 128 + # Open-file body: file type NFC_DISK. 0x1e is what VDDK sends for # VIXDISKLIB_FLAG_OPEN_READ_ONLY; writable opens clear bit 0x04 (0x1a). NFC_DISK = 2 @@ -82,6 +88,45 @@ NFC_COMPRESSION_FASTLZ = 2 +@dataclass(frozen=True, slots=True) +class DiskGeometry: + """CHS geometry, matching VDDK's ``VixDiskLibGeometry``.""" + + cylinders: int + heads: int + sectors: int + + +@dataclass(frozen=True, slots=True) +class AllocatedBlock: + """One allocated run, matching VDDK's ``VixDiskLibBlock`` (sectors).""" + + offset: int + length: int + + +@dataclass(frozen=True, slots=True) +class DiskInfo: + """Matches VDDK's ``VixDiskLibInfo``. + + ``phys_geo`` and ``capacity_sectors`` are read directly off + OPEN_FILE (offsets 40/44/48 and 28 respectively) — free, no extra + NFC round trip. ``bios_geo``, ``adapter_type``, and ``uuid`` come + from ``DDB_GET`` (see ``NfcDisk.ddb_get`` / ``query_full_info``, + ``docs/nfc_open.md``): each is a real round trip, matching what + real VDDK's ``VixDiskLib_GetInfo`` does. ``bios_geo`` defaults to + all zeros and ``adapter_type``/``uuid`` to ``None`` when the disk + has no snapshots or predates that DDB key (VDDK does the same for + a missing key). + """ + + capacity_sectors: int + phys_geo: DiskGeometry + bios_geo: DiskGeometry = DiskGeometry(cylinders=0, heads=0, sectors=0) + adapter_type: str | None = None + uuid: str | None = None + + @dataclass(frozen=True, slots=True) class ReadFragment: """One NFC AIO extra in a packed skip-decompression ``buf``. @@ -260,6 +305,7 @@ def __init__( compression: int = NFC_COMPRESSION_NONE, aio_buffer_size: int = NFC_AIO_BUFFER_SIZE, aio_buffer_count: int = NFC_AIO_BUFFER_COUNT, + info: DiskInfo | None = None, ) -> None: """Wrap an AIO session that already has ``path`` open. @@ -275,6 +321,8 @@ def __init__( at most this large. aio_buffer_count: OPEN_SESSION buffer pool count (default ``NFC_AIO_BUFFER_COUNT``). + info: Capacity/geometry from the OPEN_FILE reply. ``None`` + before the reply arrives. """ self._sock = sock self._op_id = 0 @@ -284,6 +332,7 @@ def __init__( self.compression = compression self.aio_buffer_size = aio_buffer_size self.aio_buffer_count = aio_buffer_count + self.info = info self._closed = False def _next_op_id(self) -> int: @@ -516,6 +565,150 @@ def write(self, start_sector: int, num_sectors: int, data: bytes) -> None: f"expected type={NFC_AIO_MSG_IO} opId={op_id}" ) + def query_allocated_blocks( + self, + start_sector: int, + num_sectors: int, + chunk_size_sectors: int = NFC_QUERY_ALLOCATED_BLOCKS_CHUNK_SECTORS, + ) -> tuple[AllocatedBlock, ...]: + """Return allocated (non-sparse) runs. Matches ``VixDiskLib_QueryAllocatedBlocks``. + + Captured from VDDK: a 48-byte request:: + + uint64 handle + uint64 reserved (0) + uint64 chunk_size_bytes (chunk_size_sectors * sector_size) + uint64 start_offset_bytes (start_sector * sector_size) + uint64 chunk_count (num_sectors // chunk_size_sectors) + uint64 reserved (0) + + (``start_offset_bytes`` and the first ``reserved`` field are + easy to swap — both are 0 in a ``start_sector=0`` capture, + which is what an earlier draft of this method got wrong; a + second capture with a non-zero ``start_sector`` was needed to + tell them apart.) + + The reply echoes a 48-byte body whose offset 32 carries the + same chunk count back, followed by a bitmap extra — one bit + per chunk, LSB-first, set when that chunk contains allocated + data, padded up to a **4-byte boundary** (``ceil(chunk_count / 8)`` + alone under-reads and desyncs the connection whenever that + raw byte count isn't already a multiple of 4). + This mirrors VDDK's own client-side behavior of run-length + merging contiguous set bits into ``VixDiskLibBlock`` entries + (offset/length here are in **sectors**, matching the public + VDDK struct, unlike the bytes used on the wire). See + ``docs/nfc_read.md``. + + Args: + start_sector: Sector offset from the start of the disk; + must be a multiple of ``chunk_size_sectors`` (the + server returns an AIO error otherwise). + num_sectors: Number of sectors to query; must be a multiple + of ``chunk_size_sectors``. + chunk_size_sectors: Minimum run granularity, in sectors. + """ + if num_sectors % chunk_size_sectors != 0: + raise ValueError("num_sectors must be a multiple of chunk_size_sectors") + if start_sector % chunk_size_sectors != 0: + raise ValueError("start_sector must be a multiple of chunk_size_sectors") + chunk_count = num_sectors // chunk_size_sectors + bitmap_bytes = -(-((chunk_count + 7) // 8) // 4) * 4 + request = struct.pack( + " str | None: + """Return a VMDK descriptor DDB value, or ``None`` if unset. + + Captured from VDDK: request is a 16-byte fixed payload plus the + key name as a raw ASCII extra (no NUL terminator, not counted + in ``size``, same convention as ``OPEN_FILE``'s path):: + + uint64 handle + uint32 key_name_length + uint32 reserved (0) + + + Reply is 16 bytes plus a value extra, **not** padded (unlike + ``QueryAllocatedBlocks``'s bitmap):: + + 96 bits reserved/unused (always zero in this lab) + uint32 value_length (0 = key not found) + + + Values are ASCII text even for keys that sound numeric + (``geometry.cylinders`` comes back as the bytes ``b"2088"``, + not a binary int) — this matches how a VMDK descriptor file's + DDB (disk database) section stores keys as plain text + ``ddb. = ""`` lines. See ``docs/nfc_open.md``. + + Args: + key: DDB key name without the ``ddb.`` prefix (for example + ``"adapterType"``, ``"uuid"``, ``"geometry.cylinders"``). + """ + key_bytes = key.encode("ascii") + request = struct.pack(" DiskInfo: + """Return a ``DiskInfo`` with ``bios_geo``/``adapter_type``/``uuid`` filled in. + + ``self.info`` (from OPEN_FILE) already has ``capacity_sectors`` + and ``phys_geo`` for free; this issues 5 ``DDB_GET`` round trips + for the rest, matching what real VDDK's ``VixDiskLib_GetInfo`` + does on every call. DDB values are ASCII text; geometry fields + are parsed as decimal integers, and any missing key falls back + to ``DiskInfo``'s defaults (matches VDDK: a disk with no + snapshots, or from before this DDB key existed, has none of + these set). + """ + assert self.info is not None + bios_cylinders = self.ddb_get("geometry.biosCylinders") + bios_heads = self.ddb_get("geometry.biosHeads") + bios_sectors = self.ddb_get("geometry.biosSectors") + bios_geo = DiskGeometry( + cylinders=int(bios_cylinders) if bios_cylinders else 0, + heads=int(bios_heads) if bios_heads else 0, + sectors=int(bios_sectors) if bios_sectors else 0, + ) + return DiskInfo( + capacity_sectors=self.info.capacity_sectors, + phys_geo=self.info.phys_geo, + bios_geo=bios_geo, + adapter_type=self.ddb_get("adapterType"), + uuid=self.ddb_get("uuid"), + ) + def close(self) -> None: """Close the VMDK, the AIO session, and the classic NFC session.""" if self._closed: @@ -591,16 +784,51 @@ def _aio_prepare(disk: NfcDisk) -> None: disk._aio_roundtrip(NFC_AIO_MSG_SET_RES_POOL, struct.pack(" tuple[int, int]: - if len(body) < 40: +def _decode_allocated_bitmap( + bitmap: bytes, chunk_count: int, start_sector: int, chunk_size_sectors: int +) -> tuple[AllocatedBlock, ...]: + """Run-length-merge a QueryAllocatedBlocks bitmap into ``AllocatedBlock``s. + + ``bitmap`` is one bit per chunk, LSB-first (bit 0 of byte 0 is + chunk 0), possibly longer than strictly needed for padding; only + the first ``chunk_count`` bits are read. + """ + blocks = [] + run_start = None + for chunk_idx in range(chunk_count): + allocated = (bitmap[chunk_idx // 8] >> (chunk_idx % 8)) & 1 + if allocated and run_start is None: + run_start = chunk_idx + elif not allocated and run_start is not None: + blocks.append((run_start, chunk_idx - run_start)) + run_start = None + if run_start is not None: + blocks.append((run_start, chunk_count - run_start)) + return tuple( + AllocatedBlock( + offset=start_sector + run_chunk * chunk_size_sectors, + length=run_len * chunk_size_sectors, + ) + for run_chunk, run_len in blocks + ) + + +def _parse_open_reply(body: bytes) -> tuple[int, int, DiskInfo]: + if len(body) < 52: raise NfcProtocolError(f"OPEN_FILE reply too short: {len(body)}") handle, file_type, _flags = struct.unpack_from(" nfc_open.DiskInfo: + """Return disk info. Matches ``VixDiskLib_GetInfo``. + + ``capacity_sectors``/``phys_geo`` are free (already in the + ``OPEN_FILE`` reply from ``open()``); ``bios_geo``/ + ``adapter_type``/``uuid`` cost 5 ``DDB_GET`` round trips, same + as real VDDK pays on every ``GetInfo`` call. See + ``docs/nfc_open.md``. + """ + return disk_handle.disk.query_full_info() + + def query_allocated_blocks( + self, + disk_handle: _DiskHandle, + start_sector: int, + num_sectors: int, + chunk_size_sectors: int = nfc_open.NFC_QUERY_ALLOCATED_BLOCKS_CHUNK_SECTORS, + ) -> tuple[nfc_open.AllocatedBlock, ...]: + """Return allocated runs. Matches ``VixDiskLib_QueryAllocatedBlocks``. + + Args: + disk_handle: Handle from ``open()``. + start_sector: Sector offset from the start of the disk. + num_sectors: Number of sectors to query; must be a multiple + of ``chunk_size_sectors``. + chunk_size_sectors: Minimum run granularity, in sectors. + """ + return disk_handle.disk.query_allocated_blocks( + start_sector, num_sectors, chunk_size_sectors + ) + def read( self, disk_handle: _DiskHandle, diff --git a/tests/integration/test_cbt.py b/tests/integration/test_cbt.py new file mode 100644 index 0000000..12243ea --- /dev/null +++ b/tests/integration/test_cbt.py @@ -0,0 +1,129 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Exercise Changed Block Tracking against the lab. + +Unlike VDDK/NFC features, these VIM-level calls (``nfc_auth. +enable_change_tracking`` / ``disk_change_id`` / ``query_changed_disk_areas``) +are thin wrappers around public pyVmomi; see ``docs/cbt.md``. +""" + +from pyVim.connect import Disconnect +from pyVmomi import vim + +from openvixdisklib import nfc_auth +from openvixdisklib import openvixdisklib as vixdisklib +from tests.integration.base import SECTOR_SIZE, LabEnv, _connect_vim, _wait_for_task, pattern_bytes + + +def _disk_device(vm: vim.VirtualMachine) -> vim.vm.device.VirtualDisk: + """Return the lab VM's first virtual disk device.""" + for device in vm.config.hardware.device: + if isinstance(device, vim.vm.device.VirtualDisk): + return device + raise AssertionError(f"{vm._moId} has no virtual disk") + + +def _disable_change_tracking(vm: vim.VirtualMachine) -> None: + spec = vim.vm.ConfigSpec(changeTrackingEnabled=False) + _wait_for_task(vm.ReconfigVM_Task(spec=spec)) + + +class TestCbt: + def test_full_cbt_cycle(self, lab: LabEnv) -> None: + """Enable CBT, write a known sector, and see it in a changed-areas query.""" + si = _connect_vim( + lab.host, lab.username, lab.password, lab.port, lab.thumbprint, lab.allow_untrusted + ) + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + nfc_auth.enable_change_tracking(vm) + vm.Reload() + assert vm.config.changeTrackingEnabled is True + + device_key = _disk_device(vm).key + + snap1 = _wait_for_task( + vm.CreateSnapshot_Task("cbt-baseline", "", False, False) + ) + vm.Reload() + change_id_1 = nfc_auth.disk_change_id(vm, device_key) + assert change_id_1 + + write_sector = 7777 + written = pattern_bytes(SECTOR_SIZE, b"CBT-TEST") + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted} + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, _disk_device(vm).backing.fileName, flags=0) as disk, + ): + buf = vixdisklib.get_buffer(SECTOR_SIZE) + buf[:SECTOR_SIZE] = written + handle.write(disk, write_sector, 1, buf) + + snap2 = _wait_for_task( + vm.CreateSnapshot_Task("cbt-after-write", "", False, False) + ) + vm.Reload() + + result = nfc_auth.query_changed_disk_areas( + vm, snap2, device_key, change_id_1 + ) + write_byte = write_sector * SECTOR_SIZE + assert any( + extent.start <= write_byte < extent.start + extent.length + for extent in result.changed_areas + ), f"sector {write_sector} not covered by any reported extent" + finally: + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + if vm.snapshot is not None: + _wait_for_task(vm.RemoveAllSnapshots_Task()) + _disable_change_tracking(vm) + finally: + Disconnect(si) + + def test_query_changed_disk_areas_wildcard_change_id(self, lab: LabEnv) -> None: + """changeId='*' (initial full backup) reports allocated regions. + + Not the full sparse virtual capacity -- see docs/cbt.md. + """ + si = _connect_vim( + lab.host, lab.username, lab.password, lab.port, lab.thumbprint, lab.allow_untrusted + ) + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + nfc_auth.enable_change_tracking(vm) + vm.Reload() + + device_key = _disk_device(vm).key + capacity_bytes = _disk_device(vm).capacityInKB * 1024 + + snap = _wait_for_task( + vm.CreateSnapshot_Task("cbt-wildcard", "", False, False) + ) + vm.Reload() + + result = nfc_auth.query_changed_disk_areas(vm, snap, device_key, "*") + # changeId="*" reports allocated (backed) regions, not the full + # sparse virtual capacity -- this lab disk is thin-provisioned + # and shared across the test session, so exactly how much is + # allocated depends on what earlier tests wrote. Only the + # length field (declared virtual capacity) is a fixed value; + # changed_areas is just asserted sane (non-empty, in bounds). + assert result.length == capacity_bytes + assert result.changed_areas + for extent in result.changed_areas: + assert extent.start >= 0 + assert extent.start + extent.length <= capacity_bytes + finally: + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + if vm.snapshot is not None: + _wait_for_task(vm.RemoveAllSnapshots_Task()) + _disable_change_tracking(vm) + finally: + Disconnect(si) diff --git a/tests/integration/test_encryption.py b/tests/integration/test_encryption.py new file mode 100644 index 0000000..02d5466 --- /dev/null +++ b/tests/integration/test_encryption.py @@ -0,0 +1,99 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Confirm encrypted VM disks work transparently, with no code changes. + +See ``docs/encryption.md``. Unlike the rest of ``tests/integration``, +this does not use the session-scoped ``lab`` fixture: creating and +encrypting a VM needs a vCenter with a Native Key Provider and a +manual storage-policy assignment step (``docs/encryption_lab_setup.md`` +-- the SPBM API needed to automate the last step did not work), so this +test instead points at a pre-existing encrypted VM/disk. It reuses the +vCenter settings from ``.test_config.yaml`` (see ``README.md``) and needs +two extra keys, ``encrypted_vm_moref`` and ``encrypted_disk_path``. It is +skipped if those keys are absent. +""" + +import os +from typing import Any + +import pytest +import yaml + +from openvixdisklib import openvixdisklib as vixdisklib +from tests.integration.base import SECTOR_SIZE, pattern_bytes + +_REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) +_CONFIG_PATH = os.path.join(_REPO_ROOT, ".test_config.yaml") +_ENCRYPTION_CONFIG_KEYS = ("encrypted_vm_moref", "encrypted_disk_path") + + +def _load_encryption_config() -> dict[str, Any]: + if not os.path.isfile(_CONFIG_PATH): + pytest.skip(f"{_CONFIG_PATH} not found; see docs/encryption_lab_setup.md") + with open(_CONFIG_PATH, encoding="utf-8") as config_file: + data = yaml.safe_load(config_file) or {} + missing = [key for key in _ENCRYPTION_CONFIG_KEYS if key not in data] + if missing: + pytest.skip( + "encrypted-disk test needs an already-encrypted VM; " + f"missing .test_config.yaml keys: {', '.join(missing)} " + "(see docs/encryption_lab_setup.md)" + ) + return data + + +class TestEncryption: + def test_read_write_encrypted_disk(self) -> None: + """Write a known pattern to an encrypted disk and read it back. + + No encryption-specific code path exists in openvixdisklib -- + this is exactly the same connect/open/write/read sequence as + any other disk. See docs/encryption.md for the wire-level proof + that the bytes are genuinely ciphertext at rest. + """ + cfg = _load_encryption_config() + handle = vixdisklib.VixDiskLibHandle( + vixdisklib_compatibility_version="8.0", config_path=None + ) + expected = pattern_bytes(SECTOR_SIZE, b"OVDL-CRYPTO-") + + write_buf = vixdisklib.get_buffer(SECTOR_SIZE) + write_buf[:SECTOR_SIZE] = expected + with ( + handle.connect( + server_name=cfg["host"], + port=int(cfg.get("port", 443)), + thumbprint=None, + username=cfg["username"], + password=cfg["password"], + vmx_spec=str(cfg["encrypted_vm_moref"]), + read_only=False, + allow_untrusted=bool(cfg.get("allow_untrusted", True)), + ) as conn, + handle.open(conn, str(cfg["encrypted_disk_path"]), flags=0) as disk, + ): + handle.write(disk, 0, 1, write_buf) + + read_buf = vixdisklib.get_buffer(SECTOR_SIZE) + read_buf[:SECTOR_SIZE] = b"\xa5" * SECTOR_SIZE + with ( + handle.connect( + server_name=cfg["host"], + port=int(cfg.get("port", 443)), + thumbprint=None, + username=cfg["username"], + password=cfg["password"], + vmx_spec=str(cfg["encrypted_vm_moref"]), + read_only=True, + allow_untrusted=bool(cfg.get("allow_untrusted", True)), + ) as conn, + handle.open( + conn, + str(cfg["encrypted_disk_path"]), + flags=vixdisklib.VIXDISKLIB_FLAG_OPEN_READ_ONLY, + ) as disk, + ): + handle.read(disk, 0, 1, read_buf) + + assert read_buf.raw[:SECTOR_SIZE] == expected diff --git a/tests/integration/test_nfc_auth.py b/tests/integration/test_nfc_auth.py index e074f1e..d6c32e1 100644 --- a/tests/integration/test_nfc_auth.py +++ b/tests/integration/test_nfc_auth.py @@ -11,7 +11,11 @@ def test_authd_handshake_completes(self, lab: LabEnv) -> None: """Complete VIM login and authd PROXY through ``200 Connect``.""" with lab.authenticate() as session: ticket = session.ticket - assert ticket.host + # ticket.host is unset on a direct-ESXi ticket (no vCenter): + # the authd endpoint is implicitly the host already logged + # into. connect_authd() falls back to that host, so the + # socket's peer address is the reliable check here. + assert session.authd_sock.getpeername()[0] assert ticket.port assert ticket.sessionId assert session.nfc_ssl is True diff --git a/tests/integration/test_nfc_open.py b/tests/integration/test_nfc_open.py index 9d9beb1..08fe813 100644 --- a/tests/integration/test_nfc_open.py +++ b/tests/integration/test_nfc_open.py @@ -6,7 +6,7 @@ import pytest from openvixdisklib import nfc_open -from tests.integration.base import SECTOR_SIZE, LabEnv, pattern_bytes +from tests.integration.base import _DISK_CAPACITY_KB, SECTOR_SIZE, LabEnv, pattern_bytes class TestNfcOpen: @@ -34,3 +34,18 @@ def test_open_disk_and_read_first_sector( got = disk.read(0, 1) assert got is not expected assert got == expected + + def test_open_disk_reports_capacity_and_geometry(self, lab: LabEnv) -> None: + """OPEN_FILE's reply carries capacity and physical geometry (GetInfo).""" + with ( + lab.authenticate() as session, + nfc_open.open_disk(session, lab.disk_path) as disk, + ): + assert disk.info is not None + assert disk.info.capacity_sectors == ( + _DISK_CAPACITY_KB * 1024 // SECTOR_SIZE + ) + geo = disk.info.phys_geo + assert geo.cylinders > 0 + assert geo.heads > 0 + assert geo.sectors > 0 diff --git a/tests/integration/test_openvixdisklib.py b/tests/integration/test_openvixdisklib.py index 6c853d6..6731a26 100644 --- a/tests/integration/test_openvixdisklib.py +++ b/tests/integration/test_openvixdisklib.py @@ -13,6 +13,7 @@ from openvixdisklib import openvixdisklib as vixdisklib from openvixdisklib.openvixdisklib import ReadResult from tests.integration.base import ( + _DISK_CAPACITY_KB, SECTOR_AT_1GB, SECTOR_SIZE, LabEnv, @@ -94,6 +95,58 @@ def test_write_and_read_sector_zero_and_one_gib( handle.read(disk, start, 1, read_buf) assert read_buf.raw[:SECTOR_SIZE] == expected + def test_get_info(self, lab: LabEnv) -> None: + """get_info returns the lab VM's known disk capacity and geometry.""" + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted, "read_only": True} + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open( + conn, lab.disk_path, flags=vixdisklib.VIXDISKLIB_FLAG_OPEN_READ_ONLY + ) as disk, + ): + info = handle.get_info(disk) + assert info.capacity_sectors == _DISK_CAPACITY_KB * 1024 // SECTOR_SIZE + assert info.phys_geo.cylinders > 0 + assert info.phys_geo.heads > 0 + assert info.phys_geo.sectors > 0 + # bios_geo is DDB-derived and unset (all zero) on a disk with + # no snapshots yet, matching VDDK's own default for a missing key. + assert info.bios_geo == vixdisklib.DiskGeometry( + cylinders=0, heads=0, sectors=0 + ) + assert info.adapter_type # non-empty DDB string, e.g. "lsilogic" + assert info.uuid # non-empty DDB string + + def test_query_allocated_blocks(self, lab: LabEnv) -> None: + """A written sector's chunk shows up as an allocated run.""" + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + chunk_size_sectors = 128 + # Chunk-aligned offset away from what other tests in this shared + # session-scoped VM write to (sector 0, SECTOR_AT_1GB, ...). + write_sector = 300 * chunk_size_sectors + write_buf = vixdisklib.get_buffer(SECTOR_SIZE) + write_buf[:SECTOR_SIZE] = pattern_bytes(SECTOR_SIZE, b"OVDL-QAB") + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted} + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, lab.disk_path, flags=0) as disk, + ): + handle.write(disk, write_sector, 1, write_buf) + blocks = handle.query_allocated_blocks( + disk, + start_sector=(write_sector // chunk_size_sectors) * chunk_size_sectors, + num_sectors=chunk_size_sectors, + chunk_size_sectors=chunk_size_sectors, + ) + assert any( + b.offset <= write_sector < b.offset + b.length for b in blocks + ), f"written sector {write_sector} not covered by {blocks}" + def test_read_only_open_snapshot_parent(self, lab: LabEnv) -> None: """Read-only Open uses NfcGetVmFiles, including a snapshot parent path. @@ -162,6 +215,74 @@ def read_sector(path: str) -> bytes: finally: Disconnect(si) + def test_write_and_query_allocated_blocks_on_delta_disk(self, lab: LabEnv) -> None: + """Read/write/query all work directly on a post-snapshot delta file. + + ``NFC_DELTA_DISK`` (an alternate OPEN_FILE file-type value, + per ``strings`` on ``libvixDiskLib.so``) turned out to be an + optional VMFS-only VDDK client optimization, not needed for + correctness — this exercises the plain ``NFC_DISK`` path + against an actual delta file. See + ``docs/reverse_engineering_procedure.md``. + + Also covers the gotcha documented in ``docs/nfc_read.md``: + querying allocated blocks on the *same still-open* handle a + write just went through can see stale (pre-write) data: the + write below is done in its own ``with`` block, closed, before + the separate query. + """ + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + chunk_size_sectors = 128 + write_sector = 400 * chunk_size_sectors + expected = pattern_bytes(SECTOR_SIZE, b"OVDL-DELTA") + write_buf = vixdisklib.get_buffer(SECTOR_SIZE) + write_buf[:SECTOR_SIZE] = expected + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted} + ) + read_buf = vixdisklib.get_buffer(SECTOR_SIZE) + + si = _connect_vim( + lab.host, lab.username, lab.password, lab.port, lab.thumbprint, lab.allow_untrusted + ) + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + _wait_for_task(vm.CreateSnapshot_Task("ovdl-delta", "", False, False)) + delta_path = _virtual_disk_backing(vm).fileName + assert delta_path != lab.disk_path + + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, delta_path, flags=0) as disk, + ): + handle.write(disk, write_sector, 1, write_buf) + + # Fresh handle after close, per the gotcha above. + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, delta_path, flags=0) as disk, + ): + read_buf[:SECTOR_SIZE] = b"\xa5" * SECTOR_SIZE + handle.read(disk, write_sector, 1, read_buf) + assert read_buf.raw[:SECTOR_SIZE] == expected + + blocks = handle.query_allocated_blocks( + disk, + start_sector=(write_sector // chunk_size_sectors) * chunk_size_sectors, + num_sectors=chunk_size_sectors, + chunk_size_sectors=chunk_size_sectors, + ) + assert any( + b.offset <= write_sector < b.offset + b.length for b in blocks + ), f"written sector {write_sector} on delta file not covered by {blocks}" + finally: + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + if vm.snapshot is not None: + _wait_for_task(vm.RemoveAllSnapshots_Task()) + finally: + Disconnect(si) + @pytest.mark.parametrize( "aio_buffer_size, n_sectors, n_fragments", [ diff --git a/tests/unit/test_nfc_auth.py b/tests/unit/test_nfc_auth.py new file mode 100644 index 0000000..20206ae --- /dev/null +++ b/tests/unit/test_nfc_auth.py @@ -0,0 +1,93 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Unit tests for the CBT helpers in ``nfc_auth``.""" + +from unittest import mock + +import pytest +from pyVmomi import vim + +from openvixdisklib import nfc_auth + + +def _virtual_disk(key: int, change_id: str | None = None) -> vim.vm.device.VirtualDisk: + disk = vim.vm.device.VirtualDisk() + disk.key = key + backing = vim.vm.device.VirtualDisk.FlatVer2BackingInfo() + if change_id is not None: + backing.changeId = change_id + disk.backing = backing + return disk + + +def _fake_vm(devices: list) -> mock.Mock: + vm = mock.Mock(_moId="vm-1") + vm.config.hardware.device = devices + return vm + + +class TestDiskChangeId: + def test_returns_change_id_for_matching_device(self) -> None: + """The changeId on the matching VirtualDisk's backing is returned.""" + disk = _virtual_disk(key=2000, change_id="52 aa/1") + vm = _fake_vm([disk]) + assert nfc_auth.disk_change_id(vm, 2000) == "52 aa/1" + + def test_no_matching_device_key_raises(self) -> None: + """A device key not present on the VM raises ValueError.""" + vm = _fake_vm([_virtual_disk(key=2000, change_id="52 aa/1")]) + with pytest.raises(ValueError, match="no VirtualDisk with device key 9999"): + nfc_auth.disk_change_id(vm, 9999) + + def test_empty_change_id_raises(self) -> None: + """A matching disk with no changeId yet (CBT not active) raises.""" + vm = _fake_vm([_virtual_disk(key=2000, change_id=None)]) + with pytest.raises(ValueError, match="has no changeId yet"): + nfc_auth.disk_change_id(vm, 2000) + + +class TestQueryChangedDiskAreas: + def test_converts_result_to_dataclasses(self) -> None: + """QueryChangedDiskAreas's result is converted to plain dataclasses.""" + vm = mock.Mock() + vm.QueryChangedDiskAreas.return_value = mock.Mock( + startOffset=0, + length=10737418240, + changedArea=[ + mock.Mock(start=0, length=65536), + mock.Mock(start=2555904, length=65536), + ], + ) + snapshot = mock.Mock() + + result = nfc_auth.query_changed_disk_areas(vm, snapshot, 2000, "52 aa/1") + + vm.QueryChangedDiskAreas.assert_called_once_with( + snapshot=snapshot, deviceKey=2000, startOffset=0, changeId="52 aa/1" + ) + assert result == nfc_auth.ChangedDiskAreas( + start_offset=0, + length=10737418240, + changed_areas=( + nfc_auth.ChangedExtent(start=0, length=65536), + nfc_auth.ChangedExtent(start=2555904, length=65536), + ), + ) + + def test_passes_through_start_offset_and_wildcard_change_id(self) -> None: + """A non-zero start_offset and change_id='*' are passed through as-is.""" + vm = mock.Mock() + vm.QueryChangedDiskAreas.return_value = mock.Mock( + startOffset=1024, length=0, changedArea=[] + ) + snapshot = mock.Mock() + + result = nfc_auth.query_changed_disk_areas( + vm, snapshot, 2000, "*", start_offset=1024 + ) + + vm.QueryChangedDiskAreas.assert_called_once_with( + snapshot=snapshot, deviceKey=2000, startOffset=1024, changeId="*" + ) + assert result.changed_areas == () diff --git a/tests/unit/test_nfc_open.py b/tests/unit/test_nfc_open.py new file mode 100644 index 0000000..4bf8ad1 --- /dev/null +++ b/tests/unit/test_nfc_open.py @@ -0,0 +1,215 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Unit tests for the OPEN_FILE reply parsing in ``nfc_open``.""" + +import struct + +import pytest + +from openvixdisklib import nfc_open + + +class _FakeSocket: + """A minimal socket stand-in that replays scripted bytes for recv_into.""" + + def __init__(self, replies: bytes) -> None: + self._replies = replies + self.sent: list[bytes] = [] + + def sendall(self, data: bytes) -> None: + self.sent.append(bytes(data)) + + def recv_into(self, buf: memoryview) -> int: + n = min(len(buf), len(self._replies)) + buf[:n] = self._replies[:n] + self._replies = self._replies[n:] + return n + + +def _open_reply_body( + handle: int = 0x1234, + file_type: int = nfc_open.NFC_DISK, + flags: int = nfc_open.NFC_OPEN_FLAGS_READ_ONLY, + capacity_bytes: int = 17179869184, + sector_size: int = 512, + cylinders: int = 2088, + heads: int = 255, + sectors: int = 63, +) -> bytes: + """Build a synthetic 60-byte OPEN_FILE reply payload.""" + body = bytearray(60) + struct.pack_into(" None: + """Contiguous set bits become one run; gaps split into separate ones.""" + # chunks: 1,1,1,1,0,0,0,0,1,1 (10 chunks -> 2 bytes, LSB-first) + bitmap = bytes([0b00001111, 0b00000011]) + blocks = nfc_open._decode_allocated_bitmap( + bitmap, chunk_count=10, start_sector=1000, chunk_size_sectors=128 + ) + assert blocks == ( + nfc_open.AllocatedBlock(offset=1000, length=4 * 128), + nfc_open.AllocatedBlock(offset=1000 + 8 * 128, length=2 * 128), + ) + + def test_all_zero_bitmap_returns_no_blocks(self) -> None: + """A bitmap with no set bits produces an empty result.""" + blocks = nfc_open._decode_allocated_bitmap( + bytes(4), chunk_count=16, start_sector=0, chunk_size_sectors=128 + ) + assert blocks == () + + def test_run_extending_to_the_end_is_closed(self) -> None: + """A run of set bits reaching the last chunk is still reported.""" + # chunks: 0,1,1,1 (4 chunks, 1 byte; only lower nibble meaningful) + bitmap = bytes([0b00001110]) + blocks = nfc_open._decode_allocated_bitmap( + bitmap, chunk_count=4, start_sector=0, chunk_size_sectors=1 + ) + assert blocks == (nfc_open.AllocatedBlock(offset=1, length=3),) + + def test_ignores_bits_beyond_chunk_count(self) -> None: + """Padding bits past chunk_count (from 4-byte reply alignment) are unused.""" + # 2 real chunks (both set) + 2 padding bytes with garbage bits set. + bitmap = bytes([0b00000011, 0xFF, 0xFF, 0xFF]) + blocks = nfc_open._decode_allocated_bitmap( + bitmap, chunk_count=2, start_sector=0, chunk_size_sectors=128 + ) + assert blocks == (nfc_open.AllocatedBlock(offset=0, length=256),) + + +class TestQueryAllocatedBlocksValidation: + def _disk(self) -> nfc_open.NfcDisk: + return nfc_open.NfcDisk(sock=None, path="[ds] a.vmdk", handle=1, sector_size=512) + + def test_num_sectors_not_a_multiple_raises(self) -> None: + with pytest.raises(ValueError, match="num_sectors must be a multiple"): + self._disk().query_allocated_blocks(0, 100, chunk_size_sectors=128) + + def test_start_sector_not_a_multiple_raises(self) -> None: + with pytest.raises(ValueError, match="start_sector must be a multiple"): + self._disk().query_allocated_blocks(100, 128, chunk_size_sectors=128) + + +def _ddb_get_reply(op_id: int, value: bytes | None) -> bytes: + """Build a scripted DDB_GET reply: header + 16-byte body + value extra.""" + value_length = len(value) if value is not None else 0 + body = bytes(12) + struct.pack(" nfc_open.NfcDisk: + return nfc_open.NfcDisk( + sock=_FakeSocket(replies), path="[ds] a.vmdk", handle=0x1234, sector_size=512 + ) + + def test_found_key_returns_decoded_value(self) -> None: + disk = self._disk(_ddb_get_reply(op_id=0, value=b"lsilogic")) + assert disk.ddb_get("adapterType") == "lsilogic" + + def test_missing_key_returns_none(self) -> None: + disk = self._disk(_ddb_get_reply(op_id=0, value=None)) + assert disk.ddb_get("resumeConsolidateSector") is None + + def test_sends_handle_and_key_length_in_request(self) -> None: + sock = _FakeSocket(_ddb_get_reply(op_id=0, value=b"63")) + disk = nfc_open.NfcDisk( + sock=sock, path="[ds] a.vmdk", handle=0x1234, sector_size=512 + ) + disk.ddb_get("geometry.sectors") + (sent,) = sock.sent + # header(16) + handle(8) + key_len(4) + reserved(4) + key bytes + handle, key_len, reserved = struct.unpack_from(" nfc_open.NfcDisk: + # query_full_info calls ddb_get for biosCylinders, biosHeads, + # biosSectors, adapterType, uuid, in that order. + keys = [ + "geometry.biosCylinders", + "geometry.biosHeads", + "geometry.biosSectors", + "adapterType", + "uuid", + ] + replies = b"".join( + _ddb_get_reply(op_id=i, value=values.get(k)) for i, k in enumerate(keys) + ) + disk = nfc_open.NfcDisk( + sock=_FakeSocket(replies), path="[ds] a.vmdk", handle=1, sector_size=512 + ) + disk.info = nfc_open.DiskInfo( + capacity_sectors=1024, + phys_geo=nfc_open.DiskGeometry(cylinders=10, heads=20, sectors=30), + ) + return disk + + def test_combines_open_file_info_with_ddb_values(self) -> None: + disk = self._disk_with_replies( + { + "geometry.biosCylinders": b"100", + "geometry.biosHeads": b"200", + "geometry.biosSectors": b"63", + "adapterType": b"lsilogic", + "uuid": b"some-uuid", + } + ) + info = disk.query_full_info() + assert info.capacity_sectors == 1024 + assert info.phys_geo == nfc_open.DiskGeometry(cylinders=10, heads=20, sectors=30) + assert info.bios_geo == nfc_open.DiskGeometry(cylinders=100, heads=200, sectors=63) + assert info.adapter_type == "lsilogic" + assert info.uuid == "some-uuid" + + def test_missing_ddb_keys_fall_back_to_defaults(self) -> None: + disk = self._disk_with_replies({}) + info = disk.query_full_info() + assert info.bios_geo == nfc_open.DiskGeometry(cylinders=0, heads=0, sectors=0) + assert info.adapter_type is None + assert info.uuid is None + + +class TestParseOpenReply: + def test_parses_handle_capacity_and_geometry(self) -> None: + """Capacity (offset 28, bytes) and physGeo (40/44/48) are extracted.""" + body = _open_reply_body() + handle, sector_size, info = nfc_open._parse_open_reply(body) + assert handle == 0x1234 + assert sector_size == 512 + assert info.capacity_sectors == 17179869184 // 512 + assert info.phys_geo == nfc_open.DiskGeometry( + cylinders=2088, heads=255, sectors=63 + ) + + def test_zero_sector_size_falls_back_and_still_divides_capacity(self) -> None: + """A zero sector_size falls back to NFC_SECTOR_SIZE for both uses.""" + body = _open_reply_body(sector_size=0, capacity_bytes=1024 * 512) + _handle, sector_size, info = nfc_open._parse_open_reply(body) + assert sector_size == nfc_open.NFC_SECTOR_SIZE + assert info.capacity_sectors == (1024 * 512) // nfc_open.NFC_SECTOR_SIZE + + def test_wrong_file_type_raises(self) -> None: + """A non-NFC_DISK file type is rejected.""" + body = _open_reply_body(file_type=99) + with pytest.raises(nfc_open.NfcProtocolError, match="file type 99"): + nfc_open._parse_open_reply(body) + + def test_short_body_raises(self) -> None: + """A reply shorter than the physGeo fields is rejected.""" + with pytest.raises(nfc_open.NfcProtocolError, match="too short"): + nfc_open._parse_open_reply(bytes(40))