From c7610fbd970c451d060554d538e39ba6b9e419df Mon Sep 17 00:00:00 2001 From: doccaz Date: Sat, 19 Sep 2026 10:05:00 -0300 Subject: [PATCH] Add direct-ESXi support, GetInfo, QueryAllocatedBlocks, DDB_GET, and CBT Reverse-engineered and implemented against a live standalone ESXi 8.0.3 host (no vCenter), closing most of the gap versus the proprietary VDDK: - Direct ESXi (no vCenter) connectivity: nfc_service() previously hardcoded the NfcService moref as "nfcService" (vCenter's name), which fails on bare ESXi (moref is "ha-nfc-service" there). Now resolved dynamically via RetrieveInternalContent, same as VDDK itself does. Also fixes connect_authd() for tickets that omit `host` (implicit on a direct-ESXi ticket). - VixDiskLib_GetInfo: capacity and physical geometry come free from the OPEN_FILE reply (offsets already in the wire frame). biosGeo, adapterType, and uuid are fetched via DDB_GET, matching real VDDK's behavior and cost exactly. - DDB_GET (VMDK descriptor lookups): generic key/value NFC message, values are ASCII text on the wire (not binary), matching how a VMDK descriptor's DDB section is stored. - VixDiskLib_QueryAllocatedBlocks: allocated-block bitmap query. Verified against a live disk to exactly match native VDDK's output, including two non-obvious wire details: a field-order swap that's invisible in a zero-offset capture, and 4-byte bitmap padding that only shows up for small chunk counts. - Changed Block Tracking: turned out to need no NFC work at all -- VirtualMachine.QueryChangedDiskAreas is public VIM API. Added thin wrappers (enable_change_tracking / disk_change_id / query_changed_disk_areas) and documented real-world characteristics (extent granularity, wildcard changeId semantics) from live testing. - Investigated NFC_DELTA_DISK: found it's an optional VMFS-only VDDK client optimization (per `strings` on libvixDiskLib.so), not a correctness requirement -- reading, writing, and querying allocated blocks on an actual snapshot delta file already work with the existing NFC_DISK-only implementation. Documented a real gotcha found along the way: querying allocated blocks on the same still-open handle a write just went through can see stale data. Adds unit tests (bitmap decode/merge, DDB_GET wire format, CBT dataclass conversion, validation errors) and integration tests (GetInfo, QueryAllocatedBlocks, CBT full cycle, delta-disk read/write/ query) validated against a live ESXi 8.0.3 lab. Full protocol details and the reverse-engineering process are in docs/nfc_auth.md, docs/nfc_open.md, docs/nfc_read.md, docs/cbt.md, and docs/reverse_engineering_procedure.md. --- README.md | 27 ++- docs/cbt.md | 110 +++++++++++ docs/nfc_auth.md | 90 ++++++++- docs/nfc_open.md | 88 ++++++++- docs/nfc_read.md | 77 +++++++- docs/reverse_engineering_procedure.md | 160 ++++++++++++++- docs/ssl_hook.md | 12 ++ openvixdisklib/nfc_auth.py | 159 ++++++++++++++- openvixdisklib/nfc_open.py | 237 ++++++++++++++++++++++- openvixdisklib/openvixdisklib.py | 39 +++- tests/integration/test_cbt.py | 129 ++++++++++++ tests/integration/test_nfc_auth.py | 6 +- tests/integration/test_nfc_open.py | 17 +- tests/integration/test_openvixdisklib.py | 121 ++++++++++++ tests/unit/test_nfc_auth.py | 93 +++++++++ tests/unit/test_nfc_open.py | 215 ++++++++++++++++++++ 16 files changed, 1545 insertions(+), 35 deletions(-) create mode 100644 docs/cbt.md create mode 100644 tests/integration/test_cbt.py create mode 100644 tests/unit/test_nfc_auth.py create mode 100644 tests/unit/test_nfc_open.py diff --git a/README.md b/README.md index 254794b..342ffa3 100644 --- a/README.md +++ b/README.md @@ -17,17 +17,30 @@ from VDDK 8 NBD traffic; see `docs/`. ## Status -Implemented against vCenter 8 / ESXi 8. Default transport is `nbdssl` -(`nbd` is still available): +Implemented against vCenter 8 / ESXi 8, including a standalone ESXi +host with no vCenter. Default transport is `nbdssl` (`nbd` is still +available): -- `VixDiskLib_ConnectEx` (UID credentials) +- `VixDiskLib_ConnectEx` (UID credentials; vCenter or direct ESXi) - `VixDiskLib_Open` (datastore path, read-only or read-write) - `VixDiskLib_Read` (optional ``skip_decompression`` packs FastLZ extras) - `VixDiskLib_Write` - -Not implemented: compression open flags other than FastLZ, CBT / -allocated-block queries, disk geometry (`DDB_GET`), encrypted disks, -and direct ESXi `ha-nfc` without vCenter `vpxa-nfc`. +- `VixDiskLib_GetInfo` (capacity and physical geometry from the `Open` + reply; `biosGeo`/`adapterType`/`uuid` from `DDB_GET`, matching real + VDDK's cost and behavior) +- `VixDiskLib_QueryAllocatedBlocks` (allocated-block bitmap; see + `docs/nfc_read.md`) +- Changed Block Tracking: `openvixdisklib.nfc_auth.enable_change_tracking` + / `disk_change_id` / `query_changed_disk_areas` (public VIM API, not + part of VixDiskLib itself; see `docs/cbt.md`) + +Reading/writing a snapshot delta file directly (and running +`query_allocated_blocks` against it) already works — `NFC_DELTA_DISK` +turned out to be an optional VMFS-only VDDK client optimization, not a +correctness requirement (see `docs/reverse_engineering_procedure.md`). + +Not implemented: compression open flags other than FastLZ, and +encrypted disks. Requires Python 3.10 or later. diff --git a/docs/cbt.md b/docs/cbt.md new file mode 100644 index 0000000..64b1a23 --- /dev/null +++ b/docs/cbt.md @@ -0,0 +1,110 @@ +# Changed Block Tracking (CBT) + +This is not a reverse-engineered NFC feature. VixDiskLib does not expose +CBT itself: `VixDiskLib_QueryAllocatedBlocks` (implemented separately; +see `docs/nfc_read.md`) reports which blocks are *allocated* +(non-sparse) within a single NFC-opened disk, not which byte ranges +*changed* between two points in time. Real backup tools +get changed-range information from vSphere's public +`VirtualMachine.QueryChangedDiskAreas` VIM call instead, used alongside +VDDK/NFC reads for the actual bytes. `openvixdisklib.nfc_auth` wraps +that public pyVmomi call directly — no capture, no wire format to +document, per the project rule to reuse pyVmomi for anything it already +exposes. + +## Workflow + +1. `nfc_auth.enable_change_tracking(vm)` — sets + `VirtualMachineConfigSpec.changeTrackingEnabled = True` via + `ReconfigVM_Task`. Takes effect for writes from that point forward; + it does not retroactively track earlier changes. +2. Take a snapshot (or power-cycle the VM). A disk's `changeId` is + empty until this happens. +3. `nfc_auth.disk_change_id(vm, device_key)` — reads the current + `changeId` off `VirtualDisk.backing.changeId` (for example + `"52 f3 b6 37 30 8d ea 3e-58 70 c0 fd 61 44 26 62/2"`). +4. Do backup work (VDDK/NFC reads of the disk at that point). +5. Later, take another snapshot. +6. `nfc_auth.query_changed_disk_areas(vm, new_snapshot, device_key, + change_id_from_step_3)` — returns the byte ranges written between + the two snapshots. +7. Read only those ranges via VDDK/NFC on the new snapshot's disk + chain for an incremental backup. + +For an initial full backup, pass `change_id="*"` in step 6 without a +prior snapshot. **Correction from an earlier draft of this doc:** +this does *not* report the entire disk as one changed extent — see +"Wildcard `changeId='*'` reports allocated regions, not the whole +disk" below. + +## Validated in this lab + +Confirmed end-to-end against a temporary VM on the standalone ESXi +8.0.3 lab host (no vCenter): enabled CBT, snapshotted, wrote one +sector via `openvixdisklib.openvixdisklib` at a known offset, +snapshotted again, and called `query_changed_disk_areas` with the +first snapshot's `changeId`. The single reported extent +(`start=3932160, length=65536`, i.e. sectors 7680–7807) correctly +covered the written sector (7777). Extents were 64 KiB-aligned in this +lab's observations; that granularity is server-defined, not part of +the function's contract. + +### Wildcard `changeId="*"` reports allocated regions, not the whole disk + +Tested `query_changed_disk_areas(vm, snapshot, device_key, "*")` (the +initial-full-backup path, no prior snapshot needed) against a fresh +10 GiB thin-provisioned temp-VM disk. `result.length` correctly +reports the full declared virtual capacity (10737418240 bytes), but +`result.changed_areas` only covered **1 MiB** total — not the whole +disk. For a thin-provisioned disk, `"*"` reports the regions that are +actually *allocated* (backed by real data on the datastore), not the +full sparse virtual capacity; unwritten/unallocated regions have +nothing to back up regardless. A backup tool doing an initial full +backup with `"*"` should read exactly the reported extents, not assume +it needs to read `result.length` bytes. + +### One large contiguous write is one extent; scattered writes are not + +Wrote a single 4 MiB contiguous region plus three separate one-sector +writes at scattered offsets (same disk, one CBT interval), then +queried changed areas: + +``` +4 extents reported: + start= 196608 length= 65536 (64 KiB) + start= 51183616 length= 4259840 (4160 KiB) <-- covers the whole 4 MiB write as ONE extent + start= 460783616 length= 65536 (64 KiB) + start= 921567232 length= 65536 (64 KiB) +``` + +The 4 MiB write came back as a single extent (padded slightly beyond +4 MiB — 4259840 bytes vs. the exact 4194304 written — to the 64 KiB +tail-end granularity). Each scattered single-sector write produced its +own separate 64 KiB extent. **The extent list scales with the number +of discontiguous changed regions, not with the total volume of changed +data.** A multi-hundred-GB sequential write is still one small extent +record; thousands of scattered small writes (e.g. a busy database VM +doing random I/O across a large disk) produce thousands of extent +records in one `QueryChangedDiskAreas` response, since the API has no +pagination. Real backup tools facing that scenario typically chunk the +query with `start_offset` over fixed-size windows rather than querying +the whole disk in one call — `query_changed_disk_areas`'s +`start_offset` parameter exists for this, but nothing in this module +does the chunking loop itself; that is caller responsibility. + +Disk-size scaling itself (e.g., whether extent granularity increases +for very large disks) was not tested — only reasoned about above as an +open question, not verified against a large ESXi 8 disk. + +## What this does not cover + +- `VixDiskLib_QueryAllocatedBlocks` (NFC-level allocated-block bitmap + within a single disk, useful for skipping sparse regions inside a + delta disk) — implemented separately, see `docs/nfc_read.md`. Pairs + naturally with CBT: `query_changed_disk_areas` says which byte + ranges changed, `query_allocated_blocks` says which parts of a + snapshot's delta disk are actually worth reading. Note its "same + still-open write handle" staleness gotcha in `docs/nfc_read.md` if + chaining a CBT-driven write with an allocation check. +- `DDB_GET` fields (`biosGeo`, `adapterType`, `uuid`) — also + implemented, see `docs/nfc_open.md`. diff --git a/docs/nfc_auth.md b/docs/nfc_auth.md index e19b515..ce5893b 100644 --- a/docs/nfc_auth.md +++ b/docs/nfc_auth.md @@ -57,9 +57,9 @@ VDDK logs this as `Connected to VIM Server` / `Authenticating user` / `Logged in!`. OpenVixDiskLib keeps that `ServiceInstance` and its stub for the ticket call. -Direct ESXi login is the same SOAP login against hostd, but the NFC -moref and service name differ (`ha-nfc` instead of `nfcService` / -`vpxa-nfc`). The lab path is vCenter-mediated. +Direct ESXi login is the same SOAP login, this time against hostd +instead of vCenter. The NFC moref and service/PROXY name differ; see +"Direct ESXi (no vCenter)" below for the verified values. ## Stage 2: NFC ticket @@ -253,6 +253,90 @@ are for local ESXi credentials. With a vCenter ticket: argument is not what VDDK sends. The SHA-1 value is for verifying the TLS certificate, not for the `THUMBPRINT_SHA2` command. +## Direct ESXi (no vCenter) + +Captured against a standalone ESXi 8.0.3 host (`apiType: HostAgent`, +no vCenter in the picture at all) with the same SSL-hook technique +from `docs/ssl_hook.md`, using real VDDK 8.0.3 pointed straight at the +host (`vmxSpec=moref=`, `serverName=`). This corrects an +earlier guess in this file that assumed the moref would be `ha-nfc`. + +Differences from the vCenter-mediated path above: + +| Item | vCenter-mediated | Direct ESXi (verified) | +| ------------------------------- | ---------------------- | -------------------------- | +| `NfcService` moref | `nfcService` | `ha-nfc-service` | +| `NfcGetVmFilesResponse.service` | `vpxa-nfc` | `nfc` | +| `NfcGetVmFilesResponse.host` | present (ESXi address) | **absent** (omitted field) | +| authd `PROXY` line | `PROXY vpxa-nfc` | `PROXY nfc` | +| authd success line | `200 Connect ha-nfc` | `200 Connect ha-nfc` | + +The `NfcGetVmFiles` SOAP call itself is unchanged (`vm` argument only); +only the `_this` moref and the response fields differ: + +```xml + + <_this type="NfcService">ha-nfc-service + 1 + +``` + +```xml + + + 902 + ... + nfc + 1.1 + ... + + +``` + +Since `host` is absent, the client must already know where to dial +authd: the same ESXi host it just logged into over VIM. A vCenter +ticket always fills `host` because that ESXi address is not otherwise +known to the client. + +### Finding the `ha-nfc-service` moref + +VDDK does not hardcode this moref either. Before the `NfcGetVmFiles` +call, it issues an undocumented `RetrieveInternalContent` call on the +same `ServiceInstance` moref used for the public +`RetrieveServiceContent`: + +```xml + + <_this type="ServiceInstance">ServiceInstance + +``` + +The response carries ~20 undocumented managed-object refs +(`agentManager`, `llProvisioningManager`, `diskManager`, +`nfcService`, `proxyService`, ...); only `nfcService` matters here. +Its value was `nfcService` in the earlier vCenter capture and +`ha-nfc-service` on this bare ESXi host — VDDK reads it from this +response rather than assuming either name. + +### OpenVixDiskLib fix + +`openvixdisklib/nfc_auth.py` previously hardcoded +`NFC_SERVICE_MOID = "nfcService"`, which fails outright against a bare +ESXi host with `vmodl.fault.ManagedObjectNotFound`. It now resolves +the moref the same way VDDK does: `_nfc_service_moid()` issues the +`RetrieveInternalContent` SOAP call as raw XML over the existing +authenticated stub connection (registering pyVmomi types for the full +undocumented response schema wasn't worth it for one field) and +regex-extracts `nfcService` from the reply. + +`connect_authd()` also gained a `fallback_host` parameter: when +`ticket.host` is unset (the direct-ESXi case above), it dials the VIM +connection's own host instead. `openvixdisklib.py` passes +`conn.si._stub.host` for this. + +Validated end-to-end (`ConnectEx` + `Open` + `Read`, both `nbd` and +`nbdssl` transports) against a live standalone ESXi 8.0.3 host. + ## OpenVixDiskLib | Piece | Module | Reuses pyVmomi? | diff --git a/docs/nfc_open.md b/docs/nfc_open.md index bf61550..fd42597 100644 --- a/docs/nfc_open.md +++ b/docs/nfc_open.md @@ -143,7 +143,7 @@ AIO types used for Open / Read / Close, correlated with the consecutive | 9 | `SET_SOCK_OPTS` | 12 | | | 22 | `SET_RES_POOL` | 4 | | | 4 | `OPEN_FILE` | 60 | path string | -| 11 | `DDB_GET` | 16 | key name (VDDK only) | +| 11 | `DDB_GET` | 16 | key name | | 7 | `IO` | 44 | sector bytes (read reply / write request) | | 5 | `CLOSE_FILE` | 8 | | | 3 | `CLOSE_SESSION` | 4 | | @@ -156,6 +156,65 @@ VDDK Open also issues several `DDB_GET` queries (`resumeConsolidateSector`, (16 zero bytes) on this unencrypted disk. They are not required to obtain a file handle or to read sector 0. +`VixDiskLib_GetInfo` (Step 14) triggers ~20 more `DDB_GET` calls right +after `OPEN_FILE`, for these keys (captured in request-order, key name +is the extra string after the 16-byte payload, no `ddb.` prefix on the +wire): `resumeConsolidateSector`, `isDigest` (×3), `iofilters` (×2), +`logicalSectorSize` (×2), `physicalSectorSize` (×2), +`isNativeLinkedClone` (×2), `KMFilters`, `sidecars`, `adapterType`, +`uuid`, `geometry.cylinders`, `geometry.heads`, `geometry.sectors`, +`geometry.biosCylinders`, `geometry.biosHeads`, `geometry.biosSectors`. +On this lab disk, `biosGeo` came back all zeros (key not found) and +`logicalSectorSize`/`physicalSectorSize`/the non-bios `geometry.*` keys +duplicate what OPEN_FILE already returned — only `adapterType` and +`uuid` are genuinely new information from this burst. + +### `DDB_GET` request/reply layout + +Decoded from the same capture (request/reply pairs matched by `opId` +across all ~28 calls seen in one `GetInfo`). + +Request: 16-byte fixed payload plus the key name as a raw ASCII extra +(no NUL terminator, not counted in `size` — same convention as +`OPEN_FILE`'s path): + +| Offset | Type | Meaning | +| ------ | -------- | --------------------------------------- | +| 0 | `uint64` | File handle (same value as `OPEN_FILE`) | +| 8 | `uint32` | Key name length in bytes | +| 12 | `uint32` | 0 | +| 16 | — | Key name (ASCII, no `ddb.` prefix) | + +Reply: 16 bytes plus a value extra, **not** padded (unlike +`QueryAllocatedBlocks`'s bitmap — verified by decoding all 28 replies +in sequence with no desync): + +| Offset | Type | Meaning | +| ------ | -------- | --------------------------------------- | +| 0-11 | — | Zero/unused in every capture | +| 12 | `uint32` | Value length in bytes (`0` = not found) | +| 16 | — | Value (ASCII **text**, not binary) | + +Values are ASCII text even for keys that sound numeric — +`geometry.cylinders` comes back as the literal bytes `b"2088"`, not a +binary `uint32`. This matches how a VMDK descriptor file's DDB (disk +database) section stores keys as plain-text `ddb. = ""` +lines; `adapterType` comes back as `b"lsilogic"` (a string), not +VDDK's numeric `VIXDISKLIB_ADAPTER_SCSI_LSILOGIC` enum value — VDDK's +own client does that string-to-enum mapping internally, which +OpenVixDiskLib does not reproduce (`DiskInfo.adapter_type` is the raw +DDB string). + +Implemented as `openvixdisklib.nfc_open.NfcDisk.ddb_get(key) -> str | +None` and `NfcDisk.query_full_info() -> DiskInfo` (5 round trips: +`geometry.biosCylinders`/`biosHeads`/`biosSectors`, `adapterType`, +`uuid`), wired into `VixDiskLibHandle.get_info`, which now matches +real VDDK's `VixDiskLib_GetInfo` exactly — capacity/physGeo free from +`OPEN_FILE`, the rest costing the same 5 round trips VDDK itself pays. +Validated against the live ESXi lab: matches native VDDK's `GetInfo` +output on the same disk (`adapterType=3` ↔ `"lsilogic"`, same `uuid` +string, same zeroed `biosGeo`). + ### OPEN_SESSION / sockopts / resource pool `OPEN_SESSION` payload is 16 bytes, little-endian: @@ -205,9 +264,19 @@ Reply payload (60 bytes), fields that matter: | 8 | `uint64` | File handle (opaque, per open) | | 16 | `uint32` | File type (`2` = `NFC_DISK`) | | 20 | `uint32` | Flags echoed (`0x1e` or `0x1a`) | +| 28 | `uint64` | Disk capacity in **bytes** | | 36 | `uint32` | Sector size (`512` on this VM) | - -Later AIO messages pass that handle as a `uint64`. +| 40 | `uint32` | Physical geometry cylinders | +| 44 | `uint32` | Physical geometry heads | +| 48 | `uint32` | Physical geometry sectors | + +Later AIO messages pass that handle as a `uint64`. Offset 28 was found +by capturing `VixDiskLib_GetInfo` (Step 14, +`docs/reverse_engineering_procedure.md`): it matches +`VixDiskLibInfo.capacity` converted to bytes, and offsets 40/44/48 +match `VixDiskLibInfo.physGeo` exactly — both already arrive with this +reply, no separate `GetInfo` wire call exists. `biosGeo`, `adapterType`, +and `uuid` are **not** here; VDDK gets those from `DDB_GET` (below). ### IO (read / write) @@ -232,6 +301,8 @@ classic type 4 `NFC_SESSION_COMPLETE`. | Handshake + AIO + OPEN_FILE | `openvixdisklib.nfc_open.open_disk` | | AIO extra size / pool count | `open_disk(..., aio_buffer_size=, aio_buffer_count=)` | | Sector read / write / close | `openvixdisklib.nfc_open.NfcDisk` | +| Full disk info (`GetInfo`) | `openvixdisklib.openvixdisklib.VixDiskLibHandle.get_info` | +| VMDK descriptor DDB lookup | `openvixdisklib.nfc_open.NfcDisk.ddb_get` | Run: @@ -246,10 +317,15 @@ I/O: `docs/nfc_read.md`, `docs/nfc_write.md`, and ## What is still VDDK-only -- `DDB_GET` / geometry / zlib and skipz compression / encryption keys -- `NFC_DELTA_DISK`, change-block tracking +- zlib and skipz compression / encryption keys (`DDB_GET` is + implemented for the plain, non-encrypted keys covered above) - Host-switch (`NFC_AIO_SWITCH_HOST_*`) -- Direct ESXi `ha-nfc` without vCenter `vpxa-nfc` + +Reading/writing a snapshot delta file directly, and running +`query_allocated_blocks` against it, both already work with the +existing implementation — `NFC_DELTA_DISK` turned out to be an +optional VMFS-only VDDK client optimization, not a correctness +requirement; see `docs/reverse_engineering_procedure.md`. Reads after open are in `docs/nfc_read.md`. Writes are in `docs/nfc_write.md`. diff --git a/docs/nfc_read.md b/docs/nfc_read.md index ee67652..ceafe22 100644 --- a/docs/nfc_read.md +++ b/docs/nfc_read.md @@ -189,9 +189,82 @@ uncompressed request (offsets in this read, not on disk) buf when skip_decompression=True: extras packed densely from offset 0 ``` +## `VixDiskLib_QueryAllocatedBlocks` (AIO type 13) + +Reverse-engineered by extending the SSL/write-hook capture (Step 13/14 +technique, `docs/reverse_engineering_procedure.md`) to a ctypes call to +`VixDiskLib_QueryAllocatedBlocks` after `Open`, first with +`startSector=0` then — after the first capture's field guesses turned +out wrong — again with a non-zero `startSector` against a known +already-allocated region, to disambiguate fields that are 0 in the +degenerate zero-start case. + +No SOAP or authd traffic; it is one more AIO message type in the +already-open NFC/AIO session (like `DDB_GET`). + +Request (48 bytes):: + + uint64 handle (from OPEN_FILE) + uint64 reserved (0) + uint64 chunk_size_bytes (chunk_size_sectors * sector_size) + uint64 start_offset_bytes (start_sector * sector_size) + uint64 chunk_count (num_sectors // chunk_size_sectors) + uint64 reserved (0) + +**Field-order pitfall:** `start_offset_bytes` is at byte offset 24, not +8 — offset 8 is a reserved/always-zero field. A capture with +`startSector=0` can't tell these two apart (both read 0); only a +capture with a non-zero start distinguishes them. A first +implementation attempt put `start_offset_bytes` at offset 8 and got +`chunk_count`-many all-zero bits back for every non-zero-start query, +even for byte ranges known (from a zero-start, full-range query) to be +allocated — the server was silently ignoring the offset the client +thought it was requesting and returning an artifact of a different +misread field. + +Reply: a 48-byte body (offset 32 echoes `chunk_count`) followed by a +bitmap extra, one bit per chunk (LSB-first, `1` = chunk has allocated +data), **padded up to a 4-byte boundary** — `ceil(chunk_count / 8)` +alone is correct only when that value is already a multiple of 4 +(true for the `chunk_count=16384` case tested first, which is why the +padding bug wasn't caught immediately; a `chunk_count=16` query +exposed it, since `ceil(16/8)=2` bytes under-reads the real 4-byte +reply and desyncs the connection — the *next* AIO reply's header then +reads as garbage). + +Both `start_sector` and `num_sectors` must be exact multiples of +`chunk_size_sectors`; the server returns an `NFC_AIO_MSG_ERROR` (type +1) reply otherwise (hit by accident during validation with a +non-aligned `start_sector`). + +`openvixdisklib.nfc_open.NfcDisk.query_allocated_blocks` implements +this and run-length-merges contiguous set bits into +`AllocatedBlock(offset, length)` tuples (sectors, matching VDDK's +`VixDiskLibBlock`), exposed as +`VixDiskLibHandle.query_allocated_blocks`. Validated against the live +ESXi lab: a full-disk query, a query of a known-allocated sub-range, +and an aligned empty range all match native VDDK's own +`VixDiskLib_QueryAllocatedBlocks` output on the same disk. + +### Gotcha: query on the same still-open write handle can see stale data + +Writing a sector and then immediately calling +`query_allocated_blocks` **on that same open handle, without closing +it first**, can report the just-written region as *not* allocated — +the allocation metadata this call reads apparently isn't guaranteed +current until the write handle is closed. Closing after the write and +reopening (or querying from a separate handle opened after the write +completed) reports it correctly. Confirmed on both native VDDK and +this implementation — same-session-no-close showed the write as +unallocated on both, a fresh handle after close showed it correctly +on both — so this is a real server/VMFS behavior, not a bug in either +client. Real backup tools reading allocation before a read pass +naturally do this anyway (open read-only after the writer's handle +already closed), so it's unlikely to bite in practice, but do not +call `query_allocated_blocks` right after a write on the same handle +and expect it to reflect that write. + ## What is still VDDK-only - zlib and skipz NBD compression flags - `VixDiskLib_ReadAsync` (same IO messages, different client threading) -- `VixDiskLib_QueryAllocatedBlocks` / allocation bitmaps -- `VixDiskLib_GetInfo` capacity (not required to read a known range) diff --git a/docs/reverse_engineering_procedure.md b/docs/reverse_engineering_procedure.md index b988f5c..2c69b1d 100644 --- a/docs/reverse_engineering_procedure.md +++ b/docs/reverse_engineering_procedure.md @@ -8,8 +8,10 @@ This file is the **sequence of steps**, including dead ends, so later NFC work can follow the same loop instead of rediscovering it. Scope so far: `VixDiskLib_ConnectEx` + `VixDiskLib_Open` + -`VixDiskLib_Read` + `VixDiskLib_Write` against lab vCenter 8.0.1 / -ESXi 8, transports `nbd` and `nbdssl`. Validation method: +`VixDiskLib_Read` + `VixDiskLib_Write` + `VixDiskLib_GetInfo` + +`VixDiskLib_QueryAllocatedBlocks` against lab vCenter 8.0.1 / ESXi 8, +transports `nbd` and `nbdssl`, plus a standalone ESXi 8.0.3 host with +no vCenter (Step 13). Validation method: `tests/integration/` (the session-scoped `lab` fixture creates a temporary empty VM with a 10 GiB disk and destroys it when the pytest session ends). @@ -384,6 +386,155 @@ compression on each IO. Proof: `tests/integration/test_nfc_read_write.py` (`fastlz`) and `tests/perf/test_compare.py`. +## Step 13 — Direct ESXi (`ha-nfc`) without vCenter + +Same SSL-hook technique (Step 4), this time pointing VDDK 8.0.3 +straight at a standalone ESXi 8.0.3 host (`vmxSpec=moref=`, +`serverName=`, no vCenter in the topology). Confirmed +OpenVixDiskLib's hardcoded `NFC_SERVICE_MOID = "nfcService"` fails on +this host with `vmodl.fault.ManagedObjectNotFound` *before* touching +the capture — reproduced with plain `openvixdisklib` calls, no hook +needed to see that failure. + +The capture showed VDDK does not hardcode the moref either: it calls +an undocumented `RetrieveInternalContent` on the `ServiceInstance` +moref first, and reads `nfcService` from the reply (`ha-nfc-service` +on this host, vs. `nfcService` in the earlier vCenter capture). +`NfcGetVmFilesResponse.service` was `nfc` (not `vpxa-nfc`), and its +`host` field was **absent** — the authd endpoint is implicitly the +same host already logged into. Full detail in `docs/nfc_auth.md` +("Direct ESXi (no vCenter)"). + +Fix: `_nfc_service_moid()` in `openvixdisklib/nfc_auth.py` issues the +`RetrieveInternalContent` call as raw SOAP over the existing stub +connection (its response schema has ~20 other undocumented morefs not +worth registering with pyVmomi's type system for one field), and +`connect_authd()` takes a `fallback_host` used when `ticket.host` is +unset. Validated end-to-end (`ConnectEx`/`Open`/`Read`, `nbd` and +`nbdssl`) against the live host. + +## Step 14 — `VixDiskLib_GetInfo` capacity + +Extended the ctypes probe from Step 13 to call `VixDiskLib_GetInfo` +after `Open`, under the SSL hook plus a `write`/`read` interceptor on +fd 902 (Step 7), to see what wire traffic `GetInfo` adds. + +Result: **no new SOAP or authd traffic** — the same `RetrieveContent` ++ `Login` + `NfcGetVmFiles` + authd sequence as a plain `Open`. All the +extra traffic is inside the already-open NFC/AIO session: ~20 more +`DDB_GET` (type 11) requests right after `OPEN_FILE`, for keys like +`adapterType`, `uuid`, `geometry.cylinders`, `geometry.biosCylinders`, +etc. (full list in `docs/nfc_open.md`). + +Dumping every byte of the `OPEN_FILE` reply (not just the fields the +earlier Open-only capture had labeled) found `capacity` (offset 28, +`uint64` bytes) and `physGeo` (offsets 40/44/48) already present — +verified they match `VixDiskLibInfo.capacity`/`physGeo` from the same +`GetInfo` call exactly. Only `biosGeo`, `adapterType`, and `uuid` are +genuinely `DDB_GET`-only; `biosGeo` came back "key not found" (zeros) +on this unencrypted lab disk. + +Fix: extended `_parse_open_reply` in `openvixdisklib/nfc_open.py` to +also read those offsets, added `nfc_open.DiskInfo`/`DiskGeometry`, and +exposed `VixDiskLibHandle.get_info()`. No new NFC message type was +needed — `DDB_GET` (`adapterType`/`uuid`/`biosGeo`) is still open work. +Validated against the live host: `capacity_sectors=33554432` +(16 GiB), `phys_geo=(2088, 255, 63)`, matching native VDDK's +`GetInfo` on the same disk. + +## Note — CBT needed no reverse engineering + +Investigated change-block tracking (backlog item "CBT / +`QueryAllocatedBlocks`") expecting an NFC capture like the steps above. +It turned out `VirtualMachine.QueryChangedDiskAreas` — the actual +changed-byte-range query backup tools use — is public pyVmomi API with +no VDDK/NFC involvement at all; only `VixDiskLib_QueryAllocatedBlocks` +(disk-internal allocated-block bitmap, a different and lesser feature) +needed NFC work (done separately, see Step 15 below). Implemented as +`openvixdisklib.nfc_auth.enable_change_tracking` / +`disk_change_id` / `query_changed_disk_areas`; validated end-to-end +against a temp VM on the lab (enable CBT, snapshot, write a known +sector, snapshot, query — the written sector fell inside the reported +extent). Full workflow and lab evidence: `docs/cbt.md`. + +## Step 15 — `VixDiskLib_QueryAllocatedBlocks` (AIO type 13) + +Same SSL/write-hook technique, extended to `VixDiskLib_QueryAllocatedBlocks` +after `Open`. A first capture with `startSector=0` produced a request +where two 0-valued 8-byte fields were ambiguous — either could plausibly +be `start_offset_bytes`. Implementing from that guess alone put +`start_offset_bytes` at the wrong offset (8 instead of 24) and passed +every zero-start test while silently returning wrong (all-empty) +results for any non-zero start. A second capture with a **non-zero** +`startSector` against a region already known (from the first capture) +to be allocated broke the tie and found the real field order. + +A second, independent bug (bitmap reply length) was caught the same +way: `ceil(chunk_count/8)` matched the observed reply length for +`chunk_count=16384` (already a multiple of 4) but under-read and +desynced the connection for `chunk_count=16` — the reply pads the +bitmap to a 4-byte boundary. Diagnosed by testing candidate `extra_recv` +byte counts against whether the *next* AIO round-trip (a normal +`CLOSE_FILE`) completed cleanly, rather than guessing from a single +capture. + +Full protocol detail, the exact request/reply layout, and both bugs: +`docs/nfc_read.md`. Implemented as +`openvixdisklib.nfc_open.NfcDisk.query_allocated_blocks` +(`AllocatedBlock` dataclass) and +`VixDiskLibHandle.query_allocated_blocks`. Validated against the live +ESXi lab: full-disk query, a known-allocated sub-range, and an aligned +empty range all match native VDDK's own output on the same disk. + +## Step 16 — `DDB_GET` (AIO type 11) + +Already partly captured as a side effect of Step 14 (`VixDiskLib_GetInfo` +triggers ~28 `DDB_GET` calls); no new capture was needed, just decoding +the request/reply pairs from that saved log by matching `opId` across +both directions. Confirmed the request's first 8 bytes equal the +`OPEN_FILE` handle from the same capture, and that a "found" reply's +extra is plain ASCII text (`b"lsilogic"`, `b"2088"`, ...), not binary — +matching how a VMDK descriptor's DDB section stores key/value pairs as +text. No padding on the reply extra (unlike Step 15's bitmap), +confirmed by decoding all 28 request/reply pairs from one capture in +sequence without desync. + +Implemented as `NfcDisk.ddb_get(key) -> str | None` and +`NfcDisk.query_full_info() -> DiskInfo` (the 5 keys needed for +`bios_geo`/`adapter_type`/`uuid`), wired into +`VixDiskLibHandle.get_info` in place of the OPEN_FILE-only version +from Step 14 — `get_info` now matches real VDDK's `VixDiskLib_GetInfo` +completely, including paying the same round-trip cost. Full layout: +`docs/nfc_open.md`. Validated against the live ESXi lab: matches +native VDDK's `GetInfo` output on the same disk exactly. + +## Note — `NFC_DELTA_DISK` needed no protocol work either + +Investigated the backlog item "`NFC_DELTA_DISK` (reading directly +from a snapshot chain)" expecting a distinct wire message or OPEN_FILE +variant, similar to the CBT/`QueryAllocatedBlocks` split earlier. +`strings` on `libvixDiskLib.so` found `NFC_DELTA_DISK` is a **file-type +value** (like `NFC_DISK`), used by an internal VDDK client-side +heuristic ("`"%s" would probably benefit from bitmap copying, so +overriding file type to NFC_DELTA_DISK`") — a VMFS-only optimization +for very sparse redo logs, skipped entirely on NFS per an adjacent +string, and never observed to trigger in this lab's captures (no such +log line, `strings`-confirmed heuristic notwithstanding). + +Verified end-to-end that reading, writing, and `query_allocated_blocks` +against an actual post-snapshot delta file all already work correctly +with the existing NFC_DISK-only implementation — no code change +needed. The one real finding from this investigation was a gotcha, not +a gap: an initial test that wrote a sector and immediately queried +allocated blocks *on the same still-open write handle* reported the +write as unallocated; closing the handle first (or opening a separate +one) reported it correctly. Reproduced identically against **native +VDDK** on the same delta file (two-process capture, since loading +native VDDK in the same process as pyVmomi segfaults on this host's +OpenSSL — see `docs/ssl_hook.md`'s Limits section), so this is real +server/VMFS behavior, not specific to either client. Documented as a +`query_allocated_blocks` caveat in `docs/nfc_read.md`. + ## What to write down After a stage works: @@ -404,8 +555,5 @@ OpenVixDiskLib. Not yet reversed, same loop as above: -- `DDB_GET` / disk geometry, zlib/skipz compression, encrypted disks -- `NFC_DELTA_DISK`, CBT / `QueryAllocatedBlocks` -- `VixDiskLib_GetInfo` capacity +- zlib/skipz compression, encrypted disks - Host-switch AIO messages -- Direct ESXi `ha-nfc` without vCenter `vpxa-nfc` diff --git a/docs/ssl_hook.md b/docs/ssl_hook.md index cca3a43..6d52ceb 100644 --- a/docs/ssl_hook.md +++ b/docs/ssl_hook.md @@ -143,3 +143,15 @@ skipped. that is done offline on the hex log. - It must not ship in OpenVixDiskLib. Keep it out of the library path used by `openvixdisklib/nfc_auth.py`. +- `ctypes.CDLL` on `libvixDiskLib.so` **in the same process as + pyVmomi** can segfault, at least on this lab's Python/glibc build: + VDDK's bundled OpenSSL and the system OpenSSL pyVmomi already loaded + (for its own HTTPS) collide. Symptom: `Segmentation fault (core + dumped)`, no Python traceback. Split into two separate processes + instead — one doing pyVmomi/setup work, one doing only + `ctypes.CDLL`/native VDDK calls, handing data between them via a + file (see `docs/reverse_engineering_procedure.md`'s note on + `NFC_DELTA_DISK` for an example). This is the same underlying + conflict as `tests/integration/test_vddk.py` / + `test_crosscheck.py` needing `tox -e integration`'s isolated + subprocess env rather than running inside the main pytest process. diff --git a/openvixdisklib/nfc_auth.py b/openvixdisklib/nfc_auth.py index 49d85aa..259935e 100644 --- a/openvixdisklib/nfc_auth.py +++ b/openvixdisklib/nfc_auth.py @@ -19,16 +19,22 @@ import contextlib import hashlib +import re import socket import ssl +import time +import types +from dataclasses import dataclass from pyVim.connect import Disconnect, SmartConnect from pyVmomi import vim from pyVmomi.VmomiSupport import F_OPTIONAL, CreateManagedType, GetVmodlType -NFC_SERVICE_MOID = "nfcService" AUTHD_DEFAULT_PORT = 902 _NFC_TYPES_REGISTERED = False +_NFC_SERVICE_MOID_RE = re.compile(r"]*>([^<]+)") +_TASK_POLL_S = 0.5 +_TASK_TIMEOUT_S = 300 def _ssl_client_context(verify: bool = True) -> ssl.SSLContext: @@ -121,6 +127,43 @@ def _register_nfc_types() -> None: _NFC_TYPES_REGISTERED = True +def _nfc_service_moid(si: vim.ServiceInstance) -> str: + """Return the NfcService moref via the internal ``RetrieveInternalContent`` call. + + vCenter and a bare ESXi host disagree on this moref (``nfcService`` vs. + ``ha-nfc-service``); VDDK resolves it dynamically instead of assuming + vCenter's name, which is why OpenVixDiskLib must too. The response also + carries ~20 other undocumented managed-object refs (agent manager, disk + manager, and so on) that aren't worth registering with pyVmomi's type + system just to read one field, so the call is issued as raw SOAP over + the existing authenticated connection and only ``nfcService`` is pulled + out of the XML. + """ + stub = si._stub + info = types.SimpleNamespace( + wsdlName="RetrieveInternalContent", version=stub.version, params=() + ) + request = stub.SerializeRequest(si, info, ()) + headers = { + "Cookie": stub.cookie, + "SOAPAction": stub.versionId, + "Content-Type": "text/xml; charset=utf-8", + } + conn = stub.GetConnection() + try: + conn.request("POST", stub.path, request, headers) + response = conn.getresponse() + body = response.read().decode("utf-8") + finally: + stub.ReturnConnection(conn) + match = _NFC_SERVICE_MOID_RE.search(body) + if response.status != 200 or not match: + raise RuntimeError( + f"RetrieveInternalContent (status {response.status}) had no nfcService moref" + ) + return match.group(1) + + def nfc_service(si: vim.ServiceInstance) -> vim.NfcService: """Return the vCenter/ESXi NfcService managed object on ``si``'s SOAP stub. @@ -129,7 +172,7 @@ def nfc_service(si: vim.ServiceInstance) -> vim.NfcService: """ _register_nfc_types() nfc_cls = GetVmodlType("vim.NfcService") - return nfc_cls(NFC_SERVICE_MOID, si._stub) + return nfc_cls(_nfc_service_moid(si), si._stub) def connect_vim( @@ -315,6 +358,7 @@ def connect_authd( allow_untrusted: bool = False, timeout: float = 30.0, nfc_ssl: bool = True, + fallback_host: str | None = None, ) -> ssl.SSLSocket: """Complete the ESXi authd handshake using an NFC HostServiceTicket. @@ -337,8 +381,12 @@ def connect_authd( timeout: Socket timeout in seconds. nfc_ssl: When True (the default), PROXY to the NFCSSL service used by nbdssl. Pass False for plaintext NFC (nbd). + fallback_host: Host to dial when ``ticket.host`` is unset. A ticket + issued directly by a bare ESXi host (no vCenter) omits ``host`` + entirely, since the authd endpoint is that same host; pass the + VIM connection's host in that case. """ - host = ticket.host + host = ticket.host or fallback_host port = ticket.port or AUTHD_DEFAULT_PORT raw = socket.create_connection((host, port), timeout=timeout) try: @@ -463,9 +511,112 @@ def authenticate( read_only=read_only, ) authd_sock = connect_authd( - ticket, allow_untrusted=allow_untrusted, nfc_ssl=nfc_ssl + ticket, allow_untrusted=allow_untrusted, nfc_ssl=nfc_ssl, fallback_host=host ) except Exception: Disconnect(si) raise return NfcAuthSession(si, ticket, authd_sock, nfc_ssl=nfc_ssl) + + +# --- Changed Block Tracking (CBT) --- +# +# VixDiskLib does not expose CBT itself: ``VixDiskLib_QueryAllocatedBlocks`` +# reports which blocks are allocated (non-sparse) within a single +# NFC-opened disk, not which byte ranges changed between two points in +# time. Real backup tools get that from vSphere's public +# ``VirtualMachine.QueryChangedDiskAreas`` VIM call instead, used +# alongside VDDK/NFC reads. The helpers below are thin wrappers around +# that public pyVmomi call — no NFC reverse engineering was needed for +# them. See ``docs/cbt.md``. + + +@dataclass(frozen=True, slots=True) +class ChangedExtent: + """One changed byte range, as returned by ``QueryChangedDiskAreas``.""" + + start: int + length: int + + +@dataclass(frozen=True, slots=True) +class ChangedDiskAreas: + """Result of ``QueryChangedDiskAreas``, converted to plain dataclasses.""" + + start_offset: int + length: int + changed_areas: tuple[ChangedExtent, ...] + + +def _wait_for_cbt_task(task: vim.Task): + deadline = time.monotonic() + _TASK_TIMEOUT_S + while task.info.state in (vim.TaskInfo.State.running, vim.TaskInfo.State.queued): + if time.monotonic() > deadline: + raise TimeoutError(f"timed out waiting for vSphere task {task}") + time.sleep(_TASK_POLL_S) + if task.info.state != vim.TaskInfo.State.success: + raise RuntimeError(f"vSphere task failed: {task.info.error}") + return task.info.result + + +def enable_change_tracking(vm: vim.VirtualMachine) -> None: + """Enable CBT on ``vm``. + + Takes effect for writes from this point forward; it does not + retroactively track earlier changes. A ``changeId`` for a disk only + becomes available after the next snapshot or power cycle once this + is set. + """ + spec = vim.vm.ConfigSpec(changeTrackingEnabled=True) + _wait_for_cbt_task(vm.ReconfigVM_Task(spec=spec)) + + +def disk_change_id(vm: vim.VirtualMachine, device_key: int) -> str: + """Return the current ``changeId`` for the disk with ``device_key``. + + Requires CBT to be enabled and at least one snapshot (or power + cycle) to have happened since. Raises ``ValueError`` if the disk + isn't found or has no ``changeId`` yet (CBT not active for it). + """ + for device in vm.config.hardware.device: + if isinstance(device, vim.vm.device.VirtualDisk) and device.key == device_key: + change_id = getattr(device.backing, "changeId", None) + if not change_id: + raise ValueError( + f"disk {device_key} on {vm._moId} has no changeId yet " + "(enable CBT and take a snapshot first)" + ) + return change_id + raise ValueError(f"no VirtualDisk with device key {device_key} on {vm._moId}") + + +def query_changed_disk_areas( + vm: vim.VirtualMachine, + snapshot: vim.vm.Snapshot, + device_key: int, + change_id: str, + start_offset: int = 0, +) -> ChangedDiskAreas: + """Return byte ranges changed since ``change_id``, up to ``snapshot``. + + Thin wrapper around the public + ``VirtualMachine.QueryChangedDiskAreas`` VIM call. ``change_id`` is + the value from an earlier ``disk_change_id()`` call (or ``"*"`` for + the entire disk, e.g. for an initial full backup). Extents are + 64 KiB-aligned in this lab's observations, but that granularity is + server-defined and not part of this function's contract. + """ + result = vm.QueryChangedDiskAreas( + snapshot=snapshot, + deviceKey=device_key, + startOffset=start_offset, + changeId=change_id, + ) + return ChangedDiskAreas( + start_offset=result.startOffset, + length=result.length, + changed_areas=tuple( + ChangedExtent(start=extent.start, length=extent.length) + for extent in result.changedArea + ), + ) diff --git a/openvixdisklib/nfc_open.py b/openvixdisklib/nfc_open.py index 1b25bd5..e567e7a 100644 --- a/openvixdisklib/nfc_open.py +++ b/openvixdisklib/nfc_open.py @@ -64,8 +64,14 @@ NFC_AIO_MSG_IO = 7 NFC_AIO_MSG_SET_SOCK_OPTS = 9 NFC_AIO_MSG_DDB_GET = 11 +NFC_AIO_MSG_QUERY_ALLOCATED_BLOCKS = 13 NFC_AIO_MSG_SET_RES_POOL = 22 +# Chunk size used in this project's capture/validation of +# query_allocated_blocks (128 sectors = 64 KiB); not a documented VDDK +# default, just a convenient granularity that worked in this lab. +NFC_QUERY_ALLOCATED_BLOCKS_CHUNK_SECTORS = 128 + # Open-file body: file type NFC_DISK. 0x1e is what VDDK sends for # VIXDISKLIB_FLAG_OPEN_READ_ONLY; writable opens clear bit 0x04 (0x1a). NFC_DISK = 2 @@ -82,6 +88,45 @@ NFC_COMPRESSION_FASTLZ = 2 +@dataclass(frozen=True, slots=True) +class DiskGeometry: + """CHS geometry, matching VDDK's ``VixDiskLibGeometry``.""" + + cylinders: int + heads: int + sectors: int + + +@dataclass(frozen=True, slots=True) +class AllocatedBlock: + """One allocated run, matching VDDK's ``VixDiskLibBlock`` (sectors).""" + + offset: int + length: int + + +@dataclass(frozen=True, slots=True) +class DiskInfo: + """Matches VDDK's ``VixDiskLibInfo``. + + ``phys_geo`` and ``capacity_sectors`` are read directly off + OPEN_FILE (offsets 40/44/48 and 28 respectively) — free, no extra + NFC round trip. ``bios_geo``, ``adapter_type``, and ``uuid`` come + from ``DDB_GET`` (see ``NfcDisk.ddb_get`` / ``query_full_info``, + ``docs/nfc_open.md``): each is a real round trip, matching what + real VDDK's ``VixDiskLib_GetInfo`` does. ``bios_geo`` defaults to + all zeros and ``adapter_type``/``uuid`` to ``None`` when the disk + has no snapshots or predates that DDB key (VDDK does the same for + a missing key). + """ + + capacity_sectors: int + phys_geo: DiskGeometry + bios_geo: DiskGeometry = DiskGeometry(cylinders=0, heads=0, sectors=0) + adapter_type: str | None = None + uuid: str | None = None + + @dataclass(frozen=True, slots=True) class ReadFragment: """One NFC AIO extra in a packed skip-decompression ``buf``. @@ -260,6 +305,7 @@ def __init__( compression: int = NFC_COMPRESSION_NONE, aio_buffer_size: int = NFC_AIO_BUFFER_SIZE, aio_buffer_count: int = NFC_AIO_BUFFER_COUNT, + info: DiskInfo | None = None, ) -> None: """Wrap an AIO session that already has ``path`` open. @@ -275,6 +321,8 @@ def __init__( at most this large. aio_buffer_count: OPEN_SESSION buffer pool count (default ``NFC_AIO_BUFFER_COUNT``). + info: Capacity/geometry from the OPEN_FILE reply. ``None`` + before the reply arrives. """ self._sock = sock self._op_id = 0 @@ -284,6 +332,7 @@ def __init__( self.compression = compression self.aio_buffer_size = aio_buffer_size self.aio_buffer_count = aio_buffer_count + self.info = info self._closed = False def _next_op_id(self) -> int: @@ -516,6 +565,150 @@ def write(self, start_sector: int, num_sectors: int, data: bytes) -> None: f"expected type={NFC_AIO_MSG_IO} opId={op_id}" ) + def query_allocated_blocks( + self, + start_sector: int, + num_sectors: int, + chunk_size_sectors: int = NFC_QUERY_ALLOCATED_BLOCKS_CHUNK_SECTORS, + ) -> tuple[AllocatedBlock, ...]: + """Return allocated (non-sparse) runs. Matches ``VixDiskLib_QueryAllocatedBlocks``. + + Captured from VDDK: a 48-byte request:: + + uint64 handle + uint64 reserved (0) + uint64 chunk_size_bytes (chunk_size_sectors * sector_size) + uint64 start_offset_bytes (start_sector * sector_size) + uint64 chunk_count (num_sectors // chunk_size_sectors) + uint64 reserved (0) + + (``start_offset_bytes`` and the first ``reserved`` field are + easy to swap — both are 0 in a ``start_sector=0`` capture, + which is what an earlier draft of this method got wrong; a + second capture with a non-zero ``start_sector`` was needed to + tell them apart.) + + The reply echoes a 48-byte body whose offset 32 carries the + same chunk count back, followed by a bitmap extra — one bit + per chunk, LSB-first, set when that chunk contains allocated + data, padded up to a **4-byte boundary** (``ceil(chunk_count / 8)`` + alone under-reads and desyncs the connection whenever that + raw byte count isn't already a multiple of 4). + This mirrors VDDK's own client-side behavior of run-length + merging contiguous set bits into ``VixDiskLibBlock`` entries + (offset/length here are in **sectors**, matching the public + VDDK struct, unlike the bytes used on the wire). See + ``docs/nfc_read.md``. + + Args: + start_sector: Sector offset from the start of the disk; + must be a multiple of ``chunk_size_sectors`` (the + server returns an AIO error otherwise). + num_sectors: Number of sectors to query; must be a multiple + of ``chunk_size_sectors``. + chunk_size_sectors: Minimum run granularity, in sectors. + """ + if num_sectors % chunk_size_sectors != 0: + raise ValueError("num_sectors must be a multiple of chunk_size_sectors") + if start_sector % chunk_size_sectors != 0: + raise ValueError("start_sector must be a multiple of chunk_size_sectors") + chunk_count = num_sectors // chunk_size_sectors + bitmap_bytes = -(-((chunk_count + 7) // 8) // 4) * 4 + request = struct.pack( + " str | None: + """Return a VMDK descriptor DDB value, or ``None`` if unset. + + Captured from VDDK: request is a 16-byte fixed payload plus the + key name as a raw ASCII extra (no NUL terminator, not counted + in ``size``, same convention as ``OPEN_FILE``'s path):: + + uint64 handle + uint32 key_name_length + uint32 reserved (0) + + + Reply is 16 bytes plus a value extra, **not** padded (unlike + ``QueryAllocatedBlocks``'s bitmap):: + + 96 bits reserved/unused (always zero in this lab) + uint32 value_length (0 = key not found) + + + Values are ASCII text even for keys that sound numeric + (``geometry.cylinders`` comes back as the bytes ``b"2088"``, + not a binary int) — this matches how a VMDK descriptor file's + DDB (disk database) section stores keys as plain text + ``ddb. = ""`` lines. See ``docs/nfc_open.md``. + + Args: + key: DDB key name without the ``ddb.`` prefix (for example + ``"adapterType"``, ``"uuid"``, ``"geometry.cylinders"``). + """ + key_bytes = key.encode("ascii") + request = struct.pack(" DiskInfo: + """Return a ``DiskInfo`` with ``bios_geo``/``adapter_type``/``uuid`` filled in. + + ``self.info`` (from OPEN_FILE) already has ``capacity_sectors`` + and ``phys_geo`` for free; this issues 5 ``DDB_GET`` round trips + for the rest, matching what real VDDK's ``VixDiskLib_GetInfo`` + does on every call. DDB values are ASCII text; geometry fields + are parsed as decimal integers, and any missing key falls back + to ``DiskInfo``'s defaults (matches VDDK: a disk with no + snapshots, or from before this DDB key existed, has none of + these set). + """ + assert self.info is not None + bios_cylinders = self.ddb_get("geometry.biosCylinders") + bios_heads = self.ddb_get("geometry.biosHeads") + bios_sectors = self.ddb_get("geometry.biosSectors") + bios_geo = DiskGeometry( + cylinders=int(bios_cylinders) if bios_cylinders else 0, + heads=int(bios_heads) if bios_heads else 0, + sectors=int(bios_sectors) if bios_sectors else 0, + ) + return DiskInfo( + capacity_sectors=self.info.capacity_sectors, + phys_geo=self.info.phys_geo, + bios_geo=bios_geo, + adapter_type=self.ddb_get("adapterType"), + uuid=self.ddb_get("uuid"), + ) + def close(self) -> None: """Close the VMDK, the AIO session, and the classic NFC session.""" if self._closed: @@ -591,16 +784,51 @@ def _aio_prepare(disk: NfcDisk) -> None: disk._aio_roundtrip(NFC_AIO_MSG_SET_RES_POOL, struct.pack(" tuple[int, int]: - if len(body) < 40: +def _decode_allocated_bitmap( + bitmap: bytes, chunk_count: int, start_sector: int, chunk_size_sectors: int +) -> tuple[AllocatedBlock, ...]: + """Run-length-merge a QueryAllocatedBlocks bitmap into ``AllocatedBlock``s. + + ``bitmap`` is one bit per chunk, LSB-first (bit 0 of byte 0 is + chunk 0), possibly longer than strictly needed for padding; only + the first ``chunk_count`` bits are read. + """ + blocks = [] + run_start = None + for chunk_idx in range(chunk_count): + allocated = (bitmap[chunk_idx // 8] >> (chunk_idx % 8)) & 1 + if allocated and run_start is None: + run_start = chunk_idx + elif not allocated and run_start is not None: + blocks.append((run_start, chunk_idx - run_start)) + run_start = None + if run_start is not None: + blocks.append((run_start, chunk_count - run_start)) + return tuple( + AllocatedBlock( + offset=start_sector + run_chunk * chunk_size_sectors, + length=run_len * chunk_size_sectors, + ) + for run_chunk, run_len in blocks + ) + + +def _parse_open_reply(body: bytes) -> tuple[int, int, DiskInfo]: + if len(body) < 52: raise NfcProtocolError(f"OPEN_FILE reply too short: {len(body)}") handle, file_type, _flags = struct.unpack_from(" nfc_open.DiskInfo: + """Return disk info. Matches ``VixDiskLib_GetInfo``. + + ``capacity_sectors``/``phys_geo`` are free (already in the + ``OPEN_FILE`` reply from ``open()``); ``bios_geo``/ + ``adapter_type``/``uuid`` cost 5 ``DDB_GET`` round trips, same + as real VDDK pays on every ``GetInfo`` call. See + ``docs/nfc_open.md``. + """ + return disk_handle.disk.query_full_info() + + def query_allocated_blocks( + self, + disk_handle: _DiskHandle, + start_sector: int, + num_sectors: int, + chunk_size_sectors: int = nfc_open.NFC_QUERY_ALLOCATED_BLOCKS_CHUNK_SECTORS, + ) -> tuple[nfc_open.AllocatedBlock, ...]: + """Return allocated runs. Matches ``VixDiskLib_QueryAllocatedBlocks``. + + Args: + disk_handle: Handle from ``open()``. + start_sector: Sector offset from the start of the disk. + num_sectors: Number of sectors to query; must be a multiple + of ``chunk_size_sectors``. + chunk_size_sectors: Minimum run granularity, in sectors. + """ + return disk_handle.disk.query_allocated_blocks( + start_sector, num_sectors, chunk_size_sectors + ) + def read( self, disk_handle: _DiskHandle, diff --git a/tests/integration/test_cbt.py b/tests/integration/test_cbt.py new file mode 100644 index 0000000..12243ea --- /dev/null +++ b/tests/integration/test_cbt.py @@ -0,0 +1,129 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Exercise Changed Block Tracking against the lab. + +Unlike VDDK/NFC features, these VIM-level calls (``nfc_auth. +enable_change_tracking`` / ``disk_change_id`` / ``query_changed_disk_areas``) +are thin wrappers around public pyVmomi; see ``docs/cbt.md``. +""" + +from pyVim.connect import Disconnect +from pyVmomi import vim + +from openvixdisklib import nfc_auth +from openvixdisklib import openvixdisklib as vixdisklib +from tests.integration.base import SECTOR_SIZE, LabEnv, _connect_vim, _wait_for_task, pattern_bytes + + +def _disk_device(vm: vim.VirtualMachine) -> vim.vm.device.VirtualDisk: + """Return the lab VM's first virtual disk device.""" + for device in vm.config.hardware.device: + if isinstance(device, vim.vm.device.VirtualDisk): + return device + raise AssertionError(f"{vm._moId} has no virtual disk") + + +def _disable_change_tracking(vm: vim.VirtualMachine) -> None: + spec = vim.vm.ConfigSpec(changeTrackingEnabled=False) + _wait_for_task(vm.ReconfigVM_Task(spec=spec)) + + +class TestCbt: + def test_full_cbt_cycle(self, lab: LabEnv) -> None: + """Enable CBT, write a known sector, and see it in a changed-areas query.""" + si = _connect_vim( + lab.host, lab.username, lab.password, lab.port, lab.thumbprint, lab.allow_untrusted + ) + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + nfc_auth.enable_change_tracking(vm) + vm.Reload() + assert vm.config.changeTrackingEnabled is True + + device_key = _disk_device(vm).key + + snap1 = _wait_for_task( + vm.CreateSnapshot_Task("cbt-baseline", "", False, False) + ) + vm.Reload() + change_id_1 = nfc_auth.disk_change_id(vm, device_key) + assert change_id_1 + + write_sector = 7777 + written = pattern_bytes(SECTOR_SIZE, b"CBT-TEST") + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted} + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, _disk_device(vm).backing.fileName, flags=0) as disk, + ): + buf = vixdisklib.get_buffer(SECTOR_SIZE) + buf[:SECTOR_SIZE] = written + handle.write(disk, write_sector, 1, buf) + + snap2 = _wait_for_task( + vm.CreateSnapshot_Task("cbt-after-write", "", False, False) + ) + vm.Reload() + + result = nfc_auth.query_changed_disk_areas( + vm, snap2, device_key, change_id_1 + ) + write_byte = write_sector * SECTOR_SIZE + assert any( + extent.start <= write_byte < extent.start + extent.length + for extent in result.changed_areas + ), f"sector {write_sector} not covered by any reported extent" + finally: + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + if vm.snapshot is not None: + _wait_for_task(vm.RemoveAllSnapshots_Task()) + _disable_change_tracking(vm) + finally: + Disconnect(si) + + def test_query_changed_disk_areas_wildcard_change_id(self, lab: LabEnv) -> None: + """changeId='*' (initial full backup) reports allocated regions. + + Not the full sparse virtual capacity -- see docs/cbt.md. + """ + si = _connect_vim( + lab.host, lab.username, lab.password, lab.port, lab.thumbprint, lab.allow_untrusted + ) + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + nfc_auth.enable_change_tracking(vm) + vm.Reload() + + device_key = _disk_device(vm).key + capacity_bytes = _disk_device(vm).capacityInKB * 1024 + + snap = _wait_for_task( + vm.CreateSnapshot_Task("cbt-wildcard", "", False, False) + ) + vm.Reload() + + result = nfc_auth.query_changed_disk_areas(vm, snap, device_key, "*") + # changeId="*" reports allocated (backed) regions, not the full + # sparse virtual capacity -- this lab disk is thin-provisioned + # and shared across the test session, so exactly how much is + # allocated depends on what earlier tests wrote. Only the + # length field (declared virtual capacity) is a fixed value; + # changed_areas is just asserted sane (non-empty, in bounds). + assert result.length == capacity_bytes + assert result.changed_areas + for extent in result.changed_areas: + assert extent.start >= 0 + assert extent.start + extent.length <= capacity_bytes + finally: + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + if vm.snapshot is not None: + _wait_for_task(vm.RemoveAllSnapshots_Task()) + _disable_change_tracking(vm) + finally: + Disconnect(si) diff --git a/tests/integration/test_nfc_auth.py b/tests/integration/test_nfc_auth.py index e074f1e..d6c32e1 100644 --- a/tests/integration/test_nfc_auth.py +++ b/tests/integration/test_nfc_auth.py @@ -11,7 +11,11 @@ def test_authd_handshake_completes(self, lab: LabEnv) -> None: """Complete VIM login and authd PROXY through ``200 Connect``.""" with lab.authenticate() as session: ticket = session.ticket - assert ticket.host + # ticket.host is unset on a direct-ESXi ticket (no vCenter): + # the authd endpoint is implicitly the host already logged + # into. connect_authd() falls back to that host, so the + # socket's peer address is the reliable check here. + assert session.authd_sock.getpeername()[0] assert ticket.port assert ticket.sessionId assert session.nfc_ssl is True diff --git a/tests/integration/test_nfc_open.py b/tests/integration/test_nfc_open.py index 9d9beb1..08fe813 100644 --- a/tests/integration/test_nfc_open.py +++ b/tests/integration/test_nfc_open.py @@ -6,7 +6,7 @@ import pytest from openvixdisklib import nfc_open -from tests.integration.base import SECTOR_SIZE, LabEnv, pattern_bytes +from tests.integration.base import _DISK_CAPACITY_KB, SECTOR_SIZE, LabEnv, pattern_bytes class TestNfcOpen: @@ -34,3 +34,18 @@ def test_open_disk_and_read_first_sector( got = disk.read(0, 1) assert got is not expected assert got == expected + + def test_open_disk_reports_capacity_and_geometry(self, lab: LabEnv) -> None: + """OPEN_FILE's reply carries capacity and physical geometry (GetInfo).""" + with ( + lab.authenticate() as session, + nfc_open.open_disk(session, lab.disk_path) as disk, + ): + assert disk.info is not None + assert disk.info.capacity_sectors == ( + _DISK_CAPACITY_KB * 1024 // SECTOR_SIZE + ) + geo = disk.info.phys_geo + assert geo.cylinders > 0 + assert geo.heads > 0 + assert geo.sectors > 0 diff --git a/tests/integration/test_openvixdisklib.py b/tests/integration/test_openvixdisklib.py index 6c853d6..6731a26 100644 --- a/tests/integration/test_openvixdisklib.py +++ b/tests/integration/test_openvixdisklib.py @@ -13,6 +13,7 @@ from openvixdisklib import openvixdisklib as vixdisklib from openvixdisklib.openvixdisklib import ReadResult from tests.integration.base import ( + _DISK_CAPACITY_KB, SECTOR_AT_1GB, SECTOR_SIZE, LabEnv, @@ -94,6 +95,58 @@ def test_write_and_read_sector_zero_and_one_gib( handle.read(disk, start, 1, read_buf) assert read_buf.raw[:SECTOR_SIZE] == expected + def test_get_info(self, lab: LabEnv) -> None: + """get_info returns the lab VM's known disk capacity and geometry.""" + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted, "read_only": True} + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open( + conn, lab.disk_path, flags=vixdisklib.VIXDISKLIB_FLAG_OPEN_READ_ONLY + ) as disk, + ): + info = handle.get_info(disk) + assert info.capacity_sectors == _DISK_CAPACITY_KB * 1024 // SECTOR_SIZE + assert info.phys_geo.cylinders > 0 + assert info.phys_geo.heads > 0 + assert info.phys_geo.sectors > 0 + # bios_geo is DDB-derived and unset (all zero) on a disk with + # no snapshots yet, matching VDDK's own default for a missing key. + assert info.bios_geo == vixdisklib.DiskGeometry( + cylinders=0, heads=0, sectors=0 + ) + assert info.adapter_type # non-empty DDB string, e.g. "lsilogic" + assert info.uuid # non-empty DDB string + + def test_query_allocated_blocks(self, lab: LabEnv) -> None: + """A written sector's chunk shows up as an allocated run.""" + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + chunk_size_sectors = 128 + # Chunk-aligned offset away from what other tests in this shared + # session-scoped VM write to (sector 0, SECTOR_AT_1GB, ...). + write_sector = 300 * chunk_size_sectors + write_buf = vixdisklib.get_buffer(SECTOR_SIZE) + write_buf[:SECTOR_SIZE] = pattern_bytes(SECTOR_SIZE, b"OVDL-QAB") + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted} + ) + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, lab.disk_path, flags=0) as disk, + ): + handle.write(disk, write_sector, 1, write_buf) + blocks = handle.query_allocated_blocks( + disk, + start_sector=(write_sector // chunk_size_sectors) * chunk_size_sectors, + num_sectors=chunk_size_sectors, + chunk_size_sectors=chunk_size_sectors, + ) + assert any( + b.offset <= write_sector < b.offset + b.length for b in blocks + ), f"written sector {write_sector} not covered by {blocks}" + def test_read_only_open_snapshot_parent(self, lab: LabEnv) -> None: """Read-only Open uses NfcGetVmFiles, including a snapshot parent path. @@ -162,6 +215,74 @@ def read_sector(path: str) -> bytes: finally: Disconnect(si) + def test_write_and_query_allocated_blocks_on_delta_disk(self, lab: LabEnv) -> None: + """Read/write/query all work directly on a post-snapshot delta file. + + ``NFC_DELTA_DISK`` (an alternate OPEN_FILE file-type value, + per ``strings`` on ``libvixDiskLib.so``) turned out to be an + optional VMFS-only VDDK client optimization, not needed for + correctness — this exercises the plain ``NFC_DISK`` path + against an actual delta file. See + ``docs/reverse_engineering_procedure.md``. + + Also covers the gotcha documented in ``docs/nfc_read.md``: + querying allocated blocks on the *same still-open* handle a + write just went through can see stale (pre-write) data: the + write below is done in its own ``with`` block, closed, before + the separate query. + """ + handle = vixdisklib.VixDiskLibHandle(vixdisklib_compatibility_version="8.0") + chunk_size_sectors = 128 + write_sector = 400 * chunk_size_sectors + expected = pattern_bytes(SECTOR_SIZE, b"OVDL-DELTA") + write_buf = vixdisklib.get_buffer(SECTOR_SIZE) + write_buf[:SECTOR_SIZE] = expected + connect_kwargs = lab.vixdisklib_connect_kwargs( + {"allow_untrusted": lab.allow_untrusted} + ) + read_buf = vixdisklib.get_buffer(SECTOR_SIZE) + + si = _connect_vim( + lab.host, lab.username, lab.password, lab.port, lab.thumbprint, lab.allow_untrusted + ) + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + _wait_for_task(vm.CreateSnapshot_Task("ovdl-delta", "", False, False)) + delta_path = _virtual_disk_backing(vm).fileName + assert delta_path != lab.disk_path + + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, delta_path, flags=0) as disk, + ): + handle.write(disk, write_sector, 1, write_buf) + + # Fresh handle after close, per the gotcha above. + with ( + handle.connect(**connect_kwargs) as conn, + handle.open(conn, delta_path, flags=0) as disk, + ): + read_buf[:SECTOR_SIZE] = b"\xa5" * SECTOR_SIZE + handle.read(disk, write_sector, 1, read_buf) + assert read_buf.raw[:SECTOR_SIZE] == expected + + blocks = handle.query_allocated_blocks( + disk, + start_sector=(write_sector // chunk_size_sectors) * chunk_size_sectors, + num_sectors=chunk_size_sectors, + chunk_size_sectors=chunk_size_sectors, + ) + assert any( + b.offset <= write_sector < b.offset + b.length for b in blocks + ), f"written sector {write_sector} on delta file not covered by {blocks}" + finally: + try: + vm = vim.VirtualMachine(lab.vm_moref, si._stub) + if vm.snapshot is not None: + _wait_for_task(vm.RemoveAllSnapshots_Task()) + finally: + Disconnect(si) + @pytest.mark.parametrize( "aio_buffer_size, n_sectors, n_fragments", [ diff --git a/tests/unit/test_nfc_auth.py b/tests/unit/test_nfc_auth.py new file mode 100644 index 0000000..20206ae --- /dev/null +++ b/tests/unit/test_nfc_auth.py @@ -0,0 +1,93 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Unit tests for the CBT helpers in ``nfc_auth``.""" + +from unittest import mock + +import pytest +from pyVmomi import vim + +from openvixdisklib import nfc_auth + + +def _virtual_disk(key: int, change_id: str | None = None) -> vim.vm.device.VirtualDisk: + disk = vim.vm.device.VirtualDisk() + disk.key = key + backing = vim.vm.device.VirtualDisk.FlatVer2BackingInfo() + if change_id is not None: + backing.changeId = change_id + disk.backing = backing + return disk + + +def _fake_vm(devices: list) -> mock.Mock: + vm = mock.Mock(_moId="vm-1") + vm.config.hardware.device = devices + return vm + + +class TestDiskChangeId: + def test_returns_change_id_for_matching_device(self) -> None: + """The changeId on the matching VirtualDisk's backing is returned.""" + disk = _virtual_disk(key=2000, change_id="52 aa/1") + vm = _fake_vm([disk]) + assert nfc_auth.disk_change_id(vm, 2000) == "52 aa/1" + + def test_no_matching_device_key_raises(self) -> None: + """A device key not present on the VM raises ValueError.""" + vm = _fake_vm([_virtual_disk(key=2000, change_id="52 aa/1")]) + with pytest.raises(ValueError, match="no VirtualDisk with device key 9999"): + nfc_auth.disk_change_id(vm, 9999) + + def test_empty_change_id_raises(self) -> None: + """A matching disk with no changeId yet (CBT not active) raises.""" + vm = _fake_vm([_virtual_disk(key=2000, change_id=None)]) + with pytest.raises(ValueError, match="has no changeId yet"): + nfc_auth.disk_change_id(vm, 2000) + + +class TestQueryChangedDiskAreas: + def test_converts_result_to_dataclasses(self) -> None: + """QueryChangedDiskAreas's result is converted to plain dataclasses.""" + vm = mock.Mock() + vm.QueryChangedDiskAreas.return_value = mock.Mock( + startOffset=0, + length=10737418240, + changedArea=[ + mock.Mock(start=0, length=65536), + mock.Mock(start=2555904, length=65536), + ], + ) + snapshot = mock.Mock() + + result = nfc_auth.query_changed_disk_areas(vm, snapshot, 2000, "52 aa/1") + + vm.QueryChangedDiskAreas.assert_called_once_with( + snapshot=snapshot, deviceKey=2000, startOffset=0, changeId="52 aa/1" + ) + assert result == nfc_auth.ChangedDiskAreas( + start_offset=0, + length=10737418240, + changed_areas=( + nfc_auth.ChangedExtent(start=0, length=65536), + nfc_auth.ChangedExtent(start=2555904, length=65536), + ), + ) + + def test_passes_through_start_offset_and_wildcard_change_id(self) -> None: + """A non-zero start_offset and change_id='*' are passed through as-is.""" + vm = mock.Mock() + vm.QueryChangedDiskAreas.return_value = mock.Mock( + startOffset=1024, length=0, changedArea=[] + ) + snapshot = mock.Mock() + + result = nfc_auth.query_changed_disk_areas( + vm, snapshot, 2000, "*", start_offset=1024 + ) + + vm.QueryChangedDiskAreas.assert_called_once_with( + snapshot=snapshot, deviceKey=2000, startOffset=1024, changeId="*" + ) + assert result.changed_areas == () diff --git a/tests/unit/test_nfc_open.py b/tests/unit/test_nfc_open.py new file mode 100644 index 0000000..4bf8ad1 --- /dev/null +++ b/tests/unit/test_nfc_open.py @@ -0,0 +1,215 @@ +# Copyright 2026 Cloudbase Solutions Srl +# All Rights Reserved. + +"""Unit tests for the OPEN_FILE reply parsing in ``nfc_open``.""" + +import struct + +import pytest + +from openvixdisklib import nfc_open + + +class _FakeSocket: + """A minimal socket stand-in that replays scripted bytes for recv_into.""" + + def __init__(self, replies: bytes) -> None: + self._replies = replies + self.sent: list[bytes] = [] + + def sendall(self, data: bytes) -> None: + self.sent.append(bytes(data)) + + def recv_into(self, buf: memoryview) -> int: + n = min(len(buf), len(self._replies)) + buf[:n] = self._replies[:n] + self._replies = self._replies[n:] + return n + + +def _open_reply_body( + handle: int = 0x1234, + file_type: int = nfc_open.NFC_DISK, + flags: int = nfc_open.NFC_OPEN_FLAGS_READ_ONLY, + capacity_bytes: int = 17179869184, + sector_size: int = 512, + cylinders: int = 2088, + heads: int = 255, + sectors: int = 63, +) -> bytes: + """Build a synthetic 60-byte OPEN_FILE reply payload.""" + body = bytearray(60) + struct.pack_into(" None: + """Contiguous set bits become one run; gaps split into separate ones.""" + # chunks: 1,1,1,1,0,0,0,0,1,1 (10 chunks -> 2 bytes, LSB-first) + bitmap = bytes([0b00001111, 0b00000011]) + blocks = nfc_open._decode_allocated_bitmap( + bitmap, chunk_count=10, start_sector=1000, chunk_size_sectors=128 + ) + assert blocks == ( + nfc_open.AllocatedBlock(offset=1000, length=4 * 128), + nfc_open.AllocatedBlock(offset=1000 + 8 * 128, length=2 * 128), + ) + + def test_all_zero_bitmap_returns_no_blocks(self) -> None: + """A bitmap with no set bits produces an empty result.""" + blocks = nfc_open._decode_allocated_bitmap( + bytes(4), chunk_count=16, start_sector=0, chunk_size_sectors=128 + ) + assert blocks == () + + def test_run_extending_to_the_end_is_closed(self) -> None: + """A run of set bits reaching the last chunk is still reported.""" + # chunks: 0,1,1,1 (4 chunks, 1 byte; only lower nibble meaningful) + bitmap = bytes([0b00001110]) + blocks = nfc_open._decode_allocated_bitmap( + bitmap, chunk_count=4, start_sector=0, chunk_size_sectors=1 + ) + assert blocks == (nfc_open.AllocatedBlock(offset=1, length=3),) + + def test_ignores_bits_beyond_chunk_count(self) -> None: + """Padding bits past chunk_count (from 4-byte reply alignment) are unused.""" + # 2 real chunks (both set) + 2 padding bytes with garbage bits set. + bitmap = bytes([0b00000011, 0xFF, 0xFF, 0xFF]) + blocks = nfc_open._decode_allocated_bitmap( + bitmap, chunk_count=2, start_sector=0, chunk_size_sectors=128 + ) + assert blocks == (nfc_open.AllocatedBlock(offset=0, length=256),) + + +class TestQueryAllocatedBlocksValidation: + def _disk(self) -> nfc_open.NfcDisk: + return nfc_open.NfcDisk(sock=None, path="[ds] a.vmdk", handle=1, sector_size=512) + + def test_num_sectors_not_a_multiple_raises(self) -> None: + with pytest.raises(ValueError, match="num_sectors must be a multiple"): + self._disk().query_allocated_blocks(0, 100, chunk_size_sectors=128) + + def test_start_sector_not_a_multiple_raises(self) -> None: + with pytest.raises(ValueError, match="start_sector must be a multiple"): + self._disk().query_allocated_blocks(100, 128, chunk_size_sectors=128) + + +def _ddb_get_reply(op_id: int, value: bytes | None) -> bytes: + """Build a scripted DDB_GET reply: header + 16-byte body + value extra.""" + value_length = len(value) if value is not None else 0 + body = bytes(12) + struct.pack(" nfc_open.NfcDisk: + return nfc_open.NfcDisk( + sock=_FakeSocket(replies), path="[ds] a.vmdk", handle=0x1234, sector_size=512 + ) + + def test_found_key_returns_decoded_value(self) -> None: + disk = self._disk(_ddb_get_reply(op_id=0, value=b"lsilogic")) + assert disk.ddb_get("adapterType") == "lsilogic" + + def test_missing_key_returns_none(self) -> None: + disk = self._disk(_ddb_get_reply(op_id=0, value=None)) + assert disk.ddb_get("resumeConsolidateSector") is None + + def test_sends_handle_and_key_length_in_request(self) -> None: + sock = _FakeSocket(_ddb_get_reply(op_id=0, value=b"63")) + disk = nfc_open.NfcDisk( + sock=sock, path="[ds] a.vmdk", handle=0x1234, sector_size=512 + ) + disk.ddb_get("geometry.sectors") + (sent,) = sock.sent + # header(16) + handle(8) + key_len(4) + reserved(4) + key bytes + handle, key_len, reserved = struct.unpack_from(" nfc_open.NfcDisk: + # query_full_info calls ddb_get for biosCylinders, biosHeads, + # biosSectors, adapterType, uuid, in that order. + keys = [ + "geometry.biosCylinders", + "geometry.biosHeads", + "geometry.biosSectors", + "adapterType", + "uuid", + ] + replies = b"".join( + _ddb_get_reply(op_id=i, value=values.get(k)) for i, k in enumerate(keys) + ) + disk = nfc_open.NfcDisk( + sock=_FakeSocket(replies), path="[ds] a.vmdk", handle=1, sector_size=512 + ) + disk.info = nfc_open.DiskInfo( + capacity_sectors=1024, + phys_geo=nfc_open.DiskGeometry(cylinders=10, heads=20, sectors=30), + ) + return disk + + def test_combines_open_file_info_with_ddb_values(self) -> None: + disk = self._disk_with_replies( + { + "geometry.biosCylinders": b"100", + "geometry.biosHeads": b"200", + "geometry.biosSectors": b"63", + "adapterType": b"lsilogic", + "uuid": b"some-uuid", + } + ) + info = disk.query_full_info() + assert info.capacity_sectors == 1024 + assert info.phys_geo == nfc_open.DiskGeometry(cylinders=10, heads=20, sectors=30) + assert info.bios_geo == nfc_open.DiskGeometry(cylinders=100, heads=200, sectors=63) + assert info.adapter_type == "lsilogic" + assert info.uuid == "some-uuid" + + def test_missing_ddb_keys_fall_back_to_defaults(self) -> None: + disk = self._disk_with_replies({}) + info = disk.query_full_info() + assert info.bios_geo == nfc_open.DiskGeometry(cylinders=0, heads=0, sectors=0) + assert info.adapter_type is None + assert info.uuid is None + + +class TestParseOpenReply: + def test_parses_handle_capacity_and_geometry(self) -> None: + """Capacity (offset 28, bytes) and physGeo (40/44/48) are extracted.""" + body = _open_reply_body() + handle, sector_size, info = nfc_open._parse_open_reply(body) + assert handle == 0x1234 + assert sector_size == 512 + assert info.capacity_sectors == 17179869184 // 512 + assert info.phys_geo == nfc_open.DiskGeometry( + cylinders=2088, heads=255, sectors=63 + ) + + def test_zero_sector_size_falls_back_and_still_divides_capacity(self) -> None: + """A zero sector_size falls back to NFC_SECTOR_SIZE for both uses.""" + body = _open_reply_body(sector_size=0, capacity_bytes=1024 * 512) + _handle, sector_size, info = nfc_open._parse_open_reply(body) + assert sector_size == nfc_open.NFC_SECTOR_SIZE + assert info.capacity_sectors == (1024 * 512) // nfc_open.NFC_SECTOR_SIZE + + def test_wrong_file_type_raises(self) -> None: + """A non-NFC_DISK file type is rejected.""" + body = _open_reply_body(file_type=99) + with pytest.raises(nfc_open.NfcProtocolError, match="file type 99"): + nfc_open._parse_open_reply(body) + + def test_short_body_raises(self) -> None: + """A reply shorter than the physGeo fields is rejected.""" + with pytest.raises(nfc_open.NfcProtocolError, match="too short"): + nfc_open._parse_open_reply(bytes(40))