diff --git a/test/integration/plugins/ontap/README.md b/test/integration/plugins/ontap/README.md index 2adbf449af28..09376c840c72 100644 --- a/test/integration/plugins/ontap/README.md +++ b/test/integration/plugins/ontap/README.md @@ -43,7 +43,7 @@ for the deployment, health-gate, and artifact contracts. test/integration/plugins/ontap/ ├── ontap.cfg # Environment config (IPs, credentials, zone info) ├── ontap_test_base.py # Shared base class and ONTAP REST client -├── TEST_CASES.md # Full test case reference table (72 tests) +├── TEST_CASES.md # Full test case reference table (93 tests) ├── README.md # This file │ ├── nfs3/ @@ -84,7 +84,7 @@ The ONTAP plugin (`plugins/storage/volume/ontap/`) integrates CloudStack's prima | Aspect | NFS3 | iSCSI | |--------|------|-------| -| ONTAP object per pool | FlexVol + export policy | FlexVol + igroup per KVM host | +| ONTAP object per pool | FlexVol + export policy | FlexVol; igroups are shared per KVM host and SVM | | ONTAP object per CS volume | None (FlexVol is shared) | One LUN inside the FlexVol | | Host connectivity | NFS mount | iSCSI login (IQN-based) | | Volume detach from running VM | Works via virtio hot-unplug | Requires KVM guest to support SCSI hot-unplug | @@ -270,14 +270,16 @@ pool = self.__class__.pool self.pool = pool ``` -**Guard assertion at the start of every test (except test_01)** +**Guard assertion at the start of every test that depends on a previous one** -Every test after the first starts with an assertion that the previous step's resource exists. This produces a clear, readable failure message instead of a confusing `AttributeError`: +Every test in a sequential workflow starts with an assertion that the previous step's resource exists. This produces a clear, readable failure message instead of a confusing `AttributeError`: ```python -def test_03_enable_storage_pool(self): - self.assertIsNotNone(self.__class__.pool, "Pool absent — test_01 must pass first") +def test_05_enable_storage_pool(self): + self.assertIsNotNone(self.__class__.pool, "Pool absent — test_03 must pass first") ``` +Isolated negative tests own everything they create, so they carry no such guard. In the pool lifecycle and zone-scoped suites the two create-rejection negatives are numbered `test_01` and `test_02` so that a misconfigured SVM fails within a minute rather than after the full workflow. + **Creating a storage pool — always use indexed `details[N].key` syntax** The CloudStack API for `createStoragePool` requires plugin details to be passed as indexed parameters. **Never call `StoragePool.create()` directly** — it does not support this syntax: @@ -296,6 +298,15 @@ result = self._poll_pool_state(pool.id, "Maintenance", timeout=120) self.assertEqual(result.state, "Maintenance") ``` +**Shared iSCSI igroups** + +iSCSI igroups are named from the host UUID and SVM, not from the storage pool. +Each iSCSI suite snapshots existing host igroups during `setUpClass()` and +checks that pool-only operations preserve that baseline. Tests may therefore +run while another ONTAP iSCSI pool uses the same SVM. The negative tests that +deliberately delete igroups still require exclusive SVM use and skip when +another ONTAP pool is present. + --- ## Shared base — `ontap_test_base.py` @@ -321,7 +332,7 @@ self.assertEqual(result.state, "Maintenance") | `get_data_lifs(svm_name)` | NFS data LIF count | NFS3 pool lifecycle | | `get_igroup(svm_name, name)` | iSCSI igroup existence and initiator list | iSCSI suites | | `list_luns_in_volume(svm_name, vol_name)` | LUNs present in a FlexVol | iSCSI volume/instance suites | -| `list_lun_maps_for_volume(svm_name, vol_name)` | Active LUN-maps for a volume | iSCSI instance suite | +| `list_lun_maps_for_volume(svm_name, vol_name)` | Active LUN-maps for a volume | iSCSI pool-with-volumes/instance suites | | `list_files_in_volume(svm_name, vol_name)` | Files inside a FlexVol | NFS3 instance suite | --- @@ -330,15 +341,15 @@ self.assertEqual(result.state, "Maintenance") | Suite | File | Tests | What it covers | |-------|------|-------|---------------| -| NFS3 Pool Lifecycle | `nfs3/pool/test_pool_lifecycle.py` | 8 | Create, disable, enable, maintenance, delete | -| NFS3 Pool with Volumes | `nfs3/pool/test_pool_with_volumes.py` | 7 | Same + live volume present; negative delete guard | -| NFS3 Zone-Scoped Pool | `nfs3/pool/test_zone_scoped_pool.py` | 4 | Zone scope — all hosts connected via `attachZone` | +| NFS3 Pool Lifecycle | `nfs3/pool/test_pool_lifecycle.py` | 12 | Existing lifecycle plus duplicate-name and aggregate-space create rejects, and empty-pool deletion with pre-deleted FlexVol/export policy | +| NFS3 Pool with Volumes | `nfs3/pool/test_pool_with_volumes.py` | 12 | Existing lifecycle plus deletion with pre-deleted FlexVol/export policy, and host cases for an inactive libvirt pool, a read-only NFS mount, and a deleted mount point | +| NFS3 Zone-Scoped Pool | `nfs3/pool/test_zone_scoped_pool.py` | 4 | Zone-scoped pool create, disable, enable, and delete | | NFS3 Volume Lifecycle | `nfs3/volume/test_volume_lifecycle.py` | 5 | Volume is metadata-only; FlexVol unchanged on delete | | NFS3 VM + Volume Attach | `nfs3/instance/test_vm_volume_attach.py` | 10 | Full VM lifecycle with hot-plug/detach; ROOT on tagged pool seeds/reuses template cache, which survives VM delete | | NFS3 Template Cache Negative | `nfs3/template/test_template_cache_negative.py` | 3 | Tag mismatch; undersized pool; out-of-band cache delete | -| iSCSI Pool Lifecycle | `iscsi/pool/test_pool_lifecycle.py` | 8 | Create, disable, enable, maintenance, delete + igroups | -| iSCSI Pool with Volumes | `iscsi/pool/test_pool_with_volumes.py` | 7 | Same + live LUN present; negative delete guard | -| iSCSI Zone-Scoped Pool | `iscsi/pool/test_zone_scoped_pool.py` | 4 | Zone scope | +| iSCSI Pool Lifecycle | `iscsi/pool/test_pool_lifecycle.py` | 12 | Existing lifecycle plus duplicate-name and aggregate-space create rejects, and empty-pool deletion with pre-deleted FlexVol/igroups | +| iSCSI Pool with Volumes | `iscsi/pool/test_pool_with_volumes.py` | 13 | Existing lifecycle plus deletion with pre-deleted FlexVol/igroups, maintenance with pre-deleted LUN maps, and host cases for a logged-out iSCSI session, an existing session during volume delete, and a replaced by-path symlink | +| iSCSI Zone-Scoped Pool | `iscsi/pool/test_zone_scoped_pool.py` | 4 | Zone-scoped pool create, disable, enable, and delete | | iSCSI Volume Lifecycle | `iscsi/volume/test_volume_lifecycle.py` | 5 | LUN created per CS volume; LUN removed on delete | | iSCSI VM + Volume Attach | `iscsi/instance/test_vm_volume_attach.py` | 10 | Full VM lifecycle; LUN-maps on VM start/stop/detach; ROOT on tagged pool seeds/reuses `cs_tmpl_*` LUN cache | | iSCSI Template Cache Negative | `iscsi/template/test_template_cache_negative.py` | 3 | Tag mismatch; undersized pool; out-of-band cache delete | diff --git a/test/integration/plugins/ontap/TEST_CASES.md b/test/integration/plugins/ontap/TEST_CASES.md index d3f2b1be58d1..22e41a567654 100644 --- a/test/integration/plugins/ontap/TEST_CASES.md +++ b/test/integration/plugins/ontap/TEST_CASES.md @@ -19,7 +19,7 @@ # ONTAP Integration Test Cases -Complete reference for all 72 test cases across 12 test suites. +Complete reference for all 93 test cases across 12 test suites. Each suite is sequential — tests must run in numbered order; each step builds on state created by the previous step. --- @@ -42,18 +42,22 @@ Each suite is sequential — tests must run in numbered order; each step builds **File:** `nfs3/pool/test_pool_lifecycle.py` **Class:** `TestOntapNFS3PrimaryStorageWorkflow` **Tag:** `nfs3_workflow` -**Total:** 8 tests | **Scope:** cluster-scoped NFS3 pool, no volumes for tests 01–06 +**Total:** 12 tests | **Scope:** cluster-scoped NFS3 pool, no-volume workflow through test 08. test 01 is isolated; test 03 reuses the pool from test 02; tests 11–12 are isolated | # | Test method | Goal | Depends on | CloudStack success criteria | ONTAP success criteria | Type | |---|-------------|------|------------|-----------------------------|------------------------|------| -| 01 | `test_01_create_primary_storage_pool` | Create a cluster-scoped NFS3 primary storage pool | setUpClass (zone, cluster, account) | `pool.state == "Up"`, `pool.type == "NetworkFilesystem"`, `nfsmountopts` contains `vers=3` | FlexVol exists and `state == "online"`, export policy exists with each cluster host IP as a rule, at least one NFS data LIF present on SVM | positive | -| 02 | `test_02_disable_storage_pool` | Disable the pool (admin operation) | test_01 (`pool`) | `pool.state == "Disabled"` | FlexVol still `online`; export policy still present | positive | -| 03 | `test_03_enable_storage_pool` | Re-enable the pool | test_02 | `pool.state == "Up"` | FlexVol still `online`; export policy still present | positive | -| 04 | `test_04_enter_maintenance_mode` | Put pool into maintenance (drains new volume allocations) | test_03 | `pool.state == "Maintenance"` | FlexVol still `online`; export policy still present (maintenance is CS-only state) | positive | -| 05 | `test_05_cancel_maintenance_mode` | Cancel maintenance, return pool to service | test_04 | `pool.state == "Up"` | FlexVol still `online`; export policy still present | positive | -| 06 | `test_06_delete_pool_from_maintenance` | Enter maintenance then permanently delete the pool | test_05 | Pool no longer returned by `listStoragePools` (CS 431 error expected on ID lookup) | FlexVol deleted (not found by `GET /api/storage/volumes?name=`); export policy deleted | positive | -| 07 | `test_07_create_volume_on_pool` | Create a second fresh pool and allocate a CloudStack data volume on it | test_06 (pool deleted; creates new pool) | New `pool.state == "Up"`; `createVolume` returns non-None volume object | FlexVol `online` after volume allocation; export policy present | positive | -| 08 | `test_08_delete_volume_and_pool` | Delete the volume then force-delete the pool | test_07 (`pool`, `volume`) | Volume no longer listed; pool no longer listed | FlexVol deleted; export policy deleted | positive | +| 01 | `test_01_reject_create_when_no_aggregate_space` | Reject pool creation when requested capacity exceeds every assigned online aggregate's available space | isolated | `CloudstackAPIException` containing `No suitable aggregates`; no CS pool created | No FlexVol created | negative | +| 02 | `test_02_create_primary_storage_pool` | Create a cluster-scoped NFS3 primary storage pool | setUpClass (zone, cluster, account) | `pool.state == "Up"`, `pool.type == "NetworkFilesystem"`, `nfsmountopts` contains `vers=3` | FlexVol exists and `state == "online"`, export policy exists with each cluster host IP as a rule, at least one NFS data LIF present on SVM | positive | +| 03 | `test_03_reject_create_when_flexvol_name_exists` | Reject a second pool create when the FlexVol from test 02 already exists on ONTAP | test_02 | `CloudstackAPIException`; only the test_02 pool remains, still `Up` | Existing FlexVol stays `online`; export policy still present | negative | +| 04 | `test_04_disable_storage_pool` | Disable the pool (admin operation) | test_06 | `pool.state == "Disabled"` | FlexVol still `online`; export policy still present | positive | +| 05 | `test_05_enable_storage_pool` | Re-enable the pool | test_08 | `pool.state == "Up"` | FlexVol still `online`; export policy still present | positive | +| 06 | `test_06_enter_maintenance_mode` | Put pool into maintenance (drains new volume allocations) | test_05 | `pool.state == "Maintenance"` | FlexVol still `online`; export policy still present (maintenance is CS-only state) | positive | +| 07 | `test_07_cancel_maintenance_mode` | Cancel maintenance, return pool to service | test_11 | `pool.state == "Up"` | FlexVol still `online`; export policy still present | positive | +| 08 | `test_08_delete_pool_from_maintenance` | Enter maintenance then permanently delete the original pool | test_07 | Pool no longer returned by `listStoragePools` (CS 431 error expected on ID lookup) | FlexVol and export policy deleted | positive | +| 09 | `test_09_create_volume_on_pool` | Create a fresh pool and allocate a CloudStack data volume | test_08 | New pool is `Up`; `createVolume` returns a volume | FlexVol `online`; export policy present | positive | +| 10 | `test_10_delete_volume_and_pool` | Detach and destroy the VM, delete the volume, then force-delete the pool | test_17 | VM destroyed; volume and pool no longer listed | FlexVol and export policy deleted | cleanup | +| 11 | `test_11_delete_pool_with_flexvol_predeleted` | Delete an empty pool after its FlexVol was removed directly from ONTAP | isolated | Pool removed successfully | FlexVol remains absent; export policy cleaned up | negative | +| 12 | `test_12_delete_pool_with_export_policy_predeleted` | Delete an empty pool after its export policy was removed directly from ONTAP | isolated | Pool removed successfully | FlexVol deleted; export policy remains absent | negative | --- @@ -62,7 +66,7 @@ Each suite is sequential — tests must run in numbered order; each step builds **File:** `nfs3/pool/test_pool_with_volumes.py` **Class:** `TestOntapNFS3PoolWithVolumes` **Tag:** `nfs3_with_volumes` -**Total:** 7 tests | **Scope:** cluster-scoped NFS3 pool with a live CloudStack volume throughout +**Total:** 12 tests | **Scope:** cluster-scoped NFS3 pool with a live CloudStack volume, plus isolated negative workflows | # | Test method | Goal | Depends on | CloudStack success criteria | ONTAP success criteria | Type | |---|-------------|------|------------|-----------------------------|------------------------|------| @@ -73,6 +77,11 @@ Each suite is sequential — tests must run in numbered order; each step builds | 05 | `test_05_cancel_maintenance_with_volume` | Cancel maintenance with volume — verifies the NFS3 cancel-maintenance fix | test_04 | `pool.state == "Up"`; volume still listed | FlexVol still `online` | positive | | 06 | `test_06_forced_false_delete_rejected` | Attempt to delete pool (forced=False) with volume present — must be rejected | test_05 | `deleteStoragePool(forced=False)` raises `CloudstackAPIException`; pool still listed in `Maintenance` state | FlexVol still `online`; no ONTAP objects removed | negative | | 07 | `test_07_force_delete_pool_and_cleanup` | Cancel maintenance, delete volume, then force-delete pool | test_06 | Pool no longer listed; volume no longer listed | FlexVol deleted; export policy deleted | cleanup | +| 08 | `test_08_delete_pool_with_volume_flexvol_missing` | Force-delete a pool with a CS volume after its FlexVol was removed directly from ONTAP | isolated | Pool removed; leftover volume record cleaned | FlexVol remains absent | negative | +| 09 | `test_09_delete_pool_with_volume_export_policy_missing` | Force-delete a pool with a CS volume after its export policy was removed directly from ONTAP | isolated | Pool removed; leftover volume record cleaned | FlexVol deleted; export policy remains absent | negative | +| 10 | `test_10_libvirt_pool_inactive` | Libvirt pool for this storage pool is inactive on the KVM host | isolated | Pool reaches Maintenance; volume remains | FlexVol stays online; export policy remains | negative | +| 11 | `test_11_nfs_mount_read_only` | NFS mount for this pool is read-only on the KVM host | isolated | Pool reaches Maintenance; writes to the mount fail | FlexVol stays online; export policy remains | negative | +| 12 | `test_12_nfs_mount_point_deleted` | NFS mount point for this pool is missing on the KVM host | isolated | Pool reaches Maintenance with its mount point absent | FlexVol stays online; export policy remains | negative | --- @@ -86,7 +95,7 @@ Each suite is sequential — tests must run in numbered order; each step builds | # | Test method | Goal | Depends on | CloudStack success criteria | ONTAP success criteria | Type | |---|-------------|------|------------|-----------------------------|------------------------|------| | 01 | `test_01_create_zone_scoped_pool` | Create a zone-scoped NFS3 pool; CloudStack calls `attachZone()` to connect all eligible KVM hosts | setUpClass | `pool.state == "Up"` | FlexVol `online`; export policy exists and contains **every** cluster host IP; at least one NFS data LIF present | positive | -| 02 | `test_02_disable_zone_scoped_pool` | Disable the zone-scoped pool | test_01 (`pool`) | `pool.state == "Disabled"` | FlexVol unchanged; export policy unchanged | positive | +| 02 | `test_02_disable_zone_scoped_pool` | Disable the zone-scoped pool | test_01 | `pool.state == "Disabled"` | FlexVol unchanged; export policy unchanged | positive | | 03 | `test_03_enable_zone_scoped_pool` | Re-enable the zone-scoped pool | test_02 | `pool.state == "Up"` | FlexVol unchanged; export policy unchanged | positive | | 04 | `test_04_delete_zone_scoped_pool` | Enter maintenance and force-delete the zone-scoped pool | test_03 | Pool no longer listed | FlexVol deleted; export policy deleted | positive | @@ -136,18 +145,22 @@ Each suite is sequential — tests must run in numbered order; each step builds **File:** `iscsi/pool/test_pool_lifecycle.py` **Class:** `TestOntapISCSIPoolLifecycle` **Tag:** `iscsi_workflow` -**Total:** 8 tests | **Scope:** cluster-scoped iSCSI pool, no volumes for tests 01–06 +**Total:** 12 tests | **Scope:** cluster-scoped iSCSI pool, no-volume workflow through test 08. test 01 is isolated; test 03 reuses the pool from test 02; tests 11–12 are isolated | # | Test method | Goal | Depends on | CloudStack success criteria | ONTAP success criteria | Type | |---|-------------|------|------------|-----------------------------|------------------------|------| -| 01 | `test_01_create_primary_storage_pool` | Create a cluster-scoped iSCSI primary storage pool | setUpClass | `pool.state == "Up"`, `pool.type == "OntapiSCSI"` | FlexVol `online`; one igroup per cluster host (named `cs_{svmName}_{hostShortName}`) with host IQN as initiator | positive | -| 02 | `test_02_disable_storage_pool` | Disable the pool | test_01 (`pool`) | `pool.state == "Disabled"` | FlexVol still `online` | positive | -| 03 | `test_03_enable_storage_pool` | Re-enable the pool | test_02 | `pool.state == "Up"` | FlexVol still `online` | positive | -| 04 | `test_04_enter_maintenance_mode` | Put pool into maintenance | test_03 | `pool.state == "Maintenance"` | FlexVol still `online`; igroups unchanged | positive | -| 05 | `test_05_cancel_maintenance_mode` | Cancel maintenance | test_04 | `pool.state == "Up"` | FlexVol still `online` | positive | -| 06 | `test_06_enter_maintenance_and_delete_pool` | Enter maintenance then force-delete the pool | test_05 | Pool no longer listed | FlexVol deleted; all igroups for cluster hosts deleted | positive | -| 07 | `test_07_create_volume_on_pool` | Create a second fresh pool and allocate a CloudStack data volume (creates a LUN) | test_06 (new pool) | New `pool.state == "Up"`; volume object non-None | FlexVol `online`; ≥1 LUN present inside FlexVol (`list_luns_in_volume`) | positive | -| 08 | `test_08_delete_volume_and_pool` | Delete the volume (removes LUN), enter maintenance, force-delete pool | test_07 (`pool`, `volume`) | Volume no longer listed; pool no longer listed | LUN no longer in FlexVol; FlexVol deleted; igroups deleted | positive | +| 01 | `test_01_reject_create_when_no_aggregate_space` | Reject pool creation when requested capacity exceeds every assigned online aggregate's available space | isolated | `CloudstackAPIException` containing `No suitable aggregates`; no CS pool created | No FlexVol created | negative | +| 02 | `test_02_create_primary_storage_pool` | Create a cluster-scoped iSCSI primary storage pool | setUpClass | `pool.state == "Up"`, `pool.type == "Iscsi"` | FlexVol `online`; shared `cs_{hostUuid}_{svmName}` igroups unchanged from suite-start baseline | positive | +| 03 | `test_03_reject_create_when_flexvol_name_exists` | Reject a second pool create when the FlexVol from test 02 already exists on ONTAP | test_02 | `CloudstackAPIException`; only the test_02 pool remains, still `Up` | Existing FlexVol stays `online` | negative | +| 04 | `test_04_disable_storage_pool` | Disable the pool | test_06 | `pool.state == "Disabled"` | FlexVol still `online` | positive | +| 05 | `test_05_enable_storage_pool` | Re-enable the pool | test_08 | `pool.state == "Up"` | FlexVol still `online` | positive | +| 06 | `test_06_enter_maintenance_mode` | Put pool into maintenance | test_05 | `pool.state == "Maintenance"` | FlexVol still `online`; igroups unchanged | positive | +| 07 | `test_07_cancel_maintenance_mode` | Cancel maintenance | test_11 | `pool.state == "Up"` | FlexVol still `online` | positive | +| 08 | `test_08_enter_maintenance_and_delete_pool` | Enter maintenance then delete the original pool | test_07 | Pool no longer listed | FlexVol and test-pool LUN maps deleted; shared igroup baseline restored | positive | +| 09 | `test_09_create_volume_on_pool` | Create a fresh pool and allocate a CloudStack volume | test_08 | New pool is `Up`; `createVolume` returns a volume | FlexVol `online`; at least one LUN present | positive | +| 10 | `test_10_delete_volume_and_pool` | Detach and destroy the VM, delete the volume, then force-delete the pool | test_17 | VM destroyed; volume and pool no longer listed | LUN, FlexVol, and test-pool maps deleted; shared igroup baseline restored | cleanup | +| 11 | `test_11_delete_pool_with_flexvol_predeleted` | Delete an empty pool after its FlexVol was removed directly from ONTAP | isolated | Pool removed successfully | FlexVol remains absent; igroups cleaned up | negative | +| 12 | `test_12_delete_pool_with_igroups_predeleted` | Delete an empty pool after host igroups were removed directly from ONTAP | isolated | Pool removed successfully | FlexVol deleted; igroups remain absent | negative | --- @@ -156,7 +169,7 @@ Each suite is sequential — tests must run in numbered order; each step builds **File:** `iscsi/pool/test_pool_with_volumes.py` **Class:** `TestOntapISCSIPoolWithVolumes` **Tag:** `iscsi_workflow` -**Total:** 7 tests | **Scope:** cluster-scoped iSCSI pool with a live CloudStack volume (LUN) throughout +**Total:** 13 tests | **Scope:** cluster-scoped iSCSI pool with a live CloudStack volume (LUN), plus isolated negative workflows | # | Test method | Goal | Depends on | CloudStack success criteria | ONTAP success criteria | Type | |---|-------------|------|------------|-----------------------------|------------------------|------| @@ -164,9 +177,15 @@ Each suite is sequential — tests must run in numbered order; each step builds | 02 | `test_02_disable_pool_volume_survives` | Disable pool with volume present | test_01 (`pool`, `volume`) | `pool.state == "Disabled"`; volume still listed | FlexVol still `online`; LUN still present | positive | | 03 | `test_03_enable_pool_volume_intact` | Re-enable pool with volume | test_02 | `pool.state == "Up"`; volume still listed | FlexVol still `online`; LUN still present | positive | | 04 | `test_04_enter_maintenance_volume_present` | Enter maintenance with volume | test_03 | `pool.state == "Maintenance"`; volume still listed | FlexVol still `online`; LUN still present | positive | -| 05 | `test_05_cancel_maintenance_volume_present` | Cancel maintenance with volume (TDS iSCSI cancel maintenance) | test_04 | `pool.state == "Up"`; volume still listed | FlexVol still `online`; LUN still present | positive | +| 05 | `test_05_cancel_maintenance_volume_present` | Cancel maintenance with volume | test_04 | `pool.state == "Up"`; volume still listed | FlexVol still `online`; LUN still present | positive | | 06 | `test_06_forced_false_delete_rejected` | Attempt `deleteStoragePool(forced=False)` with LUN-backed volume present — must be rejected | test_05 | `CloudstackAPIException` raised; pool still in `Maintenance` | No ONTAP objects removed | negative | -| 07 | `test_07_delete_volume_and_force_delete_pool` | Delete volume (LUN removed) then force-delete pool | test_06 (`pool`, `volume`) | Volume gone; pool gone | LUN removed; FlexVol deleted; igroups deleted | cleanup | +| 07 | `test_07_delete_volume_and_force_delete_pool` | Delete volume (LUN removed) then force-delete pool | test_06 (`pool`, `volume`) | Volume gone; pool gone | LUN and FlexVol deleted; shared igroup baseline restored | cleanup | +| 08 | `test_08_delete_pool_with_volume_flexvol_missing` | Force-delete a pool with a CS volume after its FlexVol and LUN were removed directly | isolated | Pool removed; leftover volume record cleaned | FlexVol and LUN remain absent | negative | +| 09 | `test_09_delete_pool_with_volume_igroups_missing` | Force-delete a pool with a CS volume after host igroups were removed directly | isolated | Pool removed; leftover volume record cleaned | FlexVol deleted; igroups remain absent | negative | +| 10 | `test_10_enter_maintenance_lun_maps_predeleted` | Enter maintenance after LUN maps were removed directly on ONTAP | isolated | Pool reaches Maintenance | LUN maps remain absent | negative | +| 11 | `test_11_iscsi_session_logged_out` | Enter maintenance after only the test iSCSI session is logged out | isolated | Pool reaches Maintenance; volume remains | Test LUN remains | negative | +| 12 | `test_12_delete_volume_with_existing_iscsi_session` | Delete the test volume while its iSCSI session is already logged in | isolated | Volume is removed; no extra session is created | Test LUN is removed | negative | +| 13 | `test_13_corrupt_iscsi_by_path` | Delete the test volume after its by-path symlink is replaced with a regular file | isolated | Volume is removed; the planted file remains | Test LUN is removed | negative | --- @@ -179,10 +198,10 @@ Each suite is sequential — tests must run in numbered order; each step builds | # | Test method | Goal | Depends on | CloudStack success criteria | ONTAP success criteria | Type | |---|-------------|------|------------|-----------------------------|------------------------|------| -| 01 | `test_01_create_zone_scoped_pool` | Create a zone-scoped iSCSI pool; CS calls `attachZone()` to connect all eligible KVM hosts | setUpClass | `pool.state == "Up"` | FlexVol `online`; igroup per cluster host, each with host IQN as initiator | positive | -| 02 | `test_02_disable_zone_scoped_pool` | Disable pool | test_01 (`pool`) | `pool.state == "Disabled"` | FlexVol unchanged; igroups unchanged | positive | +| 01 | `test_01_create_zone_scoped_pool` | Create a zone-scoped iSCSI pool | setUpClass | `pool.state == "Up"` | FlexVol `online`; shared host igroups unchanged from suite-start baseline | positive | +| 02 | `test_02_disable_zone_scoped_pool` | Disable pool | test_01 | `pool.state == "Disabled"` | FlexVol unchanged; igroups unchanged | positive | | 03 | `test_03_enable_zone_scoped_pool` | Re-enable pool | test_02 | `pool.state == "Up"` | FlexVol unchanged; igroups unchanged | positive | -| 04 | `test_04_delete_zone_scoped_pool` | Enter maintenance then delete pool | test_03 | Pool no longer listed | FlexVol deleted; all igroups deleted | positive | +| 04 | `test_04_delete_zone_scoped_pool` | Enter maintenance then delete pool | test_03 | Pool no longer listed | FlexVol and test-pool maps deleted; shared igroup baseline restored | positive | --- @@ -199,7 +218,7 @@ Each suite is sequential — tests must run in numbered order; each step builds | 02 | `test_02_delete_volume` | Delete the volume — the LUN is removed from the FlexVol | test_01 (`pool`, `volume`) | Volume no longer listed | LUN no longer in FlexVol; FlexVol itself still `online` | positive | | 03 | `test_03_recreate_volume_for_delete_tests` | Re-create a volume (LUN re-created) — setup for negative tests | test_02 | New volume non-None | LUN present in FlexVol again | positive | | 04 | `test_04_forced_false_delete_with_volume_fails` | Enter maintenance then attempt `deleteStoragePool(forced=False)` with LUN present — must be rejected | test_03 (`pool`, `volume`) | `CloudstackAPIException` raised; pool still in `Maintenance` | No ONTAP objects removed | negative | -| 05 | `test_05_delete_volume_and_force_delete_pool` | Delete volume (LUN removed) then force-delete pool | test_04 | Volume gone; pool gone | LUN removed; FlexVol deleted; igroups deleted | positive | +| 05 | `test_05_delete_volume_and_force_delete_pool` | Delete volume (LUN removed) then force-delete pool | test_04 | Volume gone; pool gone | LUN and FlexVol deleted; shared igroup baseline restored | positive | --- @@ -223,7 +242,7 @@ Each suite is sequential — tests must run in numbered order; each step builds | 07 | `test_07_detach_volume_from_vm` | Hot-detach the iSCSI volume from the running VM (TDS Detach iSCSI) | test_06 (`vm`, `volume`) | `volume.virtualmachineid` cleared | 0 LUN-maps; LUN still in FlexVol | positive ⚠️ | | 08 | `test_08_destroy_vm_and_cleanup` | Destroy VM (expunge), delete volume, enter maintenance, delete pool | test_07 | VM gone; spool_ref still Ready after VM expunge; volume gone; pool gone | `cs_tmpl_*` present after VM expunge; FlexVol deleted; all LUNs and igroups deleted | cleanup | -> ⚠️ **test_07 known status:** iSCSI hot-detach from a running VM relies on the KVM guest acknowledging the SCSI device removal. On this environment the guest does not acknowledge in time, causing CloudStack error 530. This is a KVM-host-level or guest-template limitation, not a test code defect. All other 61 tests pass. +> ⚠️ **test_07 known status:** iSCSI hot-detach from a running VM relies on the KVM guest acknowledging the SCSI device removal. On this environment the guest does not acknowledge in time, causing CloudStack error 530. This is a KVM-host-level or guest-template limitation, not a test code defect. --- @@ -261,16 +280,16 @@ Each suite is sequential — tests must run in numbered order; each step builds | Suite | Protocol | Scope | Tests | Status | |-------|---------|-------|-------|--------| -| NFS3 Pool Lifecycle | NFS3 | Cluster | 8 | ✅ | -| NFS3 Pool with Volumes | NFS3 | Cluster | 7 | ✅ | +| NFS3 Pool Lifecycle | NFS3 | Cluster | 12 | ✅ | +| NFS3 Pool with Volumes | NFS3 | Cluster | 12 | ✅ | | NFS3 Zone-Scoped Pool | NFS3 | Zone | 4 | ✅ | | NFS3 Volume Lifecycle | NFS3 | Cluster | 5 | ✅ | | NFS3 VM + Volume Attach | NFS3 | Cluster | 10 | 🆕 +2 template cache | | NFS3 Template Cache Negative | NFS3 | Cluster | 3 | 🆕 | -| iSCSI Pool Lifecycle | iSCSI | Cluster | 8 | ✅ | -| iSCSI Pool with Volumes | iSCSI | Cluster | 7 | ✅ | +| iSCSI Pool Lifecycle | iSCSI | Cluster | 12 | ✅ | +| iSCSI Pool with Volumes | iSCSI | Cluster | 13 | ✅ | | iSCSI Zone-Scoped Pool | iSCSI | Zone | 4 | ✅ | | iSCSI Volume Lifecycle | iSCSI | Cluster | 5 | ✅ | | iSCSI VM + Volume Attach | iSCSI | Cluster | 10 | ⚠️ 7/8 + 🆕 2 template cache | | iSCSI Template Cache Negative | iSCSI | Cluster | 3 | 🆕 | -| **Total** | | | **72** | | +| **Total** | | | **93** | **92 passing** | diff --git a/test/integration/plugins/ontap/iscsi/pool/test_pool_lifecycle.py b/test/integration/plugins/ontap/iscsi/pool/test_pool_lifecycle.py index cc87bacf0e76..bc387500b479 100644 --- a/test/integration/plugins/ontap/iscsi/pool/test_pool_lifecycle.py +++ b/test/integration/plugins/ontap/iscsi/pool/test_pool_lifecycle.py @@ -19,18 +19,22 @@ Sequential workflow integration tests for NetApp ONTAP iSCSI primary storage pool lifecycle (no volumes). -Tests are numbered test_01 ... test_08 and must run in that order. Each step +Tests are numbered test_01 ... test_12 and must run in that order. Each step builds on the shared state established by the previous step. Workflow: - 01 Create primary storage pool - 02 Disable storage pool - 03 Enable storage pool - 04 Enter maintenance mode - 05 Cancel maintenance mode - 06 Enter maintenance mode and delete the storage pool - 07 Create a new pool and allocate a CloudStack data volume (LUN created) - 08 Delete the volume (LUN removed), enter maintenance, force-delete pool + 01 Reject create when no online assigned aggregate has enough free space + 02 Create primary storage pool + 03 Reject create when a FlexVol of that name already exists on ONTAP + 04 Disable storage pool + 05 Enable storage pool + 06 Enter maintenance mode + 07 Cancel maintenance mode + 08 Enter maintenance mode and delete the storage pool + 09 Create a new pool and allocate a CloudStack data volume (LUN created) + 10 Delete the volume (LUN removed), enter maintenance, force-delete pool + 11 Delete an empty pool whose FlexVol was deleted directly on ONTAP + 12 Delete an empty pool whose igroups were deleted on ONTAP Prerequisites: - CloudStack management server with the NetApp ONTAP plugin deployed @@ -55,12 +59,11 @@ from nose.plugins.attrib import attr from marvin.cloudstackAPI import ( - cancelStorageMaintenance, createStoragePool as createStoragePoolAPI, deleteVolume as deleteVolumeAPI, - enableStorageMaintenance, updateStoragePool as updateStoragePoolAPI, ) +from marvin.cloudstackException import CloudstackAPIException from marvin.lib.base import StoragePool from marvin.lib.common import list_storage_pools @@ -88,6 +91,7 @@ class TestData: DETAIL_STORAGE_IP = "storageIP" ONTAP_MIN_VOLUME_SIZE = 1677721600 + ONTAP_MAX_VOLUME_SIZE = 300 * 1024 ** 4 def __init__(self, storage_ip, svm_name, username, password, scope="CLUSTER", provider="NetApp ONTAP", @@ -191,10 +195,14 @@ def setUpClass(cls): # Helpers # ------------------------------------------------------------------ - def _create_pool(self): + def _create_pool(self, pool_name=None, capacitybytes=None): + """Create a pool; name and capacity default to the suite's values.""" ps = self.testdata[TestData.primaryStorage] storage_ip = self.testdata[TestData.ontap][TestData.DETAIL_STORAGE_IP] - pool_name = "OntapISCSI_%d" % random.randint(0, 99999) + if pool_name is None: + pool_name = "OntapISCSI_%d" % random.randint(0, 99999) + if capacitybytes is None: + capacitybytes = ps["capacitybytes"] cmd = createStoragePoolAPI.createStoragePoolCmd() cmd.name = pool_name @@ -205,7 +213,7 @@ def _create_pool(self): cmd.scope = ps[TestData.scope] cmd.provider = ps[TestData.provider] cmd.tags = ps[TestData.tags] - cmd.capacitybytes = ps["capacitybytes"] + cmd.capacitybytes = capacitybytes cmd.hypervisor = "KVM" cmd.managed = True @@ -269,11 +277,80 @@ def _assert_pool_capacity(self, pool, label): ) # ------------------------------------------------------------------ - # Step 01 - Create primary storage pool + # Step 01 - Reject create when no aggregate has enough free space + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_workflow"], required_hardware=True) + def test_01_reject_create_when_no_aggregate_space(self): + """ + Ask for 1 GiB more than the largest online aggregate assigned to the + SVM can provide, so no aggregate qualifies and the plugin refuses + before creating anything. + + Verifies: + - createStoragePool raises CloudstackAPIException + - the error names the aggregate shortage rather than some other + failure ('No suitable aggregates') + - no pool is left in CloudStack and no FlexVol on ONTAP + """ + self._sweep_tracked_pool2() + max_free = self.ontap.max_online_aggregate_available_bytes( + self.svm_name + ) + if not max_free: + self.skipTest( + "No online aggregate with reported free space is assigned to " + "SVM '%s'; cannot build an unsatisfiable request" + % self.svm_name + ) + requested = int(max_free) + 1024 ** 3 + if requested > TestData.ONTAP_MAX_VOLUME_SIZE: + self.skipTest( + "Largest aggregate free space (%d B) + 1 GiB exceeds the " + "ONTAP FlexVol maximum (%d B); the request would be refused " + "for the size limit rather than the aggregate shortage" + % (max_free, TestData.ONTAP_MAX_VOLUME_SIZE) + ) + + pool_name = self._throwaway_pool_name("NoSpace") + log_progress( + logger, "info", + "Requesting pool '%s' of %d B; largest online aggregate on SVM " + "'%s' has %d B free (expect reject)", + pool_name, requested, self.svm_name, max_free, + ) + try: + with self.assertRaises(CloudstackAPIException) as caught: + self.__class__.pool2 = self._create_pool( + pool_name=pool_name, capacitybytes=requested + ) + error_text = str(caught.exception) + log_progress( + logger, "info", + "Rejected no-space create for '%s': %s", pool_name, error_text, + ) + self.assertIn( + "No suitable aggregates", error_text, + "Expected the rejection to report 'No suitable aggregates', " + "got: %s" % error_text, + ) + self._assert_no_pool_named(pool_name) + self.assertIsNone( + self.ontap.get_volume(pool_name), + "ONTAP FlexVol '%s' was created despite the rejected pool " + "create" % pool_name, + ) + finally: + self._cleanup_throwaway_pool( + self.__class__.pool2, flexvol_name=pool_name + ) + + # ------------------------------------------------------------------ + # Step 02 - Create primary storage pool # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_01_create_primary_storage_pool(self): + def test_02_create_primary_storage_pool(self): """ Create an iSCSI primary storage pool and verify: - CloudStack state is Up, type is OntapiSCSI @@ -327,17 +404,69 @@ def test_01_create_primary_storage_pool(self): self._assert_pool_capacity(pool, "pool-created") # ------------------------------------------------------------------ - # Step 02 - Disable storage pool + # Step 03 - Reject create when that FlexVol name already exists + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_workflow"], required_hardware=True) + def test_03_reject_create_when_flexvol_name_exists(self): + """ + The pool from test_02 already has a FlexVol of that name on ONTAP. + A second createStoragePool with the same name must be rejected, and + the existing pool and FlexVol must be left untouched. + + Verifies: + - createStoragePool raises CloudstackAPIException + - CloudStack still lists only the pool from test_02 + - the existing ONTAP FlexVol is still online + """ + pool = self.__class__.pool + self.assertIsNotNone(pool, "Pool absent - test_02 must pass first") + pool_name = pool.name + self.assertIsNotNone( + self.ontap.get_volume(pool_name), + "ONTAP FlexVol '%s' from test_02 is missing" % pool_name, + ) + log_progress( + logger, "info", + "Pool '%s' already owns a FlexVol; requesting another pool of " + "the same name (expect reject)", pool_name, + ) + duplicate = None + try: + with self.assertRaises(CloudstackAPIException) as caught: + duplicate = self._create_pool(pool_name=pool_name) + log_progress( + logger, "info", + "Rejected duplicate-name create for '%s': %s", + pool_name, caught.exception, + ) + self._assert_only_original_pool(pool) + ontap_vol = self.ontap.get_volume(pool_name) + self.assertIsNotNone( + ontap_vol, + "Existing ONTAP FlexVol '%s' was removed by the failed " + "pool create" % pool_name, + ) + self.assertEqual( + ontap_vol.get("state"), "online", + "Existing ONTAP FlexVol '%s' should still be online, got '%s'" + % (pool_name, ontap_vol.get("state")), + ) + finally: + self._warn_if_duplicate_pool(duplicate, pool) + + # ------------------------------------------------------------------ + # Step 04 - Disable storage pool # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_02_disable_storage_pool(self): + def test_04_disable_storage_pool(self): """ Disable the pool and verify: - CloudStack reports Disabled - ONTAP: FlexVol is still online (disable is a CS-only state change) """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_02 must pass first") cmd = updateStoragePoolAPI.updateStoragePoolCmd() cmd.id = self.__class__.pool.id @@ -361,13 +490,13 @@ def test_02_disable_storage_pool(self): # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_03_enable_storage_pool(self): + def test_05_enable_storage_pool(self): """ Re-enable the pool and verify: - CloudStack reports Up - ONTAP: FlexVol is still online (enable is a CS-only state change) """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_02 must pass first") cmd = updateStoragePoolAPI.updateStoragePoolCmd() cmd.id = self.__class__.pool.id @@ -391,19 +520,15 @@ def test_03_enable_storage_pool(self): # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_04_enter_maintenance_mode(self): + def test_06_enter_maintenance_mode(self): """ Put the pool into maintenance mode and verify: - CloudStack reports Maintenance - ONTAP: FlexVol is still online (maintenance is a CS-only state change) """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_02 must pass first") - cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.enableStorageMaintenance(cmd) - - result = self._poll_pool_state(self.__class__.pool.id, "Maintenance", timeout=120) + result = self._enter_maintenance(self.__class__.pool.id) self.assertEqual(result.state, "Maintenance") # ONTAP: maintenance must not touch the FlexVol @@ -420,19 +545,15 @@ def test_04_enter_maintenance_mode(self): # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_05_cancel_maintenance_mode(self): + def test_07_cancel_maintenance_mode(self): """ Cancel maintenance and verify: - CloudStack reports Up - ONTAP: FlexVol is still online """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_01 must pass first") - - cmd = cancelStorageMaintenance.cancelStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.cancelStorageMaintenance(cmd) + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_02 must pass first") - result = self._poll_pool_state(self.__class__.pool.id, "Up", timeout=120) + result = self._exit_maintenance(self.__class__.pool.id) self.assertEqual(result.state, "Up") ontap_vol = self.ontap.get_volume(self.__class__.pool.name) @@ -448,20 +569,17 @@ def test_05_cancel_maintenance_mode(self): # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_06_enter_maintenance_and_delete_pool(self): + def test_08_enter_maintenance_and_delete_pool(self): """ Enter maintenance mode then delete the pool. Verifies the pool is removed from CloudStack and the backing ONTAP FlexVol is deleted. """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_02 must pass first") pool = self.__class__.pool pool_name = pool.name - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(pool.id, "Maintenance", timeout=120) + self._enter_maintenance(pool.id) self._delete_pool(pool.id) self.__class__.pool = None @@ -497,7 +615,7 @@ def test_06_enter_maintenance_and_delete_pool(self): # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_07_create_volume_on_pool(self): + def test_09_create_volume_on_pool(self): """ Create a new iSCSI pool and allocate a CloudStack data volume. For iSCSI, createAsync creates a LUN inside the pool's ONTAP FlexVol. @@ -563,7 +681,7 @@ def test_07_create_volume_on_pool(self): # ------------------------------------------------------------------ @attr(tags=["iscsi_workflow"], required_hardware=True) - def test_08_delete_volume_and_pool(self): + def test_10_delete_volume_and_pool(self): """ Delete the volume from test_07, enter maintenance, then force-delete the pool. @@ -574,8 +692,8 @@ def test_08_delete_volume_and_pool(self): - ONTAP: FlexVol deleted - ONTAP: igroups for all cluster hosts deleted """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_07 must pass first") - self.assertIsNotNone(self.__class__.volume, "Volume absent - test_07 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_09 must pass first") + self.assertIsNotNone(self.__class__.volume, "Volume absent - test_09 must pass first") pool = self.__class__.pool pool_name = pool.name @@ -610,10 +728,7 @@ def test_08_delete_volume_and_pool(self): self._assert_pool_capacity(pool, "volume-deleted") # Enter maintenance then force-delete the pool - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(pool.id, "Maintenance", timeout=120) + self._enter_maintenance(pool.id) self._delete_pool(pool.id, forced=True) self.__class__.pool = None @@ -643,3 +758,277 @@ def test_08_delete_volume_and_pool(self): igroup, "ONTAP igroup '%s' still exists after pool deletion" % igroup_name ) + + def _throwaway_pool_name(self, suffix): + return "OntapISCSI%s_%d" % (suffix, random.randint(0, 99999)) + + + def _cleanup_throwaway_pool(self, pool, flexvol_name=None): + """Best-effort teardown for an isolated test: CloudStack, then ONTAP. + + Never raises, so a failed assertion in the test body is the error + that surfaces. + """ + if pool is not None: + try: + listed = list_storage_pools(self.apiClient, id=pool.id) + except Exception: + listed = None + if listed: + try: + if listed[0].state != "Maintenance": + self._enter_maintenance(pool.id) + except Exception as exc: + logger.warning( + "cleanup: could not put pool '%s' into Maintenance: %s", + pool.name, exc, + ) + try: + self._delete_pool(pool.id, forced=True) + except Exception as exc: + logger.warning( + "cleanup: could not delete pool '%s': %s", + pool.name, exc, + ) + if flexvol_name is None: + flexvol_name = pool.name + if flexvol_name: + try: + if self.ontap.get_volume(flexvol_name) is not None: + self.ontap.offline_and_delete_volume(flexvol_name) + except Exception as exc: + logger.warning( + "cleanup: could not delete ONTAP FlexVol '%s': %s", + flexvol_name, exc, + ) + if pool is None: + return + try: + remaining = list_storage_pools(self.apiClient, id=pool.id) + except Exception: + remaining = None + if not remaining: + self.__class__.pool2 = None + + + def _assert_only_original_pool(self, pool): + """Assert the rejected create did not add a second pool of this name.""" + try: + listed = list_storage_pools(self.apiClient, name=pool.name) or [] + except CloudstackAPIException: + listed = [] + same_name = [ + item for item in listed if getattr(item, "name", None) == pool.name + ] + self.assertEqual( + [item.id for item in same_name], + [pool.id], + "CloudStack should still list only pool '%s' after the rejected " + "create, found: %s" % (pool.name, same_name), + ) + current = list_storage_pools(self.apiClient, id=pool.id) + self.assertTrue( + current, + "Original pool '%s' disappeared after the rejected create" + % pool.name, + ) + self.assertEqual( + current[0].state, "Up", + "Original pool '%s' should still be Up, got '%s'" + % (pool.name, current[0].state), + ) + + def _warn_if_duplicate_pool(self, duplicate, pool): + """Leave a stray duplicate in place. + + Deleting it would remove the FlexVol that the rest of the suite + still uses. + """ + if duplicate is None or getattr(duplicate, "id", None) == pool.id: + return + logger.warning( + "Duplicate pool '%s' (id=%s) was created against the shared " + "FlexVol; leaving it in place so cleanup does not delete the " + "pool the rest of the suite uses", + pool.name, duplicate.id, + ) + + def _assert_no_pool_named(self, pool_name): + """Assert CloudStack holds no storage pool with this name.""" + try: + listed = list_storage_pools(self.apiClient, name=pool_name) + except CloudstackAPIException: + listed = None + self.assertFalse( + listed, + "CloudStack should hold no pool named '%s' after a rejected " + "create, found: %s" % (pool_name, listed), + ) + + + def _assert_pool_gone_from_cs(self, pool_id, pool_name): + try: + remaining = list_storage_pools(self.apiClient, id=pool_id) + except CloudstackAPIException: + remaining = None + self.assertFalse( + remaining, + "Pool '%s' still listed in CloudStack after deletion" % pool_name, + ) + + + def _sweep_tracked_pool2(self): + """Clean a prior leftover before reusing the shared pool2 slot.""" + if self.__class__.pool2 is None: + return + self._cleanup_throwaway_pool(self.__class__.pool2) + if self.__class__.pool2 is not None: + self.skipTest("A previously tracked pool could not be cleaned up") + + + def _create_throwaway_pool(self, suffix): + """Create an isolated pool, park it in pool2, and return it.""" + self._sweep_tracked_pool2() + pool = self._create_pool(pool_name=self._throwaway_pool_name(suffix)) + self.__class__.pool2 = pool + self.assertEqual( + pool.state, "Up", + "Throwaway pool '%s' should be 'Up', got '%s'" + % (pool.name, pool.state), + ) + self.assertIsNotNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol missing for throwaway pool '%s'" % pool.name, + ) + return pool + + + @attr(tags=["iscsi_workflow"], required_hardware=True) + def test_11_delete_pool_with_flexvol_predeleted(self): + """ + Delete the backing FlexVol directly on ONTAP, then delete the empty + pool through CloudStack. Deletion must tolerate the missing volume + rather than leaving an undeletable pool behind. + + Verifies: + - deleteStoragePool succeeds with the FlexVol already gone + - the pool is removed from CloudStack + """ + pool = self._create_throwaway_pool("PreDelVol") + try: + self._enter_maintenance(pool.id) + log_progress( + logger, "info", + "Deleting ONTAP FlexVol '%s' behind CloudStack's back", + pool.name, + ) + self.ontap.offline_and_delete_volume(pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol '%s' still present after direct deletion" + % pool.name, + ) + + self._delete_pool(pool.id) + self._assert_pool_gone_from_cs(pool.id, pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol '%s' reappeared after pool deletion" + % pool.name, + ) + self.__class__.pool2 = None + finally: + self._cleanup_throwaway_pool( + self.__class__.pool2, flexvol_name=pool.name + ) + + + @attr(tags=["iscsi_workflow"], required_hardware=True) + def test_12_delete_pool_with_igroups_predeleted(self): + """ + Delete the per-host igroups directly on ONTAP, then delete the empty + pool through CloudStack. Deletion must tolerate the missing igroups + and still remove the FlexVol. + + The igroup name is keyed off the host UUID and the SVM, not the pool, + so the igroups are shared by every ONTAP pool on that SVM. This test + therefore assumes no other pool is in use on the SVM, which holds + here because test_10 removed the workflow pools. + + The plugin only creates an igroup when a host is first granted access + to a LUN, so a freshly created empty pool has none. The test seeds + them on ONTAP under the exact names the plugin would use, which both + proves the naming scheme still matches and makes the pre-deletion a + real precondition rather than a no-op. + + Verifies: + - deleteStoragePool succeeds with the igroups already gone + - the pool is removed from CloudStack + - ONTAP: the FlexVol is deleted + """ + other_pools = self._other_ontap_pools_on_svm(None) + if other_pools: + self.skipTest( + "Pre-deleting SVM-wide igroups requires exclusive SVM use; " + "found other ONTAP pool(s): %s" + % ", ".join(str(getattr(p, "name", p)) for p in other_pools) + ) + specs = self._iscsi_host_specs() + if not specs: + self.skipTest( + "No cluster host advertises an iSCSI IQN, so no igroup name " + "can be derived" + ) + pool = self._create_throwaway_pool("PreDelIgroup") + seeded = [] + try: + for igroup_name, iqn in specs: + if self.ontap.get_igroup(self.svm_name, igroup_name) is None: + log_progress( + logger, "info", + "Seeding ONTAP igroup '%s' with initiator '%s'", + igroup_name, iqn, + ) + self.ontap.create_igroup(self.svm_name, igroup_name, iqn) + seeded.append(igroup_name) + present = [name for name, _ in specs] + for igroup_name in present: + self.assertIsNotNone( + self.ontap.get_igroup(self.svm_name, igroup_name), + "ONTAP igroup '%s' should exist before the pre-deletion" + % igroup_name, + ) + + self._enter_maintenance(pool.id) + log_progress( + logger, "info", + "Deleting ONTAP igroups %s behind CloudStack's back", present, + ) + for igroup_name in present: + self.ontap.delete_igroup(self.svm_name, igroup_name) + self.assertIsNone( + self.ontap.get_igroup(self.svm_name, igroup_name), + "ONTAP igroup '%s' still present after direct deletion" + % igroup_name, + ) + + self._delete_pool(pool.id) + self._assert_pool_gone_from_cs(pool.id, pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol '%s' still exists after pool deletion" + % pool.name, + ) + self.__class__.pool2 = None + finally: + for igroup_name in seeded: + try: + self.ontap.delete_igroup(self.svm_name, igroup_name) + except Exception as exc: + logger.warning( + "cleanup: could not delete seeded igroup '%s': %s", + igroup_name, exc, + ) + self._cleanup_throwaway_pool( + self.__class__.pool2, flexvol_name=pool.name + ) diff --git a/test/integration/plugins/ontap/iscsi/pool/test_pool_with_volumes.py b/test/integration/plugins/ontap/iscsi/pool/test_pool_with_volumes.py index 9dd49761c1dc..78182a7dca84 100644 --- a/test/integration/plugins/ontap/iscsi/pool/test_pool_with_volumes.py +++ b/test/integration/plugins/ontap/iscsi/pool/test_pool_with_volumes.py @@ -47,6 +47,17 @@ 06 Re-enter maintenance; forced=False delete rejected (Neg SN 5) 07 Delete volume from Maintenance, then force-delete pool (SN 7) +Isolated tests (test_08 onwards) run after that workflow and share no state +with it. Each one builds its own pool plus CloudStack volume, breaks a single +ONTAP object behind CloudStack's back, and force-deletes the pool: + + 08 FlexVol pre-deleted on ONTAP, then force-delete pool with CS volume + 09 Host igroups pre-deleted on ONTAP, then force-delete pool with CS volume + 10 Enter maintenance with CS volume after LUN maps are pre-deleted + 11 Log out only the isolated test iSCSI session, then enter maintenance + 12 Delete the test volume while that session is already logged in + 13 Replace the test LUN by-path with a regular file, then delete the volume + Prerequisites: - CloudStack management server with the NetApp ONTAP plugin deployed - KVM cluster where every host has iSCSI initiator configured @@ -62,16 +73,13 @@ import base64 import logging import random -import re import unittest from nose.plugins.attrib import attr from marvin.cloudstackAPI import ( - cancelStorageMaintenance, createStoragePool as createStoragePoolAPI, deleteVolume as deleteVolumeAPI, - enableStorageMaintenance, updateStoragePool as updateStoragePoolAPI, ) from marvin.cloudstackException import CloudstackAPIException @@ -82,7 +90,6 @@ logger = logging.getLogger("TestOntapISCSIPoolWithVolumes") - # --------------------------------------------------------------------------- # Test data # --------------------------------------------------------------------------- @@ -141,17 +148,6 @@ def __init__(self, storage_ip, svm_name, username, password, } -# --------------------------------------------------------------------------- -# Helpers -# --------------------------------------------------------------------------- - -def _igroup_name(svm_name, host_name): - """Mirror OntapStorageUtils.getIgroupName: cs_{svmName}_{sanitizedHostName}""" - short = host_name.split(".")[0] - sanitized = re.sub(r"[^a-zA-Z0-9_-]", "_", short) - return "cs_%s_%s" % (svm_name, sanitized) - - # --------------------------------------------------------------------------- # Test class # --------------------------------------------------------------------------- @@ -159,7 +155,7 @@ def _igroup_name(svm_name, host_name): class TestOntapISCSIPoolWithVolumes(OntapTestBase): """ iSCSI pool lifecycle tests with a CloudStack data volume present throughout. - All 7 tests are sequential and share class-level state. + Tests 01-07 are sequential; tests 08-10 use isolated throwaway resources. """ _vol_name_prefix = "OntapISCSIWV" @@ -201,6 +197,7 @@ def setUpClass(cls): cls.svm_name = svm_name cls._setup_cloudstack_resources(config, cls.testdata[TestData.account]) + cls._capture_igroup_baseline() # ------------------------------------------------------------------ # Helpers @@ -232,15 +229,6 @@ def _create_pool(self): response = self.apiClient.createStoragePool(cmd) return StoragePool(response.__dict__) - def _volume_exists_in_cs(self, vol_id): - """Return True if the volume is still listed by CloudStack.""" - from marvin.cloudstackAPI import listVolumes as listVolumesAPI - cmd = listVolumesAPI.listVolumesCmd() - cmd.id = vol_id - cmd.listall = True - vols = self.apiClient.listVolumes(cmd) or [] - return len(vols) > 0 - def _assert_lun_exists(self, pool_name, msg_context=""): """Assert that at least one LUN exists in the pool's ONTAP FlexVol.""" luns = self.ontap.list_luns_in_volume(self.svm_name, pool_name) @@ -339,18 +327,12 @@ def test_01_create_pool_and_volume(self): "ONTAP FlexVol should be 'online', got '%s'" % ontap_vol.get("state") ) - # ONTAP: igroup must exist for each cluster host that has an IQN - for host in self.cluster_hosts: - iqn = getattr(host, "storageurl", None) - if not iqn or not iqn.startswith("iqn."): - continue - igroup_name = _igroup_name(self.svm_name, host.name) - igroup = self.ontap.get_igroup(self.svm_name, igroup_name) - self.assertIsNotNone( - igroup, - "ONTAP igroup '%s' not found for host '%s'" - % (igroup_name, host.name) - ) + # ONTAP: the plugin creates an igroup only when a host is first + # granted access to a LUN, so the empty pool must not change the + # suite-start igroup baseline. + self._assert_igroup_baseline_unchanged( + "after creating an empty pool" + ) # Allocate a CloudStack data volume on this pool vol = self._create_volume(pool.id) @@ -359,6 +341,9 @@ def test_01_create_pool_and_volume(self): # ONTAP: a LUN must exist in the FlexVol after volume creation self._assert_lun_exists(pool.name, "after volume creation") + self._assert_igroup_baseline_unchanged( + "after creating an unattached volume" + ) # Capacity reporting: LUN allocated but FlexVol size unchanged self._assert_pool_capacity(pool, "volume-allocated") @@ -479,11 +464,7 @@ def test_04_enter_maintenance_volume_present(self): self.assertIsNotNone(self.__class__.volume, "Volume absent — test_01 must pass first") - cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.enableStorageMaintenance(cmd) - - result = self._poll_pool_state(self.__class__.pool.id, "Maintenance", timeout=120) + result = self._enter_maintenance(self.__class__.pool.id) self.assertEqual( result.state, "Maintenance", "Pool should be 'Maintenance', got '%s'" % result.state @@ -531,11 +512,7 @@ def test_05_cancel_maintenance_volume_present(self): self.assertIsNotNone(self.__class__.volume, "Volume absent — test_01 must pass first") - cmd = cancelStorageMaintenance.cancelStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.cancelStorageMaintenance(cmd) - - result = self._poll_pool_state(self.__class__.pool.id, "Up", timeout=120) + result = self._exit_maintenance(self.__class__.pool.id) self.assertEqual( result.state, "Up", "Pool should be 'Up' after cancel maintenance, got '%s'" % result.state @@ -580,10 +557,7 @@ def test_06_forced_false_delete_rejected(self): "Volume absent — test_01 must pass first") # Re-enter Maintenance (pool is Up from test_05) - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = self.__class__.pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(self.__class__.pool.id, "Maintenance", timeout=120) + self._enter_maintenance(self.__class__.pool.id) # Attempt forced=False delete — must raise with self.assertRaises(Exception, @@ -693,15 +667,688 @@ def test_07_delete_volume_and_force_delete_pool(self): "ONTAP FlexVol '%s' still exists after pool force deletion" % pool_name ) - # ONTAP: igroups for all cluster hosts must be deleted - for host in self.cluster_hosts: - iqn = getattr(host, "storageurl", None) - if not iqn or not iqn.startswith("iqn."): - continue - igroup_name = _igroup_name(self.svm_name, host.name) - igroup = self.ontap.get_igroup(self.svm_name, igroup_name) + self._assert_no_lun_maps_for_volume( + pool_name, "after pool force deletion" + ) + self._assert_igroup_baseline_unchanged("after pool force deletion") + + # ================================================================== + # Isolated tests — appended after the sequential workflow above. + # + # Each one creates its own pool and CloudStack volume in the pool2 / + # volume2 slots (which OntapTestBase.tearDownClass also sweeps), runs a + # single scenario, and cleans up in a finally block. They never reuse a + # pool destroyed by another test. + # ================================================================== + + def _host_igroup_names(self): + """igroup names the plugin creates, one per cluster host with an IQN. + + Built from the host UUID so the names match the plugin. + """ + return [name for name, _ in self._host_igroup_specs()] + + def _host_igroup_specs(self): + """(igroup name, initiator IQN) per cluster host that reports an IQN.""" + return self._iscsi_host_specs() + + def _create_isolated_pool_with_volume(self, label): + """Build a fresh pool plus CS volume for one isolated scenario. + + Returns ``(pool, volume)``. Any pool a previous isolated test could + not clean up is swept first so its ONTAP objects are never orphaned by + the overwrite of the pool2 slot. + """ + if self.__class__.pool2 is not None: + self._cleanup_isolated_pool( + self.__class__.pool2, "leftover-from-previous-isolated-test" + ) + + pool = self._create_pool() + self.__class__.pool2 = pool + logger.info("[%s] created isolated pool '%s'", label, pool.name) + + self.assertEqual( + pool.state, "Up", + "[%s] new pool state should be 'Up', got '%s'" % (label, pool.state) + ) + ontap_vol = self.ontap.get_volume(pool.name) + self.assertIsNotNone( + ontap_vol, + "[%s] ONTAP FlexVol not found for new pool '%s'" % (label, pool.name) + ) + + vol = self._create_volume(pool.id) + self.__class__.volume2 = vol + self.assertIsNotNone(vol, "[%s] createVolume returned None" % label) + self._assert_lun_exists(pool.name, "%s: after volume creation" % label) + return pool, vol + + def _assert_pool_absent(self, pool, label, delete_error=None): + """Assert CloudStack no longer lists the pool.""" + try: + remaining = list_storage_pools(self.apiClient, name=pool.name) + except Exception: + remaining = None + self.assertFalse( + remaining, + "[%s] pool '%s' is still listed after deleteStoragePool(forced=True)%s" + % (label, pool.name, + "; the API raised: %s" % delete_error if delete_error else "") + ) + + def _cleanup_isolated_pool(self, pool, label): + """Best-effort teardown of one isolated pool and its ONTAP FlexVol. + + Igroups are intentionally left alone: their names carry no pool + identity, so the pool delete owns their removal. + """ + if pool is None: + return + try: + listed = list_storage_pools(self.apiClient, name=pool.name) + except Exception: + listed = None + if listed: + self._remove_cs_volume(pool, self.__class__.volume2, label) + try: + listed = list_storage_pools( + self.apiClient, name=pool.name + ) or listed + if listed[0].state != "Maintenance": + self._enter_maintenance(pool.id) + self._delete_pool(pool.id, forced=True) + except Exception as exc: + logger.warning("[%s] could not force-delete pool '%s': %s", + label, pool.name, exc) + try: + self.ontap.offline_and_delete_volume(pool.name) + except Exception as exc: + logger.warning("[%s] ONTAP FlexVol cleanup for '%s' failed: %s", + label, pool.name, exc) + try: + listed = list_storage_pools(self.apiClient, name=pool.name) + except Exception: + listed = None + if not listed: + self.__class__.pool2 = None + + # ------------------------------------------------------------------ + # Step 08 — FlexVol deleted on ONTAP before the pool delete (negative) + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_with_volumes"], required_hardware=True) + def test_08_delete_pool_with_volume_flexvol_missing(self): + """ + Force-delete a pool that still owns a CloudStack volume after its + ONTAP FlexVol — and with it the volume's LUN — has been removed behind + CloudStack's back. + + Uses its own pool and volume, so the pool is known to be healthy up to + the point the FlexVol is destroyed. Verifies: + - deleteStoragePool is rejected while the CS volume still exists + - deleteStoragePool(forced=True) tolerates the missing FlexVol + - the CloudStack pool record is removed + - the leftover CS volume record can still be cleaned up + """ + label = "flexvol-missing" + pool, vol = self._create_isolated_pool_with_volume(label) + try: + self._enter_maintenance(pool.id) + + self.ontap.offline_and_delete_volume(pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "[%s] ONTAP FlexVol '%s' should be gone before the pool delete" + % (label, pool.name) + ) + self.assertEqual( + len(self.ontap.list_luns_in_volume(self.svm_name, pool.name)), 0, + "[%s] LUNs should have gone with the FlexVol '%s'" + % (label, pool.name) + ) + + # CloudStack rejects deleteStoragePool while the pool still owns + # a volume, even with forced=True, so the volume goes first. + with self.assertRaises(CloudstackAPIException): + self._delete_pool(pool.id, forced=True) + self.assertTrue( + self._remove_cs_volume(pool, vol, label), + "[%s] CloudStack volume could not be deleted before the pool " + "delete" % label + ) + try: + self._enter_maintenance(pool.id) + except Exception as exc: + logger.warning("[%s] could not enter maintenance on '%s': %s", + label, pool.name, exc) + + delete_error = None + try: + self._delete_pool(pool.id, forced=True) + except CloudstackAPIException as exc: + delete_error = exc + + self._assert_pool_absent(pool, label, delete_error) self.assertIsNone( - igroup, - "ONTAP igroup '%s' still exists after pool force deletion" - % igroup_name + delete_error, + "[%s] deleteStoragePool(forced=True) should tolerate a missing " + "FlexVol, but raised: %s" % (label, delete_error) + ) + + self._assert_igroup_baseline_unchanged( + "[%s] after pool delete with missing FlexVol" % label + ) + + self._purge_cs_volume_record(vol, label) + finally: + self._cleanup_isolated_pool(pool, label) + + # ------------------------------------------------------------------ + # Step 09 — Host igroups deleted on ONTAP before the delete (negative) + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_with_volumes"], required_hardware=True) + def test_09_delete_pool_with_volume_igroups_missing(self): + """ + Force-delete a pool that still owns a CloudStack volume after the host + igroups have been removed behind CloudStack's back. + + Uses its own pool and volume. The volume is not attached to any VM, so + no LUN maps reference the igroups and they delete cleanly. Unlike + test_08 the FlexVol is still present, so the plugin is expected to + remove it as part of the delete. Verifies: + - deleteStoragePool is rejected while the CS volume still exists + - deleteStoragePool(forced=True) tolerates the missing igroups + - the CloudStack pool record is removed + - the ONTAP FlexVol is deleted and no igroup is left behind + """ + other_pools = self._other_ontap_pools_on_svm(None) + if other_pools: + self.skipTest( + "Pre-deleting SVM-wide igroups requires exclusive SVM use; " + "found other ONTAP pool(s): %s" + % ", ".join(str(getattr(p, "name", p)) for p in other_pools) + ) + label = "igroups-missing" + pool, vol = self._create_isolated_pool_with_volume(label) + seeded = [] + try: + igroup_specs = self._host_igroup_specs() + self.assertTrue( + igroup_specs, + "[%s] no cluster host reports an IQN, so there is no igroup " + "to remove" % label + ) + igroup_names = [name for name, _ in igroup_specs] + + # The volume is not attached to a VM, so the plugin has never + # granted a host access and created no igroups. Seed them under + # the plugin's own names so the pre-deletion is a real one. + for name, iqn in igroup_specs: + if self.ontap.get_igroup(self.svm_name, name) is None: + logger.info("[%s] seeding igroup '%s' with initiator '%s'", + label, name, iqn) + self.ontap.create_igroup(self.svm_name, name, iqn) + seeded.append(name) + + self._enter_maintenance(pool.id) + + deleted = [] + for name in igroup_names: + if self.ontap.get_igroup(self.svm_name, name) is None: + continue + self.ontap.delete_igroup(self.svm_name, name) + deleted.append(name) + logger.info("[%s] deleted %d of %d host igroup(s): %s", + label, len(deleted), len(igroup_names), deleted) + + for name in igroup_names: + self.assertIsNone( + self.ontap.get_igroup(self.svm_name, name), + "[%s] igroup '%s' should be gone before the pool delete" + % (label, name) + ) + + # CloudStack rejects deleteStoragePool while the pool still owns + # a volume, even with forced=True, so the volume goes first. + with self.assertRaises(CloudstackAPIException): + self._delete_pool(pool.id, forced=True) + self.assertTrue( + self._remove_cs_volume(pool, vol, label), + "[%s] CloudStack volume could not be deleted before the pool " + "delete" % label + ) + try: + self._enter_maintenance(pool.id) + except Exception as exc: + logger.warning("[%s] could not enter maintenance on '%s': %s", + label, pool.name, exc) + + delete_error = None + try: + self._delete_pool(pool.id, forced=True) + except CloudstackAPIException as exc: + delete_error = exc + + self._assert_pool_absent(pool, label, delete_error) + self.assertIsNone( + delete_error, + "[%s] deleteStoragePool(forced=True) should tolerate missing " + "igroups, but raised: %s" % (label, delete_error) + ) + + self.assertIsNone( + self.ontap.get_volume(pool.name), + "[%s] ONTAP FlexVol '%s' should have been deleted with the pool" + % (label, pool.name) + ) + for name in igroup_names: + self.assertIsNone( + self.ontap.get_igroup(self.svm_name, name), + "[%s] igroup '%s' reappeared during the pool delete" + % (label, name) + ) + + self._purge_cs_volume_record(vol, label) + finally: + for name in seeded: + try: + self.ontap.delete_igroup(self.svm_name, name) + except Exception as exc: + logger.warning( + "[%s] cleanup: could not delete seeded igroup '%s': %s", + label, name, exc) + self._cleanup_isolated_pool(pool, label) + + # ------------------------------------------------------------------ + # Step 10 — Enter maintenance after LUN maps are pre-deleted + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_with_volumes"], required_hardware=True) + def test_10_enter_maintenance_lun_maps_predeleted(self): + """ + Enter maintenance with a CloudStack volume after its ONTAP LUN maps + have been deleted behind CloudStack's back. Verifies: + - enableStorageMaintenance tolerates already-absent LUN maps + - the pool reaches Maintenance and the CS volume remains present + - the LUN remains online and its maps remain absent + """ + label = "maintenance-lun-maps-missing" + pool, vol = self._create_isolated_pool_with_volume(label) + seeded_igroup = None + try: + luns = self.ontap.list_luns_in_volume(self.svm_name, pool.name) + self.assertTrue( + luns, + "[%s] no LUN found in FlexVol '%s'" % (label, pool.name) + ) + lun_path = luns[0].get("name") + + igroup_specs = self._host_igroup_specs() + self.assertTrue( + igroup_specs, + "[%s] no cluster host reports an IQN" % label + ) + igroup_name, initiator_iqn = igroup_specs[0] + if self.ontap.get_igroup(self.svm_name, igroup_name) is None: + self.ontap.create_igroup( + self.svm_name, igroup_name, initiator_iqn + ) + seeded_igroup = igroup_name + + self.ontap.create_lun_map( + self.svm_name, lun_path, igroup_name + ) + maps = self.ontap.list_lun_maps_for_volume( + self.svm_name, pool.name + ) + self.assertTrue( + maps, + "[%s] failed to seed a LUN map for '%s'" % (label, lun_path) + ) + + for lun_map in maps: + self.ontap.delete_lun_map(lun_map) + self.assertEqual( + self.ontap.list_lun_maps_for_volume( + self.svm_name, pool.name + ), + [], + "[%s] LUN maps should be absent before maintenance" % label + ) + + self._enter_maintenance(pool.id) + self.assertTrue( + self._volume_exists_in_cs(vol.id), + "[%s] CS volume disappeared after entering maintenance" % label + ) + self._assert_lun_exists( + pool.name, "after entering Maintenance with maps pre-deleted" + ) + self.assertEqual( + self.ontap.list_lun_maps_for_volume( + self.svm_name, pool.name + ), + [], + "[%s] LUN maps unexpectedly reappeared during maintenance" + % label + ) + finally: + self._cleanup_isolated_pool(pool, label) + if seeded_igroup: + try: + self.ontap.delete_igroup( + self.svm_name, seeded_igroup + ) + except Exception as exc: + logger.warning( + "[%s] cleanup: could not delete seeded igroup '%s': %s", + label, seeded_igroup, exc + ) + + def _test_iscsi_endpoint(self, host_ip=None): + """Return a reachable data LIF and target IQN for the suite SVM.""" + lifs = self.ontap.get_iscsi_data_lifs(self.svm_name) + target = self.ontap.get_iscsi_target_iqn(self.svm_name) + if not lifs or not target: + self.skipTest( + "SVM '%s' has no iSCSI data LIF or target IQN" % self.svm_name + ) + if host_ip: + for lif in lifs: + quoted_lif = lif.replace("'", "'\"'\"'") + reachable = self._kvm_run( + host_ip, + "timeout 3 bash -c " + "'exec 3<>/dev/tcp/%s/3260' >/dev/null 2>&1 " + "&& echo OPEN || true" % quoted_lif + ) + if any(line.strip() == "OPEN" for line in reachable): + return lif, target + self.skipTest( + "No iSCSI data LIF for SVM '%s' is reachable from %s" + % (self.svm_name, host_ip) + ) + return lifs[0], target + + def _iscsi_session_lines(self, host_ip, target): + quoted = target.replace("'", "'\"'\"'") + return self._kvm_run( + host_ip, + "iscsiadm -m session 2>/dev/null | grep -F '%s' || true" % quoted + ) + + def _login_test_iscsi_session(self, host_ip, portal, target): + """Log in only this target. Skip when the target is already in use.""" + if self._iscsi_session_lines(host_ip, target): + self.skipTest( + "iSCSI target '%s' is already logged in on %s; logging it " + "out would drop other LUNs" % (target, host_ip) + ) + portal_port = "%s:3260" % portal + quoted_target = target.replace("'", "'\"'\"'") + quoted_portal = portal_port.replace("'", "'\"'\"'") + self._kvm_run( + host_ip, + "iscsiadm -m node -T '%s' -p '%s' -o new >/dev/null 2>&1 || true; " + "timeout 30 iscsiadm -m node -T '%s' -p '%s' " + "--login >/dev/null 2>&1 || true" + % (quoted_target, quoted_portal, quoted_target, quoted_portal) + ) + if not self._iscsi_session_lines(host_ip, target): + self.skipTest( + "Could not log in test iSCSI target '%s' on %s" + % (target, host_ip) + ) + + def _logout_test_iscsi_session(self, host_ip, portal, target): + portal_port = "%s:3260" % portal + quoted_target = target.replace("'", "'\"'\"'") + quoted_portal = portal_port.replace("'", "'\"'\"'") + self._kvm_run( + host_ip, + "iscsiadm -m node -T '%s' -p '%s' --logout >/dev/null 2>&1 || true" + % (quoted_target, quoted_portal) + ) + + def _map_isolated_lun(self, pool, label): + """Map the isolated pool's LUN and return path, igroup and LUN id.""" + luns = self.ontap.list_luns_in_volume(self.svm_name, pool.name) + self.assertTrue(luns, "[%s] no LUN found in '%s'" % (label, pool.name)) + lun_path = luns[0].get("name") + specs = self._host_igroup_specs() + self.assertTrue(specs, "[%s] no host IQN is available" % label) + igroup_name, initiator = specs[0] + seeded = None + if self.ontap.get_igroup(self.svm_name, igroup_name) is None: + self.ontap.create_igroup(self.svm_name, igroup_name, initiator) + seeded = igroup_name + self.ontap.create_lun_map(self.svm_name, lun_path, igroup_name) + maps = self.ontap.list_lun_maps_for_volume(self.svm_name, pool.name) + self.assertTrue(maps, "[%s] LUN map was not created" % label) + lun_id = maps[0].get("logical_unit_number") + return lun_path, igroup_name, lun_id, seeded + + # ------------------------------------------------------------------ + # Step 11 — Test iSCSI session logged out on one host + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_with_volumes"], required_hardware=True) + def test_11_iscsi_session_logged_out(self): + """ + Log out only the isolated test target on one KVM host, then enter + maintenance. The session is restored before the pool is deleted. + """ + label = "iscsi-session-logout" + host_ip = self._kvm_host_ip() + portal, target = self._test_iscsi_endpoint(host_ip) + pool, vol = self._create_isolated_pool_with_volume(label) + seeded = None + logged_in = False + try: + _, _, _, seeded = self._map_isolated_lun(pool, label) + self._login_test_iscsi_session(host_ip, portal, target) + logged_in = True + self._logout_test_iscsi_session(host_ip, portal, target) + logged_in = False + self.assertFalse( + self._iscsi_session_lines(host_ip, target), + "[%s] test iSCSI session still present after logout" % label + ) + + self._enter_maintenance(pool.id) + self.assertTrue( + self._volume_exists_in_cs(vol.id), + "[%s] volume disappeared after maintenance" % label + ) + self._assert_lun_exists(pool.name, label) + finally: + if logged_in: + try: + self._logout_test_iscsi_session(host_ip, portal, target) + except Exception as exc: + logger.warning( + "[%s] could not log out test session: %s", label, exc + ) + self._cleanup_isolated_pool(pool, label) + if seeded: + try: + self.ontap.delete_igroup(self.svm_name, seeded) + except Exception as exc: + logger.warning( + "[%s] could not delete seeded igroup: %s", label, exc + ) + + # ------------------------------------------------------------------ + # Step 12 — Volume delete while the test session already exists + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_with_volumes"], required_hardware=True) + def test_12_delete_volume_with_existing_iscsi_session(self): + """ + Use the existing target session while CloudStack deletes the volume. + Delete must still remove the LUN and must not create another session. + """ + label = "iscsi-session-exists" + host_ip = self._kvm_host_ip() + portal, target = self._test_iscsi_endpoint(host_ip) + existing_sessions = self._iscsi_session_lines(host_ip, target) + created_session = False + if not existing_sessions: + self._login_test_iscsi_session(host_ip, portal, target) + created_session = True + existing_sessions = self._iscsi_session_lines(host_ip, target) + pool, vol = self._create_isolated_pool_with_volume(label) + seeded = None + try: + _, _, _, seeded = self._map_isolated_lun(pool, label) + before = len(existing_sessions) + + cmd = deleteVolumeAPI.deleteVolumeCmd() + cmd.id = vol.id + self.apiClient.deleteVolume(cmd) + self.__class__.volume2 = None + self.assertFalse( + self._volume_exists_in_cs(vol.id), + "[%s] volume remained after deleteVolume" % label + ) + self.assertEqual( + self.ontap.list_luns_in_volume(self.svm_name, pool.name), + [], + "[%s] LUN remained after deleteVolume" % label + ) + after = len(self._iscsi_session_lines(host_ip, target)) + self.assertLessEqual( + after, before, + "[%s] deleteVolume created an extra iSCSI session" % label + ) + finally: + self._cleanup_isolated_pool(pool, label) + if seeded: + try: + self.ontap.delete_igroup(self.svm_name, seeded) + except Exception as exc: + logger.warning( + "[%s] could not delete seeded igroup: %s", label, exc + ) + if created_session: + try: + self._logout_test_iscsi_session(host_ip, portal, target) + except Exception as exc: + logger.warning( + "[%s] could not log out test session: %s", label, exc + ) + + # ------------------------------------------------------------------ + # Step 13 — Regular file planted at the test LUN by-path + # ------------------------------------------------------------------ + + @attr(tags=["iscsi_with_volumes"], required_hardware=True) + def test_13_corrupt_iscsi_by_path(self): + """ + Replace only the isolated LUN's by-path symlink with a regular file. + Deleting the unattached volume must still remove the ONTAP LUN and + must leave that regular file in place. + """ + label = "iscsi-by-path-file" + host_ip = self._kvm_host_ip() + portal, target = self._test_iscsi_endpoint(host_ip) + existing_sessions = self._iscsi_session_lines(host_ip, target) + created_session = False + if not existing_sessions: + self._login_test_iscsi_session(host_ip, portal, target) + created_session = True + existing_sessions = self._iscsi_session_lines(host_ip, target) + portal_field = existing_sessions[0].split()[2] + portal = portal_field.rsplit(":3260,", 1)[0] + if portal == portal_field: + self.skipTest( + "[%s] could not parse portal from session: %s" + % (label, existing_sessions[0]) + ) + pool, vol = self._create_isolated_pool_with_volume(label) + seeded = None + by_path = None + try: + _, _, lun_id, seeded = self._map_isolated_lun(pool, label) + if lun_id is None: + self.skipTest("[%s] ONTAP did not report a LUN number" % label) + by_path = ( + "/dev/disk/by-path/ip-%s:3260-iscsi-%s-lun-%s" + % (portal, target, lun_id) + ) + quoted = by_path.replace("'", "'\"'\"'") + quoted_target = target.replace("'", "'\"'\"'") + quoted_portal = ("%s:3260" % portal).replace("'", "'\"'\"'") + self._kvm_run( + host_ip, + "iscsiadm -m node -T '%s' -p '%s' --rescan" + % (quoted_target, quoted_portal) + ) + self.assertTrue( + self._kvm_run( + host_ip, + "if [ -L '%s' ]; then echo SYMLINK; fi" % quoted + ), + "[%s] expected iSCSI by-path symlink was not found at %s" + % (label, by_path) + ) + planted = self._kvm_run( + host_ip, + "if [ -L '%s' ]; then mv '%s' '%s.bak'; fi; " + "rm -f '%s'; : > '%s'; " + "if [ -f '%s' ] && [ ! -L '%s' ]; then echo FILE; fi" + % (quoted, quoted, quoted, quoted, quoted, quoted, quoted) + ) + self.assertTrue( + any(line.strip() == "FILE" for line in planted), + "[%s] could not plant a regular file at %s" % (label, by_path) + ) + + cmd = deleteVolumeAPI.deleteVolumeCmd() + cmd.id = vol.id + self.apiClient.deleteVolume(cmd) + self.__class__.volume2 = None + self.assertEqual( + self.ontap.list_luns_in_volume(self.svm_name, pool.name), + [], + "[%s] LUN remained after deleteVolume" % label + ) + still_file = self._kvm_run( + host_ip, + "if [ -f '%s' ] && [ ! -L '%s' ]; then echo FILE; fi" + % (quoted, quoted) + ) + self.assertTrue( + any(line.strip() == "FILE" for line in still_file), + "[%s] CloudStack replaced the planted file at %s" + % (label, by_path) ) + finally: + if by_path: + quoted = by_path.replace("'", "'\"'\"'") + try: + self._kvm_run( + host_ip, + "rm -f '%s' '%s.bak'" % (quoted, quoted) + ) + except Exception as exc: + logger.warning( + "[%s] could not clean up %s: %s", label, by_path, exc + ) + self._cleanup_isolated_pool(pool, label) + if seeded: + try: + self.ontap.delete_igroup(self.svm_name, seeded) + except Exception as exc: + logger.warning( + "[%s] could not delete seeded igroup: %s", label, exc + ) + if created_session: + try: + self._logout_test_iscsi_session(host_ip, portal, target) + except Exception as exc: + logger.warning( + "[%s] could not log out test session: %s", label, exc + ) diff --git a/test/integration/plugins/ontap/iscsi/pool/test_zone_scoped_pool.py b/test/integration/plugins/ontap/iscsi/pool/test_zone_scoped_pool.py index 847a026a3bd5..1f2801cb65e4 100644 --- a/test/integration/plugins/ontap/iscsi/pool/test_zone_scoped_pool.py +++ b/test/integration/plugins/ontap/iscsi/pool/test_zone_scoped_pool.py @@ -18,16 +18,16 @@ """ Zone-scoped primary storage lifecycle tests for NetApp ONTAP (iSCSI). -Creates a zone-scoped pool (scope=ZONE, no clusterid/podid). CloudStack calls -OntapPrimaryDatastoreLifecycle.attachZone(), which connects all eligible KVM -hosts in the zone and creates igroups for each host's IQN. +Creates a zone-scoped pool (scope=ZONE, no clusterid/podid). Host igroups are +shared by host and SVM and are created only when a LUN is granted to a host, +not when an empty pool is created. -Workflow: +Test order — sequential workflow that must run in order: 01 Create zone-scoped iSCSI pool — pool.state Up; ONTAP FlexVol online; - igroup present for each cluster host IQN + pre-existing shared igroups unchanged 02 Disable zone-scoped pool — pool.state Disabled; FlexVol unchanged 03 Enable zone-scoped pool — pool.state Up; FlexVol unchanged - 04 Delete zone-scoped pool — pool gone; FlexVol deleted; igroups deleted + 04 Delete zone-scoped pool — pool gone; FlexVol deleted; baseline restored Prerequisites: - CloudStack management server with the NetApp ONTAP plugin deployed @@ -40,27 +40,28 @@ --marvin-config=test/integration/plugins/ontap/ontap.cfg \\ test/integration/plugins/ontap/iscsi/pool/test_zone_scoped_pool.py -v -Note: Tests 01-04 share class-level state (sequential). Always run the full -suite. +Note: Tests share class-level state (sequential). Always run the full suite. """ import base64 import logging import random -import re import unittest from nose.plugins.attrib import attr from marvin.cloudstackAPI import ( createStoragePool as createStoragePoolAPI, - enableStorageMaintenance, updateStoragePool as updateStoragePoolAPI, ) from marvin.lib.base import StoragePool from marvin.lib.common import list_storage_pools -from ontap_test_base import OntapRestClient, OntapTestBase, get_datacenter_config +from ontap_test_base import ( + OntapRestClient, + OntapTestBase, + get_datacenter_config, +) logger = logging.getLogger("TestOntapISCSIZoneScopedPool") @@ -122,17 +123,6 @@ def __init__(self, storage_ip, svm_name, username, password, } -# --------------------------------------------------------------------------- -# iSCSI path helpers -# --------------------------------------------------------------------------- - -def _igroup_name(svm_name, host_name): - """Mirror OntapStorageUtils.getIgroupName: cs_{svmName}_{sanitizedHostName}""" - short = host_name.split(".")[0] - sanitized = re.sub(r"[^a-zA-Z0-9_-]", "_", short) - return "cs_%s_%s" % (svm_name, sanitized) - - # --------------------------------------------------------------------------- # Sequential workflow test class # --------------------------------------------------------------------------- @@ -176,6 +166,7 @@ def setUpClass(cls): cls.svm_name = svm_name cls._setup_cloudstack_resources(config, cls.testdata[TestData.account]) + cls._capture_igroup_baseline() # No per-test tearDown — state intentionally persists between steps. @@ -209,35 +200,6 @@ def _create_zone_pool(self): response = self.apiClient.createStoragePool(cmd) return StoragePool(response.__dict__) - def _assert_igroups_for_hosts(self, expect_present): - """Assert igroups are present (or absent) for each cluster host IQN.""" - for host in self.cluster_hosts: - iqn = (getattr(host, "storageurl", None) - or getattr(host, "StorageUrl", None)) - if not iqn or not iqn.startswith("iqn."): - continue - igroup_name = _igroup_name(self.svm_name, host.name) - igroup = self.ontap.get_igroup(self.svm_name, igroup_name) - if expect_present: - self.assertIsNotNone( - igroup, - "ONTAP igroup '%s' not found for host '%s' after pool creation" - % (igroup_name, host.name) - ) - initiator_names = [ - i.get("name", "") for i in igroup.get("initiators", []) - ] - self.assertIn( - iqn, initiator_names, - "Host IQN '%s' not in igroup '%s' initiators: %s" - % (iqn, igroup_name, initiator_names) - ) - else: - self.assertIsNone( - igroup, - "ONTAP igroup '%s' still exists after pool deletion" % igroup_name - ) - # ------------------------------------------------------------------ # Step 01 — Create zone-scoped iSCSI pool # ------------------------------------------------------------------ @@ -246,12 +208,10 @@ def _assert_igroups_for_hosts(self, expect_present): def test_01_create_zone_scoped_pool(self): """ Create a zone-scoped iSCSI primary storage pool (no clusterid/podid). - CloudStack calls attachZone(), which connects all eligible KVM hosts - in the zone and creates igroups for each host's IQN. Verifies: - pool.state is Up, type is OntapiSCSI - ONTAP: FlexVol is online - - ONTAP: igroup exists for each cluster host with the correct IQN + - ONTAP: pre-existing shared igroups are unchanged """ pool = self._create_zone_pool() self.__class__.pool = pool @@ -276,8 +236,12 @@ def test_01_create_zone_scoped_pool(self): "ONTAP FlexVol should be 'online', got '%s'" % ontap_vol.get("state") ) - # ONTAP: igroups must exist for each cluster host with IQN - self._assert_igroups_for_hosts(expect_present=True) + # ONTAP: the plugin creates an igroup only when a host is first + # granted access to a LUN, so creating an empty pool must not make + # new ones. + self._assert_igroup_baseline_unchanged( + "after creating an empty zone-scoped pool" + ) # ------------------------------------------------------------------ # Step 02 — Disable zone-scoped pool @@ -308,8 +272,10 @@ def test_02_disable_zone_scoped_pool(self): "ONTAP FlexVol should still be 'online' after disable" ) - # igroups must still be present after a simple disable - self._assert_igroups_for_hosts(expect_present=True) + # The pool has no volumes, so shared igroups must remain unchanged. + self._assert_igroup_baseline_unchanged( + "after disabling an empty zone-scoped pool" + ) # ------------------------------------------------------------------ # Step 03 — Enable zone-scoped pool @@ -340,8 +306,10 @@ def test_03_enable_zone_scoped_pool(self): "ONTAP FlexVol should be 'online' after enable" ) - # igroups must still be present after re-enable - self._assert_igroups_for_hosts(expect_present=True) + # The pool has no volumes, so shared igroups must remain unchanged. + self._assert_igroup_baseline_unchanged( + "after re-enabling an empty zone-scoped pool" + ) # ------------------------------------------------------------------ # Step 04 — Delete zone-scoped pool @@ -354,17 +322,14 @@ def test_04_delete_zone_scoped_pool(self): Verifies: - Pool is removed from CloudStack - ONTAP: FlexVol deleted - - ONTAP: igroups deleted for all cluster hosts + - ONTAP: this pool's maps are gone and shared igroups are restored """ self.assertIsNotNone(self.__class__.pool, "Pool absent - test_01 must pass first") pool = self.__class__.pool pool_name = pool.name - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(pool.id, "Maintenance", timeout=120) + self._enter_maintenance(pool.id) self._delete_pool(pool.id, forced=True) self.__class__.pool = None @@ -383,5 +348,9 @@ def test_04_delete_zone_scoped_pool(self): "ONTAP FlexVol '%s' still exists after pool deletion" % pool_name ) - # ONTAP: igroups for each cluster host must be deleted - self._assert_igroups_for_hosts(expect_present=False) + self._assert_no_lun_maps_for_volume( + pool_name, "after zone-scoped pool deletion" + ) + self._assert_igroup_baseline_unchanged( + "after zone-scoped pool deletion" + ) diff --git a/test/integration/plugins/ontap/nfs3/pool/test_pool_lifecycle.py b/test/integration/plugins/ontap/nfs3/pool/test_pool_lifecycle.py index 5d1812cdad4f..4cb37cc5152a 100644 --- a/test/integration/plugins/ontap/nfs3/pool/test_pool_lifecycle.py +++ b/test/integration/plugins/ontap/nfs3/pool/test_pool_lifecycle.py @@ -18,18 +18,22 @@ """ Sequential workflow integration tests for NetApp ONTAP NFS3 primary storage pool. -Tests are numbered test_01 ... test_08 and must run in that order. Each step +Tests are numbered test_01 ... test_12 and must run in that order. Each step builds on the shared state established by the previous step. Workflow: - 01 Create primary storage pool - 02 Disable storage pool - 03 Enable storage pool - 04 Enter maintenance mode - 05 Cancel maintenance mode - 06 Delete the storage pool - 07 Create fresh pool and allocate a CloudStack volume - 08 Delete volume then force-delete the pool + 01 Reject create when no online assigned aggregate has enough free space + 02 Create primary storage pool + 03 Reject create when a FlexVol of that name already exists on ONTAP + 04 Disable storage pool + 05 Enable storage pool + 06 Enter maintenance mode + 07 Cancel maintenance mode + 08 Delete the storage pool + 09 Create fresh pool and allocate a CloudStack volume + 10 Delete volume then force-delete the pool + 11 Delete an empty pool whose FlexVol was deleted directly on ONTAP + 12 Delete an empty pool whose NFS export policy was deleted on ONTAP Prerequisites: - CloudStack management server with the NetApp ONTAP plugin deployed @@ -42,9 +46,10 @@ --marvin-config=test/integration/plugins/ontap/ontap.cfg \\ test/integration/plugins/ontap/nfs3/pool/test_pool_lifecycle.py -v -Note: Tests 01-06 share class-level state (sequential). Running a single test -with -m "test_NN" will invoke setUpClass but the guard assertion will fail -immediately if earlier steps have not yet run. Always run the full suite. +Note: Tests 02-08 share class-level state (sequential). test_03 reuses the +FlexVol created by test_02. Running a single test with -m "test_NN" will +invoke setUpClass but the guard assertion will fail immediately if earlier +steps have not yet run. Always run the full suite. """ import base64 @@ -55,10 +60,8 @@ from nose.plugins.attrib import attr from marvin.cloudstackAPI import ( - cancelStorageMaintenance, createStoragePool as createStoragePoolAPI, deleteVolume as deleteVolumeAPI, - enableStorageMaintenance, updateStoragePool as updateStoragePoolAPI, ) from marvin.cloudstackException import CloudstackAPIException @@ -96,6 +99,7 @@ class TestData: DETAIL_NFS_MOUNT_OPTS = "nfsmountopts" ONTAP_MIN_VOLUME_SIZE = 1677721600 + ONTAP_MAX_VOLUME_SIZE = 300 * 1024 ** 4 def __init__(self, storage_ip, svm_name, username, password, protocol="NFS3", scope="CLUSTER", provider="NetApp ONTAP", @@ -199,10 +203,14 @@ def setUpClass(cls): # Helpers # ------------------------------------------------------------------ - def _create_pool(self): + def _create_pool(self, pool_name=None, capacitybytes=None): + """Create a pool; name and capacity default to the suite's values.""" ps = self.testdata[TestData.primaryStorage] storage_ip = self.testdata[TestData.ontap][TestData.DETAIL_STORAGE_IP] - pool_name = "OntapNFS3_%d" % random.randint(0, 99999) + if pool_name is None: + pool_name = "OntapNFS3_%d" % random.randint(0, 99999) + if capacitybytes is None: + capacitybytes = ps["capacitybytes"] cmd = createStoragePoolAPI.createStoragePoolCmd() cmd.name = pool_name @@ -213,7 +221,7 @@ def _create_pool(self): cmd.scope = ps[TestData.scope] cmd.provider = ps[TestData.provider] cmd.tags = ps[TestData.tags] - cmd.capacitybytes = ps["capacitybytes"] + cmd.capacitybytes = capacitybytes cmd.hypervisor = "KVM" cmd.managed = True @@ -305,15 +313,6 @@ def _assert_pool_capacity(self, pool, label): % (label, ontap_size, configured) ) - def _volume_exists_in_cs(self, vol_id): - """Return True if the volume is still listed by CloudStack.""" - from marvin.cloudstackAPI import listVolumes as listVolumesAPI - cmd = listVolumesAPI.listVolumesCmd() - cmd.id = vol_id - cmd.listall = True - vols = self.apiClient.listVolumes(cmd) or [] - return len(vols) > 0 - def _assert_pool_gone_from_cs(self, pool_id, pool_name): try: remaining = list_storage_pools(self.apiClient, id=pool_id) @@ -380,10 +379,7 @@ def _delete_volume_then_force_delete_pool(self, pool, vol, ep_name): self.assertTrue(listed, "Pool '%s' not found before delete" % pool.name) if listed[0].state != "Maintenance": self._assert_pool_capacity(pool, "volume-deleted") - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(pool.id, "Maintenance", timeout=120) + self._enter_maintenance(pool.id) self._cleanup_kvm_storage_pool_mounts(pool.id) try: @@ -397,11 +393,82 @@ def _delete_volume_then_force_delete_pool(self, pool, vol, ep_name): self._assert_ontap_pool_gone(pool.name, ep_name) # ------------------------------------------------------------------ - # Step 01 — Create primary storage pool + # Step 01 — Reject create when no aggregate has enough free space + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_workflow"], required_hardware=True) + def test_01_reject_create_when_no_aggregate_space(self): + """ + Ask for 1 GiB more than the largest online aggregate assigned to the + SVM can provide, so no aggregate qualifies and the plugin refuses + before creating anything. + + Verifies: + - createStoragePool raises CloudstackAPIException + - the error names the aggregate shortage rather than some other + failure ('No suitable aggregates') + - no pool is left in CloudStack and no FlexVol on ONTAP + """ + self._sweep_tracked_pool2() + max_free = self.ontap.max_online_aggregate_available_bytes( + self.svm_name + ) + if not max_free: + self.skipTest( + "No online aggregate with reported free space is assigned to " + "SVM '%s'; cannot build an unsatisfiable request" + % self.svm_name + ) + requested = int(max_free) + 1024 ** 3 + if requested > TestData.ONTAP_MAX_VOLUME_SIZE: + self.skipTest( + "Largest aggregate free space (%d B) + 1 GiB exceeds the " + "ONTAP FlexVol maximum (%d B); the request would be refused " + "for the size limit rather than the aggregate shortage" + % (max_free, TestData.ONTAP_MAX_VOLUME_SIZE) + ) + + pool_name = self._throwaway_pool_name("NoSpace") + log_progress( + logger, "info", + "Requesting pool '%s' of %d B; largest online aggregate on SVM " + "'%s' has %d B free (expect reject)", + pool_name, requested, self.svm_name, max_free, + ) + try: + with self.assertRaises(CloudstackAPIException) as caught: + self.__class__.pool2 = self._create_pool( + pool_name=pool_name, capacitybytes=requested + ) + error_text = str(caught.exception) + log_progress( + logger, "info", + "Rejected no-space create for '%s': %s", pool_name, error_text, + ) + self.assertIn( + "No suitable aggregates", error_text, + "Expected the rejection to report 'No suitable aggregates', " + "got: %s" % error_text, + ) + self._assert_no_pool_named(pool_name) + self.assertIsNone( + self.ontap.get_volume(pool_name), + "ONTAP FlexVol '%s' was created despite the rejected pool " + "create" % pool_name, + ) + finally: + self._cleanup_throwaway_pool( + self.__class__.pool2, + flexvol_name=pool_name, + ep_name="cs-%s-%s" % (self.svm_name, pool_name), + ) + + # ------------------------------------------------------------------ + # Step 02 — Create primary storage pool # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_01_create_primary_storage_pool(self): + def test_02_create_primary_storage_pool(self): """ Create an NFS3 primary storage pool and verify: - CloudStack state is Up, type is NetworkFilesystem @@ -458,17 +525,80 @@ def test_01_create_primary_storage_pool(self): self._assert_pool_capacity(pool, "pool-created") # ------------------------------------------------------------------ - # Step 02 — Disable storage pool + # Step 03 — Reject create when that FlexVol name already exists + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_workflow"], required_hardware=True) + def test_03_reject_create_when_flexvol_name_exists(self): + """ + The pool from test_02 already has a FlexVol of that name on ONTAP. + A second createStoragePool with the same name must be rejected, and + the existing pool, FlexVol, and export policy must be left untouched. + + Verifies: + - createStoragePool raises CloudstackAPIException + - CloudStack still lists only the pool from test_02 + - the existing ONTAP FlexVol is still online + - the existing export policy is still present + """ + pool = self.__class__.pool + self.assertIsNotNone(pool, "Pool absent — test_02 must pass first") + pool_name = pool.name + self.assertIsNotNone( + self.ontap.get_volume(pool_name), + "ONTAP FlexVol '%s' from test_02 is missing" % pool_name, + ) + ep_name = self.__class__.pool_ep_name + self.assertIsNotNone( + self.ontap.get_export_policy(ep_name), + "Export policy '%s' from test_02 is missing" % ep_name, + ) + log_progress( + logger, "info", + "Pool '%s' already owns a FlexVol; requesting another pool of " + "the same name (expect reject)", pool_name, + ) + duplicate = None + try: + with self.assertRaises(CloudstackAPIException) as caught: + duplicate = self._create_pool(pool_name=pool_name) + log_progress( + logger, "info", + "Rejected duplicate-name create for '%s': %s", + pool_name, caught.exception, + ) + self._assert_only_original_pool(pool) + ontap_vol = self.ontap.get_volume(pool_name) + self.assertIsNotNone( + ontap_vol, + "Existing ONTAP FlexVol '%s' was removed by the failed " + "pool create" % pool_name, + ) + self.assertEqual( + ontap_vol.get("state"), "online", + "Existing ONTAP FlexVol '%s' should still be online, got '%s'" + % (pool_name, ontap_vol.get("state")), + ) + self.assertIsNotNone( + self.ontap.get_export_policy(ep_name), + "Export policy '%s' was removed by the rejected pool create" + % ep_name, + ) + finally: + self._warn_if_duplicate_pool(duplicate, pool) + + # ------------------------------------------------------------------ + # Step 04 — Disable storage pool # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_02_disable_storage_pool(self): + def test_04_disable_storage_pool(self): """ Disable the pool and verify: - CloudStack reports Disabled - ONTAP: FlexVol is still online and export policy unchanged """ - self.assertIsNotNone(self.__class__.pool, "Pool absent — test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent — test_02 must pass first") cmd = updateStoragePoolAPI.updateStoragePoolCmd() cmd.id = self.__class__.pool.id @@ -498,13 +628,13 @@ def test_02_disable_storage_pool(self): # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_03_enable_storage_pool(self): + def test_05_enable_storage_pool(self): """ Re-enable the pool and verify: - CloudStack reports Up - ONTAP: FlexVol is still online and export policy unchanged """ - self.assertIsNotNone(self.__class__.pool, "Pool absent — test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent — test_02 must pass first") cmd = updateStoragePoolAPI.updateStoragePoolCmd() cmd.id = self.__class__.pool.id @@ -534,20 +664,16 @@ def test_03_enable_storage_pool(self): # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_04_enter_maintenance_mode(self): + def test_06_enter_maintenance_mode(self): """ Put the pool into maintenance mode and verify: - CloudStack reports Maintenance - ONTAP: FlexVol is still online and export policy unchanged (maintenance is a CS-only state change) """ - self.assertIsNotNone(self.__class__.pool, "Pool absent — test_01 must pass first") - - cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.enableStorageMaintenance(cmd) + self.assertIsNotNone(self.__class__.pool, "Pool absent — test_02 must pass first") - result = self._poll_pool_state(self.__class__.pool.id, "Maintenance", timeout=120) + result = self._enter_maintenance(self.__class__.pool.id) self.assertEqual(result.state, "Maintenance") ontap_vol = self.ontap.get_volume(self.__class__.pool.name) @@ -570,7 +696,7 @@ def test_04_enter_maintenance_mode(self): # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_05_cancel_maintenance_mode(self): + def test_07_cancel_maintenance_mode(self): """ Cancel maintenance mode and verify the pool returns to Up. @@ -580,15 +706,9 @@ def test_05_cancel_maintenance_mode(self): - ONTAP: NFS export policy still present """ self.assertIsNotNone(self.__class__.pool, - "Pool absent — test_01 must pass first") + "Pool absent — test_02 must pass first") - cmd = cancelStorageMaintenance.cancelStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.cancelStorageMaintenance(cmd) - - result = self._poll_pool_state( - self.__class__.pool.id, "Up", timeout=120 - ) + result = self._exit_maintenance(self.__class__.pool.id) self.assertEqual( result.state, "Up", "Pool should be 'Up' after cancel maintenance, got '%s'" @@ -620,7 +740,7 @@ def test_05_cancel_maintenance_mode(self): # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_06_delete_pool_from_maintenance(self): + def test_08_delete_pool_from_maintenance(self): """ Enter maintenance mode then delete the storage pool. @@ -629,16 +749,13 @@ def test_06_delete_pool_from_maintenance(self): - ONTAP: FlexVol is deleted - ONTAP: NFS export policy is deleted """ - self.assertIsNotNone(self.__class__.pool, "Pool absent — test_01 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent — test_02 must pass first") pool = self.__class__.pool pool_name = pool.name ep_name = self.__class__.pool_ep_name # Pool is Up after test_05 succeeded; must enter Maintenance before deletion. - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(pool.id, "Maintenance", timeout=120) + self._enter_maintenance(pool.id) self._delete_pool(pool.id) self.__class__.pool = None @@ -671,7 +788,7 @@ def test_06_delete_pool_from_maintenance(self): # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_07_create_volume_on_pool(self): + def test_09_create_volume_on_pool(self): """ Create a new NFS3 pool and allocate a CloudStack data volume. For NFS3, createAsync is a no-op on ONTAP (volume is a CloudStack record @@ -737,7 +854,7 @@ def test_07_create_volume_on_pool(self): # ------------------------------------------------------------------ @attr(tags=["nfs3_workflow"], required_hardware=True) - def test_08_delete_volume_and_pool(self): + def test_10_delete_volume_and_pool(self): """ Delete the volume from test_07, enter maintenance, then force-delete the pool. @@ -748,8 +865,8 @@ def test_08_delete_volume_and_pool(self): - ONTAP: FlexVol deleted - ONTAP: export policy deleted """ - self.assertIsNotNone(self.__class__.pool, "Pool absent - test_07 must pass first") - self.assertIsNotNone(self.__class__.volume, "Volume absent - test_07 must pass first") + self.assertIsNotNone(self.__class__.pool, "Pool absent - test_09 must pass first") + self.assertIsNotNone(self.__class__.volume, "Volume absent - test_09 must pass first") pool = self.__class__.pool pool_name = pool.name @@ -779,3 +896,267 @@ def test_08_delete_volume_and_pool(self): ) self.__class__.pool2 = None self.__class__.pool2_ep_name = None + + def _throwaway_pool_name(self, suffix): + return "OntapNFS3%s_%d" % (suffix, random.randint(0, 99999)) + + + def _cleanup_throwaway_pool(self, pool, flexvol_name=None, ep_name=None): + """Best-effort teardown for an isolated test: CloudStack, then ONTAP. + + Never raises, so a failed assertion in the test body is the error + that surfaces. The KVM unmount happens while the export is still + reachable, i.e. before the FlexVol goes away. + """ + backend_cleanup_safe = pool is None + if pool is not None: + try: + listed = list_storage_pools(self.apiClient, id=pool.id) + except Exception: + listed = None + backend_cleanup_safe = not listed + if listed: + try: + if listed[0].state != "Maintenance": + self._enter_maintenance(pool.id) + except Exception as exc: + logger.warning( + "cleanup: could not put pool '%s' into Maintenance: %s", + pool.name, exc, + ) + try: + self._cleanup_kvm_storage_pool_mounts(pool.id) + except Exception as exc: + logger.warning( + "cleanup: could not safely unmount pool '%s': %s", + pool.name, exc, + ) + return + backend_cleanup_safe = True + try: + self._delete_pool(pool.id, forced=True) + except Exception as exc: + logger.warning( + "cleanup: could not delete pool '%s': %s", + pool.name, exc, + ) + if flexvol_name is None: + flexvol_name = pool.name + if backend_cleanup_safe and flexvol_name: + try: + if self.ontap.get_volume(flexvol_name) is not None: + self.ontap.offline_and_delete_volume(flexvol_name) + except Exception as exc: + logger.warning( + "cleanup: could not delete ONTAP FlexVol '%s': %s", + flexvol_name, exc, + ) + if backend_cleanup_safe and ep_name: + try: + if self.ontap.get_export_policy(ep_name) is not None: + self.ontap.delete_export_policy(ep_name) + except Exception as exc: + logger.warning( + "cleanup: could not delete export policy '%s': %s", + ep_name, exc, + ) + if pool is None: + return + try: + remaining = list_storage_pools(self.apiClient, id=pool.id) + except Exception: + remaining = None + if not remaining: + self.__class__.pool2 = None + self.__class__.pool2_ep_name = None + + + def _assert_only_original_pool(self, pool): + """Assert the rejected create did not add a second pool of this name.""" + try: + listed = list_storage_pools(self.apiClient, name=pool.name) or [] + except CloudstackAPIException: + listed = [] + same_name = [ + item for item in listed if getattr(item, "name", None) == pool.name + ] + self.assertEqual( + [item.id for item in same_name], + [pool.id], + "CloudStack should still list only pool '%s' after the rejected " + "create, found: %s" % (pool.name, same_name), + ) + current = list_storage_pools(self.apiClient, id=pool.id) + self.assertTrue( + current, + "Original pool '%s' disappeared after the rejected create" + % pool.name, + ) + self.assertEqual( + current[0].state, "Up", + "Original pool '%s' should still be Up, got '%s'" + % (pool.name, current[0].state), + ) + + def _warn_if_duplicate_pool(self, duplicate, pool): + """Leave a stray duplicate in place. + + Deleting it would remove the FlexVol and export policy that the + rest of the suite still uses. + """ + if duplicate is None or getattr(duplicate, "id", None) == pool.id: + return + logger.warning( + "Duplicate pool '%s' (id=%s) was created against the shared " + "FlexVol; leaving it in place so cleanup does not delete the " + "pool the rest of the suite uses", + pool.name, duplicate.id, + ) + + def _assert_no_pool_named(self, pool_name): + """Assert CloudStack holds no storage pool with this name.""" + try: + listed = list_storage_pools(self.apiClient, name=pool_name) + except CloudstackAPIException: + listed = None + self.assertFalse( + listed, + "CloudStack should hold no pool named '%s' after a rejected " + "create, found: %s" % (pool_name, listed), + ) + + + def _sweep_tracked_pool2(self): + """Clean a prior leftover before reusing the shared pool2 slot.""" + if self.__class__.pool2 is None: + return + self._cleanup_throwaway_pool( + self.__class__.pool2, + ep_name=self.__class__.pool2_ep_name, + ) + if self.__class__.pool2 is not None: + self.skipTest("A previously tracked pool could not be cleaned up safely") + + + def _create_throwaway_pool(self, suffix): + """Create an isolated pool, park it in pool2, and return it.""" + self._sweep_tracked_pool2() + pool = self._create_pool(pool_name=self._throwaway_pool_name(suffix)) + self.__class__.pool2 = pool + ep_name = self._get_export_policy_name(pool) + self.__class__.pool2_ep_name = ep_name + self.assertEqual( + pool.state, "Up", + "Throwaway pool '%s' should be 'Up', got '%s'" + % (pool.name, pool.state), + ) + self.assertIsNotNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol missing for throwaway pool '%s'" % pool.name, + ) + return pool, ep_name + + + @attr(tags=["nfs3_workflow"], required_hardware=True) + def test_11_delete_pool_with_flexvol_predeleted(self): + """ + Delete the backing FlexVol directly on ONTAP, then delete the empty + pool through CloudStack. Deletion must tolerate the missing volume + rather than leaving an undeletable pool behind. + + Verifies: + - deleteStoragePool succeeds with the FlexVol already gone + - the pool is removed from CloudStack + - the export policy is cleaned up as well + """ + pool, ep_name = self._create_throwaway_pool("PreDelVol") + try: + self._enter_maintenance(pool.id) + # Unmount on the KVM hosts first: once the FlexVol is gone the + # export is unreachable and a stale mount can wedge the host. + self._cleanup_kvm_storage_pool_mounts(pool.id) + log_progress( + logger, "info", + "Deleting ONTAP FlexVol '%s' behind CloudStack's back", + pool.name, + ) + self.ontap.offline_and_delete_volume(pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol '%s' still present after direct deletion" + % pool.name, + ) + + self._delete_pool(pool.id) + self._assert_pool_gone_from_cs(pool.id, pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol '%s' reappeared after pool deletion" + % pool.name, + ) + self.assertIsNone( + self.ontap.get_export_policy(ep_name), + "Export policy '%s' still exists after pool deletion" + % ep_name, + ) + self.__class__.pool2 = None + self.__class__.pool2_ep_name = None + finally: + self._cleanup_throwaway_pool( + self.__class__.pool2, + flexvol_name=pool.name, + ep_name=ep_name, + ) + + + @attr(tags=["nfs3_workflow"], required_hardware=True) + def test_12_delete_pool_with_export_policy_predeleted(self): + """ + Delete the NFS export policy directly on ONTAP, then delete the + empty pool through CloudStack. Deletion must tolerate the missing + policy and still remove the FlexVol. + + Verifies: + - deleteStoragePool succeeds with the export policy already gone + - the pool is removed from CloudStack + - ONTAP: the FlexVol is deleted + """ + pool, ep_name = self._create_throwaway_pool("PreDelEp") + try: + self.assertIsNotNone( + self.ontap.get_export_policy(ep_name), + "Export policy '%s' missing for the fresh throwaway pool" + % ep_name, + ) + self._enter_maintenance(pool.id) + # Dropping the policy revokes the hosts' NFS access, so unmount + # before it disappears. + self._cleanup_kvm_storage_pool_mounts(pool.id) + log_progress( + logger, "info", + "Deleting NFS export policy '%s' behind CloudStack's back", + ep_name, + ) + self.ontap.reassign_volume_export_policy(pool.name) + self.ontap.delete_export_policy(ep_name) + self.assertIsNone( + self.ontap.get_export_policy(ep_name), + "Export policy '%s' still present after direct deletion" + % ep_name, + ) + + self._delete_pool(pool.id) + self._assert_pool_gone_from_cs(pool.id, pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "ONTAP FlexVol '%s' still exists after pool deletion" + % pool.name, + ) + self.__class__.pool2 = None + self.__class__.pool2_ep_name = None + finally: + self._cleanup_throwaway_pool( + self.__class__.pool2, + flexvol_name=pool.name, + ep_name=ep_name, + ) diff --git a/test/integration/plugins/ontap/nfs3/pool/test_pool_with_volumes.py b/test/integration/plugins/ontap/nfs3/pool/test_pool_with_volumes.py index b266c1920f9d..5fa74ff843be 100644 --- a/test/integration/plugins/ontap/nfs3/pool/test_pool_with_volumes.py +++ b/test/integration/plugins/ontap/nfs3/pool/test_pool_with_volumes.py @@ -30,6 +30,17 @@ 06 Forced=False delete rejected — pool stays in Maintenance (negative) 07 Cleanup — cancel maintenance, delete volume, force-delete pool +Isolated tests (test_08 onwards) run after that workflow and share no state +with it. Each one builds its own pool plus CloudStack volume. Tests 08 and +09 remove one ONTAP object and then force-delete the pool. Tests 10-12 break +only that pool's view on the KVM host: + + 08 FlexVol already removed on ONTAP + 09 Export policy already removed on ONTAP + 10 Libvirt pool for this storage pool inactive on the KVM host + 11 NFS mount for this pool read-only on the KVM host + 12 NFS mount point for this pool missing on the KVM host + Prerequisites: - CloudStack management server with the NetApp ONTAP plugin deployed - KVM cluster registered in CloudStack @@ -69,7 +80,6 @@ cancelStorageMaintenance, createStoragePool as createStoragePoolAPI, deleteVolume as deleteVolumeAPI, - enableStorageMaintenance, updateStoragePool as updateStoragePoolAPI, ) from marvin.cloudstackException import CloudstackAPIException @@ -150,6 +160,7 @@ class TestOntapNFS3PoolWithVolumes(OntapTestBase): """ pool_ep_name = None # NFS export policy name extracted at pool creation + pool2_ep_name = None # export policy of the pool an isolated test builds _vol_name_prefix = "OntapNFS3WV" @@ -230,15 +241,6 @@ def _get_export_policy_name(self, pool): ep_name = "cs-%s-%s" % (self.svm_name, pool.name) return ep_name - def _volume_exists_in_cs(self, vol_id): - """Return True if the volume is still listed by CloudStack.""" - from marvin.cloudstackAPI import listVolumes as listVolumesAPI - cmd = listVolumesAPI.listVolumesCmd() - cmd.id = vol_id - cmd.listall = True - vols = self.apiClient.listVolumes(cmd) or [] - return len(vols) > 0 - def _assert_pool_capacity(self, pool, label): """Assert CloudStack capacity fields and ONTAP FlexVol size are consistent. @@ -462,11 +464,7 @@ def test_04_enter_maintenance_volume_present(self): self.assertIsNotNone(self.__class__.volume, "Volume absent — test_01 must pass first") - cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.enableStorageMaintenance(cmd) - - result = self._poll_pool_state(self.__class__.pool.id, "Maintenance", timeout=120) + result = self._enter_maintenance(self.__class__.pool.id) self.assertEqual( result.state, "Maintenance", "Pool should be 'Maintenance', got '%s'" % result.state @@ -515,13 +513,7 @@ def test_05_cancel_maintenance_with_volume(self): self.assertIsNotNone(self.__class__.volume, "Volume absent — test_01 must pass first") - cmd = cancelStorageMaintenance.cancelStorageMaintenanceCmd() - cmd.id = self.__class__.pool.id - self.apiClient.cancelStorageMaintenance(cmd) - - result = self._poll_pool_state( - self.__class__.pool.id, "Up", timeout=120 - ) + result = self._exit_maintenance(self.__class__.pool.id) self.assertEqual( result.state, "Up", "Pool should be 'Up' after cancel maintenance, got '%s'" % result.state @@ -576,10 +568,7 @@ def test_06_forced_false_delete_rejected(self): # Pool is Up after test_05 (cancel maintenance); re-enter Maintenance # before attempting the delete so it reaches the forced=False gate. - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = self.__class__.pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(self.__class__.pool.id, "Maintenance", timeout=120) + self._enter_maintenance(self.__class__.pool.id) with self.assertRaises(CloudstackAPIException, msg="deleteStoragePool(forced=False) with a live " @@ -690,16 +679,8 @@ def test_07_force_delete_pool_and_cleanup(self): self.__class__.volume = None vol = None try: - mc = enableStorageMaintenance.enableStorageMaintenanceCmd() - mc.id = pool.id - self.apiClient.enableStorageMaintenance(mc) - deadline = time.time() + 60 - while time.time() < deadline: - ps = list_storage_pools(self.apiClient, id=pool.id) - if ps and ps[0].state == "Maintenance": - pool_state = "Maintenance" - break - time.sleep(5) + self._enter_maintenance(pool.id, timeout=60) + pool_state = "Maintenance" except Exception: pass @@ -749,3 +730,482 @@ def test_07_force_delete_pool_and_cleanup(self): policy, "NFS export policy '%s' should be removed after cleanup" % ep_name ) + + # ================================================================== + # Isolated tests — appended after the sequential workflow above. + # + # Each one creates its own pool and CloudStack volume in the pool2 / + # volume2 slots (which OntapTestBase.tearDownClass also sweeps), runs a + # single scenario, and cleans up in a finally block. They never reuse a + # pool destroyed by another test. + # ================================================================== + + def _create_isolated_pool_with_volume(self, label): + """Build a fresh pool plus CS volume for one isolated scenario. + + Returns ``(pool, volume, export_policy_name)``. Any pool a previous + isolated test could not clean up is swept first so its ONTAP objects + are never orphaned by the overwrite of the pool2 slot. + """ + if self.__class__.pool2 is not None: + self._cleanup_isolated_pool( + self.__class__.pool2, self.__class__.pool2_ep_name, + "leftover-from-previous-isolated-test" + ) + + pool = self._create_pool() + self.__class__.pool2 = pool + ep_name = self._get_export_policy_name(pool) + self.__class__.pool2_ep_name = ep_name + logger.info("[%s] created isolated pool '%s' (ep '%s')", + label, pool.name, ep_name) + + self.assertEqual( + pool.state, "Up", + "[%s] new pool state should be 'Up', got '%s'" % (label, pool.state) + ) + ontap_vol = self.ontap.get_volume(pool.name) + self.assertIsNotNone( + ontap_vol, + "[%s] ONTAP FlexVol not found for new pool '%s'" % (label, pool.name) + ) + self.assertIsNotNone( + self.ontap.get_export_policy(ep_name), + "[%s] export policy '%s' not found for new pool '%s'" + % (label, ep_name, pool.name) + ) + + vol = self._create_volume(pool.id) + self.__class__.volume2 = vol + self.assertIsNotNone(vol, "[%s] createVolume returned None" % label) + return pool, vol, ep_name + + def _assert_pool_absent(self, pool, label, delete_error=None): + """Assert CloudStack no longer lists the pool.""" + try: + remaining = list_storage_pools(self.apiClient, id=pool.id) + except CloudstackAPIException: + remaining = None + self.assertFalse( + remaining, + "[%s] pool '%s' is still listed after deleteStoragePool(forced=True)%s" + % (label, pool.name, + "; the API raised: %s" % delete_error if delete_error else "") + ) + + def _cleanup_isolated_pool(self, pool, ep_name, label): + """Best-effort teardown of one isolated pool and its ONTAP objects.""" + if pool is None: + return + try: + listed = list_storage_pools(self.apiClient, id=pool.id) + except CloudstackAPIException: + listed = None + backend_cleanup_safe = not listed + if listed: + self._remove_cs_volume(pool, self.__class__.volume2, label) + try: + listed = list_storage_pools(self.apiClient, id=pool.id) or listed + if listed[0].state != "Maintenance": + self._enter_maintenance(pool.id) + self._cleanup_kvm_storage_pool_mounts(pool.id) + except Exception as exc: + logger.warning("[%s] could not safely unmount pool '%s': %s", + label, pool.name, exc) + return + backend_cleanup_safe = True + try: + self._delete_pool(pool.id, forced=True) + except Exception as exc: + logger.warning("[%s] could not force-delete pool '%s': %s", + label, pool.name, exc) + if not backend_cleanup_safe: + return + try: + self.ontap.offline_and_delete_volume(pool.name) + except Exception as exc: + logger.warning("[%s] ONTAP FlexVol cleanup for '%s' failed: %s", + label, pool.name, exc) + if ep_name: + try: + self.ontap.delete_export_policy(ep_name) + except Exception as exc: + logger.warning( + "[%s] ONTAP export policy cleanup for '%s' failed: %s", + label, ep_name, exc) + try: + listed = list_storage_pools(self.apiClient, id=pool.id) + except CloudstackAPIException: + listed = None + if not listed: + self.__class__.pool2 = None + self.__class__.pool2_ep_name = None + + # ------------------------------------------------------------------ + # Step 08 — FlexVol deleted on ONTAP before the pool delete (negative) + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_with_volumes"], required_hardware=True) + def test_08_delete_pool_with_volume_flexvol_missing(self): + """ + Force-delete a pool that still owns a CloudStack volume after its + ONTAP FlexVol has been removed behind CloudStack's back. + + Uses its own pool and volume, so the pool is known to be healthy up to + the point the FlexVol is destroyed. Verifies: + - deleteStoragePool is rejected while the CS volume still exists + - deleteStoragePool(forced=True) tolerates the missing FlexVol + - the CloudStack pool record is removed + - the leftover CS volume record can still be cleaned up + """ + label = "flexvol-missing" + pool, vol, ep_name = self._create_isolated_pool_with_volume(label) + try: + self._enter_maintenance(pool.id) + + # Unmount on every KVM host while the NFS export is still + # reachable, before the FlexVol is destroyed underneath it. + self._cleanup_kvm_storage_pool_mounts(pool.id) + + self.ontap.offline_and_delete_volume(pool.name) + self.assertIsNone( + self.ontap.get_volume(pool.name), + "[%s] ONTAP FlexVol '%s' should be gone before the pool delete" + % (label, pool.name) + ) + + # CloudStack rejects deleteStoragePool while the pool still owns + # a volume, even with forced=True, so the volume goes first. + with self.assertRaises(CloudstackAPIException): + self._delete_pool(pool.id, forced=True) + self.assertTrue( + self._remove_cs_volume(pool, vol, label), + "[%s] CloudStack volume could not be deleted before the pool " + "delete" % label + ) + try: + self._enter_maintenance(pool.id) + except Exception as exc: + logger.warning("[%s] could not enter maintenance on '%s': %s", + label, pool.name, exc) + + delete_error = None + try: + self._delete_pool(pool.id, forced=True) + except CloudstackAPIException as exc: + delete_error = exc + + self._assert_pool_absent(pool, label, delete_error) + self.assertIsNone( + delete_error, + "[%s] deleteStoragePool(forced=True) should tolerate a missing " + "FlexVol, but raised: %s" % (label, delete_error) + ) + + self._purge_cs_volume_record(vol, label) + finally: + self._cleanup_isolated_pool(pool, ep_name, label) + + # ------------------------------------------------------------------ + # Step 09 — Export policy deleted on ONTAP before the delete (negative) + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_with_volumes"], required_hardware=True) + def test_09_delete_pool_with_volume_export_policy_missing(self): + """ + Force-delete a pool that still owns a CloudStack volume after its NFS + export policy has been removed behind CloudStack's back. + + Uses its own pool and volume. Unlike test_08 the FlexVol is still + present, so the plugin is expected to remove it as part of the delete. + Verifies: + - deleteStoragePool is rejected while the CS volume still exists + - deleteStoragePool(forced=True) tolerates the missing export policy + - the CloudStack pool record is removed + - the ONTAP FlexVol is deleted despite the missing policy + """ + label = "export-policy-missing" + pool, vol, ep_name = self._create_isolated_pool_with_volume(label) + try: + self._enter_maintenance(pool.id) + + # Unmount before the export policy goes away, otherwise the KVM + # hosts are left holding a mount they can no longer reach. + self._cleanup_kvm_storage_pool_mounts(pool.id) + + self.ontap.reassign_volume_export_policy(pool.name) + self.ontap.delete_export_policy(ep_name) + self.assertIsNone( + self.ontap.get_export_policy(ep_name), + "[%s] export policy '%s' should be gone before the pool delete" + % (label, ep_name) + ) + + # CloudStack rejects deleteStoragePool while the pool still owns + # a volume, even with forced=True, so the volume goes first. + with self.assertRaises(CloudstackAPIException): + self._delete_pool(pool.id, forced=True) + self.assertTrue( + self._remove_cs_volume(pool, vol, label), + "[%s] CloudStack volume could not be deleted before the pool " + "delete" % label + ) + try: + self._enter_maintenance(pool.id) + except Exception as exc: + logger.warning("[%s] could not enter maintenance on '%s': %s", + label, pool.name, exc) + + delete_error = None + try: + self._delete_pool(pool.id, forced=True) + except CloudstackAPIException as exc: + delete_error = exc + + self._assert_pool_absent(pool, label, delete_error) + self.assertIsNone( + delete_error, + "[%s] deleteStoragePool(forced=True) should tolerate a missing " + "export policy, but raised: %s" % (label, delete_error) + ) + + self.assertIsNone( + self.ontap.get_volume(pool.name), + "[%s] ONTAP FlexVol '%s' should have been deleted with the pool" + % (label, pool.name) + ) + self.assertIsNone( + self.ontap.get_export_policy(ep_name), + "[%s] export policy '%s' reappeared during the pool delete" + % (label, ep_name) + ) + + self._purge_cs_volume_record(vol, label) + finally: + self._cleanup_isolated_pool(pool, ep_name, label) + + def _libvirt_pool_present(self, host_ip, pool_uuid): + output = self._kvm_run( + host_ip, + "virsh pool-info '%s' >/dev/null 2>&1 && echo PRESENT || true" + % pool_uuid + ) + return any(line.strip() == "PRESENT" for line in output) + + def _restore_libvirt_pool(self, host_ip, pool_uuid, xml_path, undefined): + quoted = pool_uuid.replace("'", "'\"'\"'") + quoted_xml = xml_path.replace("'", "'\"'\"'") + if undefined and xml_path: + self._kvm_run( + host_ip, + "virsh pool-define '%s' >/dev/null 2>&1 || true" % quoted_xml + ) + cleanup = "rm -f '%s'" % quoted_xml if xml_path else "true" + self._kvm_run( + host_ip, + "virsh pool-start '%s' >/dev/null 2>&1 || true; %s" + % (quoted, cleanup) + ) + + # ------------------------------------------------------------------ + # Step 10 — Test libvirt pool made inactive + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_with_volumes"], required_hardware=True) + def test_10_libvirt_pool_inactive(self): + """ + Destroy only the isolated pool's libvirt definition so it is inactive, + then enter maintenance. The libvirt pool is started again before + CloudStack cleanup. + """ + label = "libvirt-pool-inactive" + host_ip = self._kvm_host_ip() + pool, vol, ep_name = self._create_isolated_pool_with_volume(label) + xml_path = "/tmp/cs-ontap-%s.xml" % pool.id + destroyed = False + temporarily_defined = False + try: + if not self._libvirt_pool_present(host_ip, pool.id): + self.skipTest( + "Libvirt has no storage pool named '%s' on %s" + % (pool.id, host_ip) + ) + self._kvm_run( + host_ip, + "virsh pool-dumpxml '%s' > '%s' && " + "virsh pool-destroy '%s' >/dev/null 2>&1 || true" + % (pool.id, xml_path, pool.id) + ) + destroyed = True + info = self._kvm_run( + host_ip, "virsh pool-info '%s' 2>/dev/null || true" % pool.id + ) + if not any(line.strip() for line in info): + self._kvm_run( + host_ip, + "virsh pool-define '%s' >/dev/null" % xml_path + ) + temporarily_defined = True + info = self._kvm_run( + host_ip, + "virsh pool-info '%s' 2>/dev/null || true" % pool.id + ) + self.assertTrue( + any("inactive" in line.lower() for line in info), + "[%s] libvirt pool was not inactive: %s" % (label, info) + ) + + self._enter_maintenance(pool.id) + self.assertTrue( + self._volume_exists_in_cs(vol.id), + "[%s] volume disappeared after maintenance" % label + ) + ontap_vol = self.ontap.get_volume(pool.name) + self.assertIsNotNone(ontap_vol, "[%s] FlexVol disappeared" % label) + self.assertEqual(ontap_vol.get("state"), "online") + self.assertIsNotNone( + self.ontap.get_export_policy(ep_name), + "[%s] export policy disappeared" % label + ) + finally: + if destroyed: + try: + self._kvm_run( + host_ip, + "virsh pool-start '%s' >/dev/null 2>&1 || true; " + "%s; rm -f '%s'" + % ( + pool.id, + "virsh pool-undefine '%s' >/dev/null 2>&1 || true" + % pool.id if temporarily_defined else "true", + xml_path, + ) + ) + except Exception as exc: + logger.warning( + "[%s] could not restart libvirt pool: %s", label, exc + ) + self._cleanup_isolated_pool(pool, ep_name, label) + + # ------------------------------------------------------------------ + # Step 11 — Test NFS mount made read-only + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_with_volumes"], required_hardware=True) + def test_11_nfs_mount_read_only(self): + """ + Remount only the isolated test pool read-only, verify writes fail, + then verify the pool can still enter maintenance. + """ + label = "nfs-mount-read-only" + host_ip = self._kvm_host_ip() + pool, vol, ep_name = self._create_isolated_pool_with_volume(label) + mount_path = "/mnt/%s" % pool.id + quoted_mount = mount_path.replace("'", "'\"'\"'") + remounted_read_only = False + try: + mounted = self._kvm_run( + host_ip, + "mountpoint -q '%s' && echo MOUNTED || true" % quoted_mount + ) + self.assertTrue( + any(line.strip() == "MOUNTED" for line in mounted), + "[%s] %s was not mounted before the test" + % (label, mount_path) + ) + + self._kvm_run( + host_ip, "mount -o remount,ro '%s'" % quoted_mount + ) + remounted_read_only = True + options = self._kvm_run( + host_ip, + "findmnt -n -o OPTIONS --target '%s'" % quoted_mount + ) + self.assertTrue( + any( + "ro" in [item.strip() for item in line.split(",")] + for line in options + ), + "[%s] %s was not remounted read-only: %s" + % (label, mount_path, options) + ) + write_probe = self._kvm_run( + host_ip, + "if touch '%s/.cloudstack-ro-probe' 2>/dev/null; then " + "rm -f '%s/.cloudstack-ro-probe'; echo WRITABLE; " + "else echo READ_ONLY; fi" % (quoted_mount, quoted_mount) + ) + self.assertTrue( + any(line.strip() == "READ_ONLY" for line in write_probe), + "[%s] writes unexpectedly succeeded on %s" + % (label, mount_path) + ) + + self._enter_maintenance(pool.id) + self.assertTrue(self._volume_exists_in_cs(vol.id)) + ontap_vol = self.ontap.get_volume(pool.name) + self.assertIsNotNone(ontap_vol, "[%s] FlexVol disappeared" % label) + self.assertEqual(ontap_vol.get("state"), "online") + self.assertIsNotNone(self.ontap.get_export_policy(ep_name)) + finally: + if remounted_read_only: + try: + self._kvm_run( + host_ip, + "if mountpoint -q '%s'; then " + "mount -o remount,rw '%s'; fi" + % (quoted_mount, quoted_mount) + ) + except Exception as exc: + logger.warning( + "[%s] could not restore read-write mount: %s", + label, exc + ) + self._cleanup_isolated_pool(pool, ep_name, label) + + # ------------------------------------------------------------------ + # Step 12 — Test NFS mount point deleted + # ------------------------------------------------------------------ + + @attr(tags=["nfs3_with_volumes"], required_hardware=True) + def test_12_nfs_mount_point_deleted(self): + """ + Unmount only the isolated test pool and delete its mount-point + directory, then verify the pool can still enter maintenance. + """ + label = "nfs-mount-point-deleted" + host_ip = self._kvm_host_ip() + pool, vol, ep_name = self._create_isolated_pool_with_volume(label) + mount_path = "/mnt/%s" % pool.id + quoted_mount = mount_path.replace("'", "'\"'\"'") + try: + mounted = self._kvm_run( + host_ip, + "mountpoint -q '%s' && echo MOUNTED || true" % quoted_mount + ) + self.assertTrue( + any(line.strip() == "MOUNTED" for line in mounted), + "[%s] %s was not mounted before the test" + % (label, mount_path) + ) + + removed = self._kvm_run( + host_ip, + "umount -f -l '%s' && rmdir '%s'; " + "if [ ! -e '%s' ]; then echo REMOVED; fi" + % (quoted_mount, quoted_mount, quoted_mount) + ) + self.assertTrue( + any(line.strip() == "REMOVED" for line in removed), + "[%s] could not remove mount point %s" % (label, mount_path) + ) + + self._enter_maintenance(pool.id) + self.assertTrue(self._volume_exists_in_cs(vol.id)) + ontap_vol = self.ontap.get_volume(pool.name) + self.assertIsNotNone(ontap_vol, "[%s] FlexVol disappeared" % label) + self.assertEqual(ontap_vol.get("state"), "online") + self.assertIsNotNone(self.ontap.get_export_policy(ep_name)) + finally: + self._cleanup_isolated_pool(pool, ep_name, label) diff --git a/test/integration/plugins/ontap/nfs3/pool/test_zone_scoped_pool.py b/test/integration/plugins/ontap/nfs3/pool/test_zone_scoped_pool.py index 88a6309f1ee1..21260151da9f 100644 --- a/test/integration/plugins/ontap/nfs3/pool/test_zone_scoped_pool.py +++ b/test/integration/plugins/ontap/nfs3/pool/test_zone_scoped_pool.py @@ -23,9 +23,9 @@ hosts in the zone to the pool and creates an NFS export policy covering their IPs. -Workflow: +Test order — sequential workflow that must run in order: 01 Create zone-scoped NFS3 pool — pool.state Up; ONTAP FlexVol online; - export policy has all cluster host IPs + export policy has all cluster host IPs 02 Disable zone-scoped pool — pool.state Disabled; FlexVol unchanged 03 Enable zone-scoped pool — pool.state Up; FlexVol unchanged 04 Delete zone-scoped pool — pool gone; FlexVol deleted; export policy deleted @@ -41,7 +41,7 @@ --marvin-config=test/integration/plugins/ontap/ontap.cfg \\ test/integration/plugins/ontap/nfs3/pool/test_zone_scoped_pool.py -v -Note: Tests 01-04 share class-level state (sequential). Running a single test +Note: Tests share class-level state (sequential). Running a single test with -m "test_NN" will invoke setUpClass but the guard assertion will fail immediately if earlier steps have not yet run. Always run the full suite. """ @@ -55,13 +55,17 @@ from marvin.cloudstackAPI import ( createStoragePool as createStoragePoolAPI, - enableStorageMaintenance, updateStoragePool as updateStoragePoolAPI, ) from marvin.lib.base import StoragePool from marvin.lib.common import list_storage_pools -from ontap_test_base import OntapRestClient, OntapTestBase, _parse_pool_details, get_datacenter_config +from ontap_test_base import ( + OntapRestClient, + OntapTestBase, + _parse_pool_details, + get_datacenter_config, +) logger = logging.getLogger("TestOntapZoneScopedPool") @@ -380,10 +384,7 @@ def test_04_delete_zone_scoped_pool(self): pool_name = pool.name ep_name = self.__class__.pool_ep_name - maint_cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() - maint_cmd.id = pool.id - self.apiClient.enableStorageMaintenance(maint_cmd) - self._poll_pool_state(pool.id, "Maintenance", timeout=120) + self._enter_maintenance(pool.id) # Unmount the NFS on each KVM host BEFORE deleteStoragePool removes # the ONTAP export. Without this, the mount becomes stale and diff --git a/test/integration/plugins/ontap/ontap_test_base.py b/test/integration/plugins/ontap/ontap_test_base.py index 17bec2c1251c..bf6a084308a5 100644 --- a/test/integration/plugins/ontap/ontap_test_base.py +++ b/test/integration/plugins/ontap/ontap_test_base.py @@ -30,6 +30,7 @@ import logging import random +import re import requests import sys import time @@ -41,6 +42,7 @@ createVolume as createVolumeAPI, deleteStoragePool as deleteStoragePoolAPI, deleteVolume as deleteVolumeAPI, + destroyVolume as destroyVolumeAPI, enableStorageMaintenance, listDiskOfferings as listDiskOfferingsAPI, listVirtualMachines as listVirtualMachinesAPI, @@ -48,6 +50,7 @@ updateStoragePool as updateStoragePoolAPI, ) from marvin.cloudstackAPI import listHosts as listHostsAPI +from marvin.cloudstackException import CloudstackAPIException from marvin.cloudstackTestCase import cloudstackTestCase from marvin.jsonHelper import jsonDump from marvin.lib.base import Account, DiskOffering @@ -166,25 +169,249 @@ def _delete(self, path, params=None): url = self._base + path resp = requests.delete(url, auth=self._auth, params=params, verify=False, timeout=30) - if not resp.ok: - raise requests.HTTPError( - "%s for url: %s body=%s" - % (resp.status_code, resp.url, resp.text), - response=resp, - ) + self._raise_http(resp) + return self._response_data(resp) - def _patch(self, path, payload=None, params=None): - url = self._base + path - resp = requests.patch(url, auth=self._auth, params=params, - json=payload or {}, verify=False, timeout=30) - resp.raise_for_status() + @staticmethod + def _response_data(resp): + """Return response JSON, including async job UUID from Location.""" if resp.content: try: return resp.json() except ValueError: - return None + pass + location = resp.headers.get("Location", "") + marker = "/cluster/jobs/" + if marker in location: + return {"job": {"uuid": location.split(marker, 1)[1].split("?", 1)[0]}} return None + def _raise_http(self, resp): + if resp.ok: + return + body = "" + try: + body = resp.text + except Exception: + body = "" + raise requests.HTTPError( + "%s Client Error: %s for url: %s body: %s" + % (resp.status_code, resp.reason, resp.url, body), + response=resp, + ) + + def _patch(self, path, params=None, json_body=None, timeout=60): + url = self._base + path + resp = requests.patch( + url, auth=self._auth, params=params, json=json_body, + verify=False, timeout=timeout, + ) + self._raise_http(resp) + return self._response_data(resp) + + def _post(self, path, params=None, data=None, json_body=None, timeout=60, + headers=None, files=None): + url = self._base + path + resp = requests.post( + url, auth=self._auth, params=params, data=data, json=json_body, + headers=headers, files=files, verify=False, timeout=timeout, + ) + self._raise_http(resp) + return self._response_data(resp) + + def _wait_for_job(self, response, timeout=120): + """Wait for an asynchronous ONTAP response, if it contains a job.""" + job = (response or {}).get("job") or {} + job_uuid = job.get("uuid") + if not job_uuid: + return response + deadline = time.time() + timeout + while time.time() < deadline: + current = self._get("/cluster/jobs/%s" % job_uuid) + state = (current.get("state") or "").lower() + if state == "success": + return current + if state in ("failure", "failed", "error"): + message = current.get("message") or "unknown ONTAP job failure" + raise RuntimeError("ONTAP job %s failed: %s" % (job_uuid, message)) + time.sleep(2) + raise RuntimeError("Timed out waiting for ONTAP job %s" % job_uuid) + + def _svm_aggregates(self, svm_name): + """Return detailed aggregate records assigned to an SVM.""" + svms = self._get( + "/svm/svms", + params={"name": svm_name, "fields": "aggregates"}, + ).get("records", []) + if not svms: + raise RuntimeError("ONTAP SVM '%s' was not found" % svm_name) + aggregates = [] + for aggregate in svms[0].get("aggregates", []): + uuid = aggregate.get("uuid") + if not uuid: + continue + aggregates.append(self._get( + "/storage/aggregates/%s" % uuid, + params={"fields": "name,uuid,state,space.block_storage.available"}, + )) + return aggregates + + def max_online_aggregate_available_bytes(self, svm_name): + """Return the largest free-space value among assigned online aggregates.""" + available = [] + for aggregate in self._svm_aggregates(svm_name): + if (aggregate.get("state") or "").lower() != "online": + continue + free = (aggregate.get("space", {}) + .get("block_storage", {}).get("available")) + if free is not None: + available.append(int(float(free))) + if not available: + raise RuntimeError( + "SVM '%s' has no online aggregate with space data" % svm_name + ) + return max(available) + + def create_flexvol(self, svm_name, volume_name, size_bytes, nas_path=True): + """Create a thin FlexVol directly on a suitable SVM aggregate.""" + suitable = [] + for aggregate in self._svm_aggregates(svm_name): + free = (aggregate.get("space", {}) + .get("block_storage", {}).get("available")) + if ((aggregate.get("state") or "").lower() == "online" + and free is not None and int(float(free)) > int(size_bytes)): + suitable.append((int(float(free)), aggregate)) + if not suitable: + raise RuntimeError( + "No ONTAP aggregate can hold FlexVol '%s'" % volume_name + ) + aggregate = max(suitable, key=lambda item: item[0])[1] + request = { + "name": volume_name, + "svm": {"name": svm_name}, + "size": int(size_bytes), + "aggregates": [{"name": aggregate.get("name")}], + "guarantee": {"type": "none"}, + } + if nas_path: + request["nas"] = {"path": "/" + volume_name} + response = self._post( + "/storage/volumes", + params={"return_timeout": 15}, + json_body=request, + ) + self._wait_for_job(response) + deadline = time.time() + 120 + while time.time() < deadline: + volume = self.get_volume(volume_name) + if volume is not None: + return volume + time.sleep(2) + raise RuntimeError( + "FlexVol '%s' was not visible after creation" % volume_name + ) + + def offline_and_delete_volume(self, name): + """Offline and delete a FlexVol directly; no-op when already absent.""" + volume = self.get_volume(name) + if not volume: + return + uuid = volume.get("uuid") + if not uuid: + raise RuntimeError("FlexVol '%s' has no UUID" % name) + if (volume.get("nas") or {}).get("path"): + response = self._patch( + "/storage/volumes/%s" % uuid, + params={"return_timeout": 15}, + json_body={"nas": {"path": ""}}, + ) + self._wait_for_job(response) + if (volume.get("state") or "").lower() != "offline": + response = self._patch( + "/storage/volumes/%s" % uuid, + params={"return_timeout": 15}, + json_body={"state": "offline"}, + ) + self._wait_for_job(response) + response = self._delete( + "/storage/volumes/%s" % uuid, + params={"return_timeout": 15}, + ) + self._wait_for_job(response) + deadline = time.time() + 120 + while time.time() < deadline: + if self.get_volume(name) is None: + return + time.sleep(2) + raise RuntimeError("FlexVol '%s' still exists after deletion" % name) + + def reassign_volume_export_policy(self, volume_name, policy_name="default"): + """Assign a FlexVol to another export policy before deleting its policy.""" + volume = self.get_volume(volume_name) + if not volume: + raise RuntimeError("FlexVol '%s' was not found" % volume_name) + uuid = volume.get("uuid") + if not uuid: + raise RuntimeError("FlexVol '%s' has no UUID" % volume_name) + response = self._patch( + "/storage/volumes/%s" % uuid, + params={"return_timeout": 15}, + json_body={"nas": {"export_policy": {"name": policy_name}}}, + ) + self._wait_for_job(response) + + def create_igroup(self, svm_name, igroup_name, initiator_iqn): + """Create an ONTAP igroup holding a single initiator.""" + self._post( + "/protocols/san/igroups", + json_body={ + "svm": {"name": svm_name}, + "name": igroup_name, + "os_type": "linux", + "protocol": "iscsi", + "initiators": [{"name": initiator_iqn}], + }, + ) + igroup = self.get_igroup(svm_name, igroup_name) + if igroup is None: + raise RuntimeError( + "ONTAP igroup '%s' absent right after creation" % igroup_name + ) + return igroup + + def delete_igroup(self, svm_name, igroup_name): + """Delete an ONTAP igroup by name; no-op when already absent.""" + igroup = self.get_igroup(svm_name, igroup_name) + if not igroup: + return + uuid = igroup.get("uuid") + if not uuid: + raise RuntimeError("ONTAP igroup '%s' has no UUID" % igroup_name) + self._delete("/protocols/san/igroups/%s" % uuid) + + def create_lun_map(self, svm_name, lun_path, igroup_name): + """Map an existing LUN to an existing igroup.""" + response = self._post( + "/protocols/san/lun-maps", + params={"return_timeout": 15}, + json_body={ + "svm": {"name": svm_name}, + "lun": {"name": lun_path}, + "igroup": {"name": igroup_name}, + }, + ) + self._wait_for_job(response) + + def delete_lun_map(self, lun_map): + """Delete one LUN map returned by list_lun_maps_for_volume.""" + lun_uuid = lun_map.get("lun", {}).get("uuid") + igroup_uuid = lun_map.get("igroup", {}).get("uuid") + if not lun_uuid or not igroup_uuid: + raise RuntimeError("ONTAP LUN map is missing LUN or igroup UUID") + self._delete( + "/protocols/san/lun-maps/%s/%s" % (lun_uuid, igroup_uuid) + ) + def delete_volume(self, name): """Delete the ONTAP FlexVol with the given name. No-op if not found.""" data = self._get("/storage/volumes", params={"name": name}) @@ -214,7 +441,7 @@ def get_volume(self, name): uuid = records[0].get("uuid") if uuid: return self._get("/storage/volumes/%s" % uuid, - params={"fields": "name,uuid,state,space"}) + params={"fields": "name,uuid,state,space,nas.path,nas.export_policy"}) return records[0] # -- NFS helpers --------------------------------------------------------- @@ -244,6 +471,29 @@ def get_data_lifs(self, svm_name): return [r.get("ip", {}).get("address") for r in records if r.get("ip", {}).get("address")] + def get_iscsi_data_lifs(self, svm_name): + """Return iSCSI data LIF addresses for the given SVM.""" + data = self._get( + "/network/ip/interfaces", + params={"svm.name": svm_name, "services": "data-iscsi", + "fields": "ip,name"} + ) + return [r.get("ip", {}).get("address") + for r in data.get("records", []) + if r.get("ip", {}).get("address")] + + def get_iscsi_target_iqn(self, svm_name): + """Return the SVM iSCSI target IQN, or None when the service is absent.""" + data = self._get( + "/protocols/san/iscsi/services", + params={"svm.name": svm_name, "fields": "target", + "max_records": 1} + ) + records = data.get("records", []) + if not records: + return None + return (records[0].get("target") or {}).get("name") + # -- iSCSI helpers ------------------------------------------------------- def get_igroup(self, svm_name, igroup_name): @@ -276,7 +526,8 @@ def list_lun_maps_for_volume(self, svm_name, vol_name): prefix = "/vol/%s/" % vol_name data = self._get("/protocols/san/lun-maps", params={"svm.name": svm_name, - "fields": "lun.name,igroup.name"}) + "fields": "lun.name,lun.uuid,igroup.name," + "igroup.uuid,logical_unit_number"}) return [r for r in data.get("records", []) if r.get("lun", {}).get("name", "").startswith(prefix)] @@ -449,6 +700,8 @@ class OntapTestBase(cloudstackTestCase): svm_name = None cluster_hosts = None kvm_hosts_ssh_creds = [] # [{'host': '10.x.x.x', 'user': 'root', 'password': '...'}] + _host_iqn_cache = {} + igroup_baseline = {} ontap = None testdata = None zone = None @@ -768,6 +1021,255 @@ def _poll_pool_state(self, pool_id, target_state, timeout=120, interval=5): % (pool_id, target_state, timeout, current_state) ) + def _enter_maintenance(self, pool_id, timeout=120): + """Put a pool into Maintenance and wait for CloudStack to report it.""" + cmd = enableStorageMaintenance.enableStorageMaintenanceCmd() + cmd.id = pool_id + self.apiClient.enableStorageMaintenance(cmd) + return self._poll_pool_state(pool_id, "Maintenance", timeout=timeout) + + def _exit_maintenance(self, pool_id, timeout=120): + """Cancel Maintenance on a pool and wait for CloudStack to report Up.""" + cmd = cancelStorageMaintenance.cancelStorageMaintenanceCmd() + cmd.id = pool_id + self.apiClient.cancelStorageMaintenance(cmd) + return self._poll_pool_state(pool_id, "Up", timeout=timeout) + + def _exit_maintenance_quietly(self, pool, label): + """Bring a pool out of Maintenance without failing the test. + + Returns True when the pool is no longer in Maintenance, False when + it is not listed or the cancel did not take effect. + """ + try: + listed = list_storage_pools(self.apiClient, id=pool.id) + except CloudstackAPIException: + return False + if not listed: + return False + if listed[0].state != "Maintenance": + return True + try: + self._exit_maintenance(pool.id) + return True + except Exception as exc: + logger.warning("[%s] could not cancel maintenance on '%s': %s", + label, pool.name, exc) + return False + + @classmethod + def host_iqn(cls, host): + """The iSCSI initiator IQN for a cluster host, or None.""" + host_ip = getattr(host, "ipaddress", None) + if not host_ip: + return None + if host_ip in cls._host_iqn_cache: + return cls._host_iqn_cache[host_ip] + iqn = None + creds = next( + (c for c in cls.kvm_hosts_ssh_creds if c["host"] == host_ip), None + ) + if creds is not None: + try: + ssh = SshClient(host_ip, 22, creds["user"], creds["password"], + retries=3, delay=3, timeout=15.0) + out = ssh.execute( + "awk -F= '/^InitiatorName=/{print $2}' " + "/etc/iscsi/initiatorname.iscsi 2>/dev/null" + ) + for line in out or []: + line = line.strip() + if line.startswith("iqn."): + iqn = line + break + except Exception as ex: + logger.warning("host_iqn: SSH to %s failed: %s", host_ip, ex) + cls._host_iqn_cache[host_ip] = iqn + return iqn + + def _kvm_host_ip(self): + """Return one KVM host that this suite can reach over SSH.""" + if not self.kvm_hosts_ssh_creds: + self.skipTest("KVM SSH credentials are not configured") + return self.kvm_hosts_ssh_creds[0]["host"] + + def _kvm_run(self, host_ip, command): + """Run one command on a KVM host and return its output lines.""" + creds = next( + (c for c in self.kvm_hosts_ssh_creds if c["host"] == host_ip), + None + ) + if creds is None: + self.skipTest("No SSH credentials for KVM host %s" % host_ip) + ssh = SshClient( + host_ip, 22, creds["user"], creds["password"], + retries=3, delay=3, timeout=20.0 + ) + return ssh.execute(command) or [] + + @classmethod + def _igroup_name(cls, host_uuid): + """Return the igroup name used by OntapStorageUtils.""" + sanitized = re.sub(r"[^a-zA-Z0-9_-]", "_", str(host_uuid)) + return ("cs_%s_%s" % (sanitized, cls.svm_name))[:96] + + @classmethod + def _iscsi_host_specs(cls): + """Return (igroup name, initiator IQN) for iSCSI cluster hosts.""" + specs = [] + for host in cls.cluster_hosts or []: + iqn = ( + getattr(host, "storageurl", None) + or getattr(host, "StorageUrl", None) + or cls.host_iqn(host) + ) + host_uuid = getattr(host, "id", None) + if not iqn or not iqn.startswith("iqn.") or not host_uuid: + continue + specs.append((cls._igroup_name(host_uuid), iqn)) + return specs + + @staticmethod + def _igroup_initiators(igroup): + if igroup is None: + return None + return tuple(sorted( + i.get("name", "") for i in igroup.get("initiators", []) + )) + + @classmethod + def _capture_igroup_baseline(cls): + """Snapshot shared host igroups before an iSCSI suite creates a pool.""" + cls.igroup_baseline = {} + for igroup_name, _ in cls._iscsi_host_specs(): + igroup = cls.ontap.get_igroup(cls.svm_name, igroup_name) + cls.igroup_baseline[igroup_name] = cls._igroup_initiators(igroup) + logger.info( + "Captured iSCSI igroup baseline for SVM '%s': %s", + cls.svm_name, cls.igroup_baseline, + ) + + def _assert_igroup_baseline_unchanged(self, context): + """Assert an operation did not change pre-existing shared igroups.""" + for igroup_name, expected_initiators in self.igroup_baseline.items(): + igroup = self.ontap.get_igroup(self.svm_name, igroup_name) + actual_initiators = self._igroup_initiators(igroup) + self.assertEqual( + actual_initiators, expected_initiators, + "ONTAP igroup '%s' changed %s: expected initiators %s, got %s" + % (igroup_name, context, expected_initiators, + actual_initiators), + ) + + def _assert_no_lun_maps_for_volume(self, volume_name, context): + """Assert no LUN in a test FlexVol remains mapped to any igroup.""" + maps = self.ontap.list_lun_maps_for_volume( + self.svm_name, volume_name + ) + self.assertFalse( + maps, + "LUN maps for FlexVol '%s' remain %s: %s" + % (volume_name, context, maps), + ) + + def _get_cs_volume(self, vol_id): + """Return the CloudStack volume object, or None if it is gone.""" + from marvin.cloudstackAPI import listVolumes as listVolumesAPI + cmd = listVolumesAPI.listVolumesCmd() + cmd.id = vol_id + cmd.listall = True + vols = self.apiClient.listVolumes(cmd) or [] + return vols[0] if vols else None + + def _volume_exists_in_cs(self, vol_id): + """Return True if the volume is still listed by CloudStack.""" + return self._get_cs_volume(vol_id) is not None + + def _purge_cs_volume_record(self, vol, label): + """Remove a CS volume record the forced pool delete may have left. + + The backing storage is already gone at this point, so a failure here + only affects tidiness — the volume stays in the volume2 slot for + tearDownClass to retry and no exception is raised. + """ + if vol is None: + return + if self._volume_exists_in_cs(vol.id): + try: + cmd = deleteVolumeAPI.deleteVolumeCmd() + cmd.id = vol.id + self.apiClient.deleteVolume(cmd) + logger.info("[%s] deleted leftover CS volume record %s", + label, vol.id) + except Exception as exc: + logger.warning("[%s] could not delete leftover CS volume %s: %s", + label, vol.id, exc) + else: + logger.info("[%s] CS volume %s was removed along with the pool", + label, vol.id) + if not self._volume_exists_in_cs(vol.id): + self.__class__.volume2 = None + + def _remove_cs_volume(self, pool, vol, label): + """Clear the CloudStack volume so the pool can be force-deleted. + + deleteStoragePool(forced=True) refuses while any volume on the pool is + in a state other than Destroy. deleteVolume is tried first because it + also reclaims the backing storage, but it expunges through libvirt and + fails when the FlexVol is already gone. destroyVolume(expunge=False) + is the fallback: it only moves the record to Destroy, which is all the + forced pool delete requires - it expunges the leftovers itself. + + Returns True once the volume no longer blocks the forced pool delete. + """ + if vol is None or not self._volume_exists_in_cs(vol.id): + self.__class__.volume2 = None + return True + self._exit_maintenance_quietly(pool, label) + try: + cmd = deleteVolumeAPI.deleteVolumeCmd() + cmd.id = vol.id + self.apiClient.deleteVolume(cmd) + except Exception as exc: + logger.warning("[%s] deleteVolume failed for %s (%s); falling back " + "to destroyVolume without expunge", + label, vol.id, exc) + try: + cmd = destroyVolumeAPI.destroyVolumeCmd() + cmd.id = vol.id + cmd.expunge = False + self.apiClient.destroyVolume(cmd) + except Exception as destroy_exc: + logger.warning("[%s] destroyVolume also failed for %s: %s", + label, vol.id, destroy_exc) + remaining = self._get_cs_volume(vol.id) + if remaining is None: + self.__class__.volume2 = None + return True + state = getattr(remaining, "state", None) + return state is None or state.lower() in ( + "destroy", "destroyed", "expunging", "expunged" + ) + + def _other_ontap_pools_on_svm(self, current_pool_id): + """Return other CloudStack ONTAP pools that use this suite's SVM.""" + try: + pools = list_storage_pools(self.apiClient) or [] + except Exception: + return ["unable to list storage pools"] + others = [] + for pool in pools: + if str(getattr(pool, "id", "")) == str(current_pool_id): + continue + details = _parse_pool_details(pool) + if details.get("svmName") == getattr(self, "svm_name", None): + others.append(pool) + continue + provider = (getattr(pool, "provider", "") or "").lower() + if not details and "netapp" in provider and "ontap" in provider: + others.append(pool) + return others + def _poll_vm_state(self, vm_id, target_state, timeout=300, interval=10): """ Poll listVirtualMachines until the VM reaches target_state.