From 218ce1459185447f045f46a5637b2d82da9ce1c3 Mon Sep 17 00:00:00 2001 From: rahimahisah17 Date: Mon, 21 Sep 2026 19:16:05 +0100 Subject: [PATCH] test: add policy SQL tests for rightsizing, snapshots, container registries and app service plans Adds a shared run_policy fixture in tests/policies/conftest.py and unit tests for four policies: rightsizing_utilization (idle vs underutilized grading, metric averaging, the 5% and 20% CPU limits, cost summing, case-insensitive join), old_snapshots (the 30 day boundary and cost), idle_container_registries and idle_app_service_plans (pricing tiers). Refs #6. --- tests/policies/conftest.py | 37 +++++ tests/policies/test_idle_app_service_plans.py | 43 +++++ .../test_idle_container_registries.py | 49 ++++++ tests/policies/test_old_snapshots.py | 63 +++++++ .../policies/test_rightsizing_utilization.py | 157 ++++++++++++++++++ 5 files changed, 349 insertions(+) create mode 100644 tests/policies/conftest.py create mode 100644 tests/policies/test_idle_app_service_plans.py create mode 100644 tests/policies/test_idle_container_registries.py create mode 100644 tests/policies/test_old_snapshots.py create mode 100644 tests/policies/test_rightsizing_utilization.py diff --git a/tests/policies/conftest.py b/tests/policies/conftest.py new file mode 100644 index 0000000..e2aeb22 --- /dev/null +++ b/tests/policies/conftest.py @@ -0,0 +1,37 @@ +from pathlib import Path + +import duckdb +import pyarrow as pa +import pytest + +from cloudcost.config.loader import PolicyConfig +from cloudcost.policies.models import Finding +from cloudcost.policies.runner import PolicyRunner + +POLICIES_DIR = Path(__file__).resolve().parents[2] / "policies" + + +@pytest.fixture +def run_policy(tmp_path: Path): + """Run one policy from the policies folder against small in-memory tables and return its findings.""" + + def run(policy_name: str, tables: dict[str, pa.Table]) -> list[Finding]: + db_path = tmp_path / "costs.duckdb" + conn = duckdb.connect(str(db_path)) + try: + for table_name, table in tables.items(): + conn.register("staged", table) + conn.execute(f"CREATE TABLE {table_name} AS SELECT * FROM staged") + conn.unregister("staged") + finally: + conn.close() + + policy = PolicyConfig( + name=policy_name, + type="sql", + query_file=str(POLICIES_DIR / f"{policy_name}.sql"), + severity="medium", + ) + return PolicyRunner(str(db_path)).run_policy(policy) + + return run diff --git a/tests/policies/test_idle_app_service_plans.py b/tests/policies/test_idle_app_service_plans.py new file mode 100644 index 0000000..d90f444 --- /dev/null +++ b/tests/policies/test_idle_app_service_plans.py @@ -0,0 +1,43 @@ +import pyarrow as pa + + +def plans(*rows: tuple[str, str]) -> pa.Table: + """Each row is (resource_id, sku).""" + return pa.table( + { + "resource_id": pa.array([r[0] for r in rows], type=pa.string()), + "resource_name": pa.array([f"name-{r[0]}" for r in rows], type=pa.string()), + "resource_group": pa.array(["rg" for _ in rows], type=pa.string()), + "sku": pa.array([r[1] for r in rows], type=pa.string()), + } + ) + + +def test_free_tier_plan_is_reported_at_zero_dollars(run_policy) -> None: + findings = run_policy("idle_app_service_plans", {"fact_idle_app_service_plans": plans(("plan1", "F1"))}) + + assert len(findings) == 1 + finding = findings[0] + assert finding.provider == "azure" + assert finding.service_name == "App Service Plan" + assert finding.estimated_impact == 0.0 + assert finding.evidence["evidence_reason"] == "App Service Plan with 0 deployed apps (F1 tier)" + + +def test_paid_tier_plan_uses_the_documented_basic_price_estimate(run_policy) -> None: + findings = run_policy("idle_app_service_plans", {"fact_idle_app_service_plans": plans(("plan1", "B1"))}) + + assert findings[0].estimated_impact == 13.14 + assert findings[0].evidence["evidence_reason"] == "App Service Plan with 0 deployed apps (B1 tier)" + + +def test_every_plan_row_becomes_a_finding(run_policy) -> None: + table = plans(("a", "F1"), ("b", "B1"), ("c", "S1")) + + findings = run_policy("idle_app_service_plans", {"fact_idle_app_service_plans": table}) + + assert sorted(f.resource_id for f in findings) == ["a", "b", "c"] + + +def test_empty_table_returns_no_findings(run_policy) -> None: + assert run_policy("idle_app_service_plans", {"fact_idle_app_service_plans": plans()}) == [] diff --git a/tests/policies/test_idle_container_registries.py b/tests/policies/test_idle_container_registries.py new file mode 100644 index 0000000..d1da21c --- /dev/null +++ b/tests/policies/test_idle_container_registries.py @@ -0,0 +1,49 @@ +import pyarrow as pa + + +def registries(*rows: tuple[str, str]) -> pa.Table: + """Each row is (resource_id, sku).""" + return pa.table( + { + "resource_id": pa.array([r[0] for r in rows], type=pa.string()), + "resource_name": pa.array([f"name-{r[0]}" for r in rows], type=pa.string()), + "resource_group": pa.array(["rg" for _ in rows], type=pa.string()), + "sku": pa.array([r[1] for r in rows], type=pa.string()), + } + ) + + +def test_standard_registry_is_priced_at_the_daily_rate_times_30(run_policy) -> None: + findings = run_policy( + "idle_container_registries", {"fact_idle_container_registries": registries(("reg1", "Standard"))} + ) + + assert len(findings) == 1 + finding = findings[0] + assert finding.provider == "azure" + assert finding.service_name == "Container Registry" + assert finding.estimated_impact == 20.0 + assert finding.evidence["evidence_reason"] == "idle Container Registry, 0 repositories (Standard tier)" + + +def test_premium_registry_is_priced_higher_than_standard(run_policy) -> None: + findings = run_policy( + "idle_container_registries", {"fact_idle_container_registries": registries(("reg1", "Premium"))} + ) + + assert findings[0].estimated_impact == 50.0 + assert findings[0].evidence["evidence_reason"] == "idle Container Registry, 0 repositories (Premium tier)" + + +def test_every_registry_row_becomes_a_finding(run_policy) -> None: + table = registries(("a", "Standard"), ("b", "Premium"), ("c", "Standard")) + + findings = run_policy("idle_container_registries", {"fact_idle_container_registries": table}) + + assert sorted(f.resource_id for f in findings) == ["a", "b", "c"] + + +def test_empty_table_returns_no_findings(run_policy) -> None: + table = registries() + + assert run_policy("idle_container_registries", {"fact_idle_container_registries": table}) == [] diff --git a/tests/policies/test_old_snapshots.py b/tests/policies/test_old_snapshots.py new file mode 100644 index 0000000..ff503b2 --- /dev/null +++ b/tests/policies/test_old_snapshots.py @@ -0,0 +1,63 @@ +import pyarrow as pa + + +def snapshots(*rows: tuple[str, int, int]) -> pa.Table: + """Each row is (resource_id, size_gb, age_days).""" + return pa.table( + { + "resource_id": pa.array([r[0] for r in rows], type=pa.string()), + "resource_name": pa.array([f"name-{r[0]}" for r in rows], type=pa.string()), + "resource_group": pa.array(["rg" for _ in rows], type=pa.string()), + "size_gb": pa.array([r[1] for r in rows], type=pa.int64()), + "sku": pa.array(["Standard_LRS" for _ in rows], type=pa.string()), + "age_days": pa.array([r[2] for r in rows], type=pa.int64()), + "location": pa.array(["eastus" for _ in rows], type=pa.string()), + } + ) + + +def test_flags_a_snapshot_older_than_30_days(run_policy) -> None: + findings = run_policy("old_snapshots", {"fact_old_snapshots": snapshots(("snap1", 100, 45))}) + + assert len(findings) == 1 + finding = findings[0] + assert finding.provider == "azure" + assert finding.service_name == "Snapshot" + assert finding.resource_id == "snap1" + assert finding.evidence["evidence_reason"] == "snapshot is 45 days old (100GB Standard_LRS)" + + +def test_a_snapshot_exactly_30_days_old_is_not_flagged(run_policy) -> None: + findings = run_policy("old_snapshots", {"fact_old_snapshots": snapshots(("snap1", 100, 30))}) + + assert findings == [] + + +def test_a_snapshot_31_days_old_is_flagged(run_policy) -> None: + findings = run_policy("old_snapshots", {"fact_old_snapshots": snapshots(("snap1", 100, 31))}) + + assert [f.resource_id for f in findings] == ["snap1"] + + +def test_monthly_cost_is_size_times_five_cents(run_policy) -> None: + findings = run_policy("old_snapshots", {"fact_old_snapshots": snapshots(("snap1", 100, 45))}) + + assert findings[0].estimated_impact == 5.0 + + +def test_monthly_cost_scales_with_snapshot_size(run_policy) -> None: + findings = run_policy("old_snapshots", {"fact_old_snapshots": snapshots(("snap1", 33, 45))}) + + assert findings[0].estimated_impact == 1.65 + + +def test_only_old_snapshots_are_returned_from_a_mixed_table(run_policy) -> None: + table = snapshots(("old", 10, 90), ("recent", 10, 3), ("older", 10, 400)) + + findings = run_policy("old_snapshots", {"fact_old_snapshots": table}) + + assert sorted(f.resource_id for f in findings) == ["old", "older"] + + +def test_empty_table_returns_no_findings(run_policy) -> None: + assert run_policy("old_snapshots", {"fact_old_snapshots": snapshots()}) == [] diff --git a/tests/policies/test_rightsizing_utilization.py b/tests/policies/test_rightsizing_utilization.py new file mode 100644 index 0000000..dd346c8 --- /dev/null +++ b/tests/policies/test_rightsizing_utilization.py @@ -0,0 +1,157 @@ +import pyarrow as pa +import pytest + +IDLE = "idle (high confidence)" +UNDERUTILIZED = "underutilized" + + +def cost(resource_id: str = "vm1", service: str = "Virtual Machines", billed: float = 30.0) -> dict: + return {"resource_id": resource_id, "service_name": service, "billed_cost": billed} + + +def metric( + resource_id: str = "vm1", + cpu: float = 2.0, + net_in: float = 0.0, + net_out: float = 0.0, + disk_read: float = 0.0, + disk_write: float = 0.0, +) -> dict: + return { + "resource_id": resource_id, + "avg_cpu_percent": cpu, + "avg_network_in_bytes": net_in, + "avg_network_out_bytes": net_out, + "avg_disk_read_ops": disk_read, + "avg_disk_write_ops": disk_write, + } + + +def costs(*rows: dict) -> pa.Table: + return pa.table( + { + "provider": pa.array(["azure" for _ in rows], type=pa.string()), + "resource_id": pa.array([r["resource_id"] for r in rows], type=pa.string()), + "service_name": pa.array([r["service_name"] for r in rows], type=pa.string()), + "billed_cost": pa.array([r["billed_cost"] for r in rows], type=pa.float64()), + } + ) + + +def metrics(*rows: dict) -> pa.Table: + return pa.table( + { + "resource_id": pa.array([r["resource_id"] for r in rows], type=pa.string()), + "avg_cpu_percent": pa.array([r["avg_cpu_percent"] for r in rows], type=pa.float64()), + "avg_network_in_bytes": pa.array([r["avg_network_in_bytes"] for r in rows], type=pa.float64()), + "avg_network_out_bytes": pa.array([r["avg_network_out_bytes"] for r in rows], type=pa.float64()), + "avg_disk_read_ops": pa.array([r["avg_disk_read_ops"] for r in rows], type=pa.float64()), + "avg_disk_write_ops": pa.array([r["avg_disk_write_ops"] for r in rows], type=pa.float64()), + } + ) + + +def rightsizing(run_policy, cost_rows: pa.Table, metric_rows: pa.Table): + return run_policy("rightsizing_utilization", {"fact_cost": cost_rows, "fact_metrics": metric_rows}) + + +def test_vm_with_no_activity_on_any_axis_is_idle_with_high_confidence(run_policy) -> None: + findings = rightsizing( + run_policy, + costs(cost(billed=30.0)), + metrics(metric(cpu=2.0, net_in=100.0, net_out=100.0, disk_read=1.0, disk_write=1.0)), + ) + + assert len(findings) == 1 + finding = findings[0] + assert finding.provider == "azure" + assert finding.service_name == "Virtual Machines" + assert finding.estimated_impact == 30.0 + assert finding.evidence["confidence"] == IDLE + assert finding.evidence["evidence_reason"].startswith("CPU 2.0%") + + +@pytest.mark.parametrize( + "busy_axis", + [ + {"net_in": 1_000_000.0}, + {"net_out": 1_000_000.0}, + {"disk_read": 50.0}, + {"disk_write": 50.0}, + {"net_in": 50_000.0}, + {"net_out": 50_000.0}, + {"disk_read": 5.0}, + {"disk_write": 5.0}, + ], +) +def test_low_cpu_with_activity_on_any_other_axis_is_only_underutilized(run_policy, busy_axis) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(cpu=2.0, **busy_axis))) + + assert findings[0].evidence["confidence"].startswith(UNDERUTILIZED) + + +def test_cpu_of_5_percent_is_no_longer_idle(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(cpu=5.0))) + + assert findings[0].evidence["confidence"].startswith(UNDERUTILIZED) + + +def test_moderate_cpu_is_underutilized_even_with_no_other_traffic(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(cpu=12.0))) + + assert findings[0].evidence["confidence"].startswith(UNDERUTILIZED) + + +def test_cpu_at_20_percent_is_not_reported(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(cpu=20.0))) + + assert findings == [] + + +def test_cpu_just_under_20_percent_is_reported(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(cpu=19.9))) + + assert [f.resource_id for f in findings] == ["vm1"] + + +def test_metrics_are_averaged_across_rows_before_grading(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(cpu=2.0), metric(cpu=8.0))) + + finding = findings[0] + assert finding.evidence["evidence_reason"].startswith("CPU 5.0%") + assert finding.evidence["confidence"].startswith(UNDERUTILIZED) + + +def test_costs_are_summed_per_vm(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost(billed=10.0), cost(billed=5.5)), metrics(metric())) + + assert len(findings) == 1 + assert findings[0].estimated_impact == 15.5 + + +def test_resource_ids_are_matched_ignoring_case(run_policy) -> None: + findings = rightsizing( + run_policy, + costs(cost(resource_id="/SUB/RG/VM1")), + metrics(metric(resource_id="/sub/rg/vm1")), + ) + + assert [f.resource_id for f in findings] == ["/SUB/RG/VM1"] + + +def test_services_other_than_virtual_machines_are_ignored(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost(service="Storage")), metrics(metric())) + + assert findings == [] + + +def test_vm_with_zero_cost_is_ignored(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost(billed=0.0)), metrics(metric())) + + assert findings == [] + + +def test_vm_without_metrics_is_not_reported(run_policy) -> None: + findings = rightsizing(run_policy, costs(cost()), metrics(metric(resource_id="other"))) + + assert findings == []