Files
gyxx-flow/tests/test_shadow_comparison.py
T

271 lines
8.9 KiB
Python

from __future__ import annotations
import csv
import json
from datetime import datetime, timezone
from pathlib import Path
import pytest
from gyxx_flow.migration.shadow import (
ComparisonSpec,
MetricTolerance,
compare_structured_files,
write_comparison_report,
)
FIXED_TIME = datetime(2026, 7, 27, 8, 30, tzinfo=timezone.utc)
def _write_jsonl(path: Path, rows: list[dict[str, object]]) -> None:
path.write_text(
"".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows),
encoding="utf-8",
)
def _write_csv(path: Path, rows: list[dict[str, object]]) -> None:
with path.open("w", encoding="utf-8", newline="") as handle:
writer = csv.DictWriter(handle, fieldnames=list(rows[0]))
writer.writeheader()
writer.writerows(rows)
def test_compare_jsonl_to_csv_reports_key_metric_and_error_differences_without_rows(
tmp_path: Path,
) -> None:
baseline = tmp_path / "baseline.jsonl"
candidate = tmp_path / "candidate.csv"
_write_jsonl(
baseline,
[
{
"shop_id": "shop-secret-1",
"sku": "sku-secret-a",
"category": "tops",
"revenue": 100.0,
"errors": ["upstream token exposed"],
"customer_email": "private@example.test",
},
{
"shop_id": "shop-secret-1",
"sku": "sku-secret-b",
"category": "shoes",
"revenue": 20.0,
"errors": [],
"customer_email": "hidden@example.test",
},
],
)
_write_csv(
candidate,
[
{
"shop_id": "shop-secret-1",
"sku": "sku-secret-a",
"category": "tops",
"revenue": "101.5",
"errors": "replacement failure",
"customer_email": "private@example.test",
},
{
"shop_id": "shop-secret-2",
"sku": "sku-secret-c",
"category": "accessories",
"revenue": "9.0",
"errors": "",
"customer_email": "do-not-report@example.test",
},
],
)
spec = ComparisonSpec(
primary_key=("shop_id", "sku"),
metrics={"revenue": MetricTolerance(absolute=1.0, relative=0.0)},
error_fields=("errors",),
safe_fields=("category",),
hash_secret="test-only-hmac-key",
)
report = compare_structured_files(
baseline, candidate, spec=spec, generated_at=FIXED_TIME
)
payload = report.to_dict()
assert payload["schema_version"] == 1
assert payload["status"] == "mismatch"
assert payload["generated_at"] == "2026-07-27T08:30:00+00:00"
assert payload["sources"] == {
"baseline_format": "jsonl",
"candidate_format": "csv",
}
assert payload["summary"] == {
"baseline_rows": 2,
"candidate_rows": 2,
"common_keys": 1,
"missing_keys": 1,
"extra_keys": 1,
"metric_mismatches": 1,
"error_set_differences": 2,
}
assert payload["keys"]["missing"][0]["safe_fields"] == {"category": "shoes"}
assert payload["keys"]["extra"][0]["safe_fields"] == {
"category": "accessories"
}
assert len(payload["keys"]["missing"][0]["key_hash"]) == 64
assert payload["metrics"]["revenue"]["mismatch_count"] == 1
assert payload["metrics"]["revenue"]["max_absolute_delta"] == 1.5
assert len(payload["errors"]["only_baseline_hashes"][0]) == 64
assert len(payload["errors"]["only_candidate_hashes"][0]) == 64
rendered = json.dumps(payload, ensure_ascii=False, sort_keys=True)
for sensitive in (
"shop-secret-1",
"shop-secret-2",
"sku-secret-a",
"sku-secret-b",
"sku-secret-c",
"upstream token exposed",
"replacement failure",
"private@example.test",
"hidden@example.test",
"do-not-report@example.test",
"test-only-hmac-key",
):
assert sensitive not in rendered
def test_metric_values_within_absolute_or_relative_tolerance_match(tmp_path: Path) -> None:
baseline = tmp_path / "baseline.json"
candidate = tmp_path / "candidate.jsonl"
baseline.write_text(
json.dumps(
{
"records": [
{"id": "one", "amount": 100.0},
{"id": "two", "amount": 0.0},
]
}
),
encoding="utf-8",
)
_write_jsonl(
candidate,
[
{"id": "one", "amount": 100.5},
{"id": "two", "amount": 0.01},
],
)
spec = ComparisonSpec(
primary_key=("id",),
metrics={
"amount": MetricTolerance(absolute=0.01, relative=0.01),
},
)
payload = compare_structured_files(baseline, candidate, spec=spec).to_dict()
assert payload["status"] == "match"
assert payload["metrics"]["amount"]["mismatch_count"] == 0
assert payload["summary"]["metric_mismatches"] == 0
def test_comparison_rejects_duplicate_or_missing_primary_keys(tmp_path: Path) -> None:
baseline = tmp_path / "baseline.jsonl"
candidate = tmp_path / "candidate.jsonl"
_write_jsonl(baseline, [{"id": "same"}, {"id": "same"}])
_write_jsonl(candidate, [{"id": "same"}])
with pytest.raises(ValueError, match="duplicate primary key"):
compare_structured_files(
baseline, candidate, spec=ComparisonSpec(primary_key=("id",))
)
_write_jsonl(baseline, [{"name": "missing"}])
with pytest.raises(ValueError, match="missing primary key field"):
compare_structured_files(
baseline, candidate, spec=ComparisonSpec(primary_key=("id",))
)
def test_comparison_rejects_unsupported_or_malformed_inputs(tmp_path: Path) -> None:
bad_extension = tmp_path / "rows.txt"
candidate = tmp_path / "candidate.jsonl"
bad_extension.write_text("{}\n", encoding="utf-8")
_write_jsonl(candidate, [{"id": "1", "amount": 1}])
spec = ComparisonSpec(
primary_key=("id",), metrics={"amount": MetricTolerance()}
)
with pytest.raises(ValueError, match="unsupported structured data format"):
compare_structured_files(bad_extension, candidate, spec=spec)
malformed = tmp_path / "malformed.jsonl"
malformed.write_text('[1, 2, 3]\n', encoding="utf-8")
with pytest.raises(ValueError, match="record must be a JSON object"):
compare_structured_files(malformed, candidate, spec=spec)
too_large = tmp_path / "too-large.jsonl"
_write_jsonl(too_large, [{"id": "1", "amount": "1e9999"}])
with pytest.raises(ValueError, match="exceeds report range"):
compare_structured_files(too_large, candidate, spec=spec)
def test_difference_samples_are_deterministic_and_bounded(tmp_path: Path) -> None:
baseline = tmp_path / "baseline.jsonl"
candidate = tmp_path / "candidate.jsonl"
_write_jsonl(baseline, [{"id": f"old-{index}"} for index in range(5)])
_write_jsonl(candidate, [{"id": f"new-{index}"} for index in range(5)])
spec = ComparisonSpec(primary_key=("id",), difference_limit=2)
payload = compare_structured_files(baseline, candidate, spec=spec).to_dict()
assert payload["summary"]["missing_keys"] == 5
assert payload["keys"]["missing_total"] == 5
assert payload["keys"]["missing_truncated"] is True
assert len(payload["keys"]["missing"]) == 2
assert payload["keys"]["missing"] == sorted(
payload["keys"]["missing"], key=lambda item: item["key_hash"]
)
def test_write_comparison_report_is_atomic_and_machine_readable(tmp_path: Path) -> None:
baseline = tmp_path / "baseline.jsonl"
candidate = tmp_path / "candidate.jsonl"
destination = tmp_path / "reports" / "comparison.json"
_write_jsonl(baseline, [{"id": "same"}])
_write_jsonl(candidate, [{"id": "same"}])
report = compare_structured_files(
baseline,
candidate,
spec=ComparisonSpec(primary_key=("id",)),
generated_at=FIXED_TIME,
)
written = write_comparison_report(destination, report)
assert written == destination
assert json.loads(destination.read_text(encoding="utf-8")) == report.to_dict()
assert not list(destination.parent.glob(".*.tmp"))
@pytest.mark.parametrize(
"kwargs, message",
[
({"primary_key": ()}, "primary_key"),
({"primary_key": ("id", "id")}, "primary_key"),
({"primary_key": ("id",), "difference_limit": 0}, "difference_limit"),
],
)
def test_comparison_spec_validates_declarations(
kwargs: dict[str, object], message: str
) -> None:
with pytest.raises(ValueError, match=message):
ComparisonSpec(**kwargs) # type: ignore[arg-type]
def test_metric_tolerance_rejects_negative_or_non_finite_values() -> None:
with pytest.raises(ValueError, match="absolute"):
MetricTolerance(absolute=-0.1)
with pytest.raises(ValueError, match="relative"):
MetricTolerance(relative=float("inf"))