"""Machine-readable acceptance evidence derived from repository state.""" from __future__ import annotations import hashlib import json import re from dataclasses import asdict, dataclass from pathlib import Path from gyxx_flow.catalog import CatalogError, WorkflowCatalog from gyxx_flow.core.config import Settings from gyxx_flow.script_catalog import ScriptCatalog, ScriptCatalogError from gyxx_flow.security import scan_repository _CHECKLIST = re.compile(r"^\s*-\s*\[([ xX])\]\s+(P\d+\.\d+)\b", re.MULTILINE) @dataclass(frozen=True, slots=True) class PlanItem: item_id: str completed: bool @dataclass(frozen=True, slots=True) class AcceptanceReport: items: tuple[PlanItem, ...] checks: dict[str, bool] @property def is_complete(self) -> bool: return bool(self.items) and all(item.completed for item in self.items) and all( self.checks.values() ) def as_dict(self) -> dict[str, object]: completed = sum(item.completed for item in self.items) return { "schema_version": 1, "is_complete": self.is_complete, "summary": { "total": len(self.items), "completed": completed, "pending": len(self.items) - completed, }, "items": [asdict(item) for item in self.items], "checks": dict(sorted(self.checks.items())), } def parse_plan_checklist(path: Path) -> tuple[PlanItem, ...]: text = Path(path).read_text(encoding="utf-8") items = tuple( PlanItem(item_id=match.group(2), completed=match.group(1).casefold() == "x") for match in _CHECKLIST.finditer(text) ) if len({item.item_id for item in items}) != len(items): raise ValueError("plan contains duplicate acceptance item IDs") return items def build_acceptance_report(settings: Settings) -> AcceptanceReport: project_root = settings.project_root items = parse_plan_checklist(project_root / "docs" / "plan.md") checks = { "catalog_22_tasks": _catalog_has_22_tasks(project_root), "scheduled_graphs_explicit": _scheduled_graphs_are_explicit(project_root), "python_scheduler_only": _python_scheduler_is_the_only_scheduler(project_root), "runtime_service_policy": _runtime_service_policy_is_configured(project_root), "native_entrypoints_local": _native_entrypoints_are_local(project_root), "runtime_sources_decoupled": _runtime_sources_are_decoupled(project_root), "public_command_registry_complete": _public_command_registry_is_complete( project_root ), "source_manifests_verified": _source_manifests_are_verified(project_root), "secret_scan_clean": not scan_repository(project_root), } return AcceptanceReport(items, checks) def _catalog_has_22_tasks(project_root: Path) -> bool: try: catalog = WorkflowCatalog.load(project_root / "config") except Exception: return False scheduled = catalog.scheduled_workflows() scheduled_ids = {workflow.workflow_id for workflow in scheduled} schedule_ids = {schedule.workflow_id for schedule in catalog.schedules} return bool(scheduled) and scheduled_ids == schedule_ids def _scheduled_graphs_are_explicit(project_root: Path) -> bool: try: catalog = WorkflowCatalog.load(project_root / "config") except Exception: return False return all(workflow.steps for workflow in catalog.scheduled_workflows()) def _python_scheduler_is_the_only_scheduler(project_root: Path) -> bool: scheduler_service = project_root / "src" / "gyxx_flow" / "scheduler_service.py" retired_windows_scheduler = project_root / "src" / "gyxx_flow" / "scheduler.py" systemd_unit = project_root / "deploy" / "gyxx-flow.service" legacy_installer = project_root / "deploy" / "windows-service" / "install.ps1" try: unit = systemd_unit.read_text(encoding="utf-8").casefold() except OSError: return False legacy_is_safe = True if legacy_installer.exists(): try: installer = legacy_installer.read_text(encoding="utf-8").casefold() except OSError: return False legacy_is_safe = ( "schedule run" in installer and "schtasks" not in installer and "new-scheduledtask" not in installer ) return ( scheduler_service.is_file() and not retired_windows_scheduler.exists() and "execstart=/opt/gyxx-flow/.venv/bin/python -m gyxx_flow schedule run" in unit and "environment=gyxx_data_root=/var/lib/gyxx-flow" in unit and "oncalendar=" not in unit and legacy_is_safe ) def _runtime_service_policy_is_configured(project_root: Path) -> bool: try: payload = json.loads( (project_root / "config" / "runtime-bindings.json").read_text( encoding="utf-8" ) ) services = payload["services"] except (OSError, json.JSONDecodeError, KeyError, TypeError): return False local_urls = ( services.get("hermes_url", ""), services.get("hermes_collector_url", ""), services.get("hermes_analyzer_gateway_url", ""), services.get("hermes_collector_gateway_url", ""), ) return ( services.get("postgres") == "cloud" and not services.get("postgres_host") and not services.get("postgres_database") and not services.get("postgres_user") and services.get("hermes") == "local" and all(url.startswith("http://127.0.0.1:") for url in local_urls) ) def _native_entrypoints_are_local(project_root: Path) -> bool: try: catalog = WorkflowCatalog.load(project_root / "config") for workflow in catalog.workflows: if workflow.trigger == "unavailable": continue target = ( project_root / "src" / "gyxx_flow" / "modules" / workflow.module / workflow.entry ).resolve(strict=True) if not target.is_file() or not target.is_relative_to(project_root): return False config_text = (project_root / "config" / "workflows.json").read_text( encoding="utf-8" ) return "legacy" not in config_text.casefold() except (OSError, ValueError, CatalogError): return False def _public_command_registry_is_complete(project_root: Path) -> bool: try: workflows = WorkflowCatalog.load(project_root / "config").workflows commands = ScriptCatalog.discover_default() except (OSError, ValueError, ScriptCatalogError): return False runnable = tuple(item for item in workflows if item.trigger != "unavailable") expected_modules = {item.module for item in runnable} scripts = commands.scripts if ( not scripts or len(commands.command_ids) != len(set(commands.command_ids)) or {item.module for item in scripts} != expected_modules ): return False resolved_root = project_root.resolve() for script in scripts: if not script.path.is_file() or not script.path.is_relative_to(resolved_root): return False try: for workflow in runnable: entries = ( tuple(step.entry for step in workflow.steps) if workflow.steps else (workflow.entry,) ) for entry in entries: command = commands.get(f"{workflow.module}:{entry}") if command.module != workflow.module or command.entry != entry: return False except ScriptCatalogError: return False return True def _runtime_sources_are_decoupled(project_root: Path) -> bool: module_root = project_root / "src" / "gyxx_flow" / "modules" forbidden = ( "d:\\yingxiaoyunying", "d:\\shop-data-flow", "d:\\product-collector-analyze-flow", "e:\\auto-flow", "gyxx_legacy_", "deferredlegacycommandstep", ) try: for path in module_root.rglob("*"): if not path.is_file() or path.suffix.casefold() in {".pyc", ".pyo"}: continue if path.is_symlink() or not path.resolve().is_relative_to(project_root): return False content = path.read_bytes() if b"\x00" in content: continue text: str | None = None for encoding in ("utf-8-sig", "gb18030"): try: text = content.decode(encoding) break except UnicodeDecodeError: continue if text is None: continue normalized = text.casefold().replace("/", "\\") while "\\\\" in normalized: normalized = normalized.replace("\\\\", "\\") if any(value in normalized for value in forbidden): return False except OSError: return False return True def _source_manifests_are_verified(project_root: Path) -> bool: modules = ( "content_marketing", "product_commerce", "shop_intelligence", "supply_chain", ) try: for module in modules: payload = json.loads( ( project_root / "config" / "source-manifests" / f"{module}.json" ).read_text(encoding="utf-8-sig") ) if payload.get("schema_version") != 1 or payload.get("module") != module: return False files = payload.get("files") if not isinstance(files, list) or not files: return False for item in files: target = project_root.joinpath(*Path(item["target_relative_path"]).parts) resolved = target.resolve(strict=True) if ( target.is_symlink() or not resolved.is_relative_to(project_root) or not resolved.is_file() or _sha256(resolved) != item["target_sha256"] ): return False except (OSError, json.JSONDecodeError, KeyError, TypeError, ValueError): return False return True def _sha256(path: Path) -> str: digest = hashlib.sha256() with path.open("rb") as stream: for chunk in iter(lambda: stream.read(1024 * 1024), b""): digest.update(chunk) return digest.hexdigest()