Files
awoooi/docs/operations/sre-k3s-controlled-automation-work-items.snapshot.json
2026-07-19 06:32:28 +08:00

1023 lines
60 KiB
JSON
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"schema_version": "sre_k3s_controlled_automation_work_items_v1",
"governance_version": "global_product_governance_v2",
"program_id": "AIA-SRE-P0-20260715",
"generated_at": "2026-07-19T05:07:28+08:00",
"status": "phase_6_release_blocked_cd_5384_failed_runtime_unchanged",
"scope_complete": false,
"current_p0": {
"id": "AIA-SRE-013",
"title": "Claude/Gemini protected-secret 受控上線與五路 SRE 效果驗證",
"phase": "phase_3",
"terminal_condition": "同一 source SHA 通過 focused/full tests 與 Gitea CDdurable Gate5 authorization receipt 綁定同一 runprotected-secret metadata、GCP-A/GCP-B/host111/Claude/Gemini 同 fixture、Claude/Gemini finalized receipt、latency/token/cost、paired atomic rollback 與 production status readback 全部可驗證;不代表尚未完成的 HolmesGPT 主 pipeline promotion"
},
"architecture": {
"pipeline": [
"Alert",
"Canonical Asset Normalize",
"Typed Domain Router",
"HolmesGPT Investigator",
"Ollama RCA / Gemini Critic",
"Deterministic Policy",
"Single Controlled Executor",
"Independent Verifier",
"Incident Closure + KM/RAG/MCP/PlayBook"
],
"cross_domain_fallback_allowed": false,
"unknown_asset_terminal": "asset_identity_unresolved",
"circuit_open_terminal": "stay_in_same_domain_and_create_repair_work_item",
"same_run_identity_required": [
"trace_id",
"run_id",
"work_item_id"
],
"provider_failover_is_not_a_pipeline_stage": true
},
"provider_policy": {
"ordered_route": [
"ollama_gcp_a",
"ollama_gcp_b",
"ollama_local",
"claude",
"gemini"
],
"route_label": "GCP-A -> GCP-B -> host111 Ollama -> Anthropic Claude -> Gemini API",
"hops": [
{
"position": 1,
"provider": "ollama_gcp_a",
"identity": "GCP-A"
},
{
"position": 2,
"provider": "ollama_gcp_b",
"identity": "GCP-B"
},
{
"position": 3,
"provider": "ollama_local",
"identity": "host111"
},
{
"position": 4,
"provider": "claude",
"identity": "Anthropic Claude API"
},
{
"position": 5,
"provider": "gemini",
"identity": "Gemini API"
}
],
"ollama_role": "primary_rca_and_action_planning",
"claude_role": "paid_architecture_rca_and_debug_fallback",
"gemini_role": "final_fallback_and_optional_critic_only",
"claude_paid_call_allowed": true,
"gemini_paid_call_allowed": true,
"paid_execution_mode": "canary",
"paid_canary_percent": 5,
"paid_canary_authorization_ref": "AIA-SRE-013-user-approved-20260716",
"paid_canary_authorization_status": "user_direction_and_durable_gate5_consumed_first_canary_terminal_failed",
"durable_gate5_authorization": {
"required": true,
"binding": [
"trace_id",
"run_id",
"work_item_id",
"provider_pair",
"cost_caps",
"expiry"
],
"accepted_source": "durable_operation_log_or_authorization_receipt",
"free_form_authorization_ref_is_sufficient": false,
"status": "consumed_terminal_failed",
"authorization_ref": "019f6c09-e77c-7800-35af-9179358c8f37",
"run_id": "019f6c09-e77c-7800-35af-9179358c8f37",
"trace_id": "00-4a83cb2dea7749b398a5e6d6c90737df6ed6c3b8e9534e8b-89789ec6474d4a52-01",
"approval_durable": true,
"single_use_claim_consumed": true,
"terminal_state": "failed",
"terminal_error_code": "E-PAID-CANARY-VERIFY"
},
"paid_canary_scope": "one committed sanitized fixture per provider; no infrastructure remediation",
"claude_cost_cap_status": "generation_finalized_and_aggregate_run_cap_verified",
"gemini_cost_cap_status": "failed_generation_estimate_finalized_and_aggregate_run_cap_verified",
"claude_credential_status": "runtime_secret_reference_generation_authenticated_no_secret_value_read",
"gemini_credential_status": "runtime_secret_reference_present_but_provider_authentication_rejected_no_secret_value_read",
"credential_metadata": {
"claude": {
"user_supplied_credential_noted": true,
"raw_value_recorded_in_repository": false,
"raw_value_readback_allowed": false,
"protected_secret_reference_status": "k8s_secret_ref_present_and_generation_authenticated"
},
"gemini": {
"user_supplied_credential_noted": true,
"raw_value_recorded_in_repository": false,
"raw_value_readback_allowed": false,
"protected_secret_reference_status": "k8s_secret_ref_present_but_generation_authentication_rejected"
}
},
"cloud_transport_contract": {
"ollama_gcp_a": {
"observed_transport": "public_http",
"execution_scope": "sanitized_candidate_only",
"tool_loop_terminal": "fail_closed",
"pre_generation_receipt": "blocked_no_transport_receipt_tcp_timeout",
"exact_prometheus_target_runtime_status": "missing"
},
"ollama_gcp_b": {
"observed_transport": "public_http",
"execution_scope": "sanitized_candidate_only",
"tool_loop_terminal": "fail_closed",
"pre_generation_receipt": "transport_receipt_present_but_not_same_run_and_public_http_degraded",
"exact_prometheus_target_runtime_status": "missing"
},
"production_sensitive_payload_allowed": false,
"secure_mesh_or_tls_required_for_promotion": true
},
"paid_pair_contract": {
"enable_and_rollback_as_pair": true,
"partial_five_lane_terminal": "rollback_both_paid_providers_and_fail_canary",
"daily_monthly_and_per_incident_cost_caps_required": true,
"source_status": "same_run_gate5_atomic_single_use_claim_and_aggregate_cap_ready",
"runtime_status": "first_canary_partial_failed_paired_rollback_verified"
},
"latest_canary_receipt": {
"run_id": "019f6c09-e77c-7800-35af-9179358c8f37",
"status": "failed_rollback_verified",
"provider_results": {
"ollama_gcp_a": "transport_unreachable_no_generation",
"ollama_gcp_b": "provider_canary_timeout_after_120_seconds",
"ollama_local": "http_connection_reset_no_generation",
"claude": "generation_succeeded_contract_verifier_failed",
"gemini": "authentication_rejected"
},
"claude_latency_ms": 4879.9,
"claude_tokens": 1886,
"claude_cost_usd": 0.011802,
"aggregate_accounted_cost_usd": 0.012099,
"aggregate_cost_cap_usd": 0.25,
"paired_rollback_verified": true,
"persistent_paid_routes_enabled": false,
"raw_prompt_persisted": false,
"raw_response_persisted": false,
"secret_value_read_or_returned": false,
"learning_writeback_status": "pending_provider_scorecard_km_rag_mcp_playbook_ack"
},
"production_provider_route_switch_allowed": false,
"required_before_paid_provider_enablement": [
"offline_replay_scorecard",
"shadow_comparison",
"bounded_canary",
"daily_and_monthly_cost_cap",
"per_incident_token_cap",
"rate_limit_and_circuit_breaker",
"rollback_to_ollama_chain",
"production_cost_readback"
]
},
"agent99_host_operations_bridge": {
"role": "on_prem_policy_controlled_ai_execution_node",
"single_executor_domains": [
"windows_vmware",
"control_plane_recovery"
],
"relay_only_domains": [
"host_systemd",
"docker_container"
],
"dispatch_identity_schema": "agent99_controlled_dispatch_identity_v1",
"command_envelope_schema": "agent99_controlled_dispatch_input_v1",
"dispatch_receipt_schema": "agent99_controlled_dispatch_receipt_v1",
"outcome_schema": "agent99_outcome_contract_v1",
"completion_callback": "/api/v1/agents/agent99/completion-callback",
"arbitrary_command_allowed": false,
"unknown_asset_dispatch_allowed": false,
"cross_domain_fallback_allowed": false,
"runtime_terminal": "callback_plus_independent_verifier_plus_learning_writeback"
},
"domain_routes": [
{
"domain": "kubernetes_workload",
"executor": "kubernetes_controlled_executor",
"verifier": "kubernetes_rollout_verifier",
"fallback": "forbidden"
},
{
"domain": "host_systemd",
"executor": "host_ansible_executor",
"verifier": "host_runtime_independent_verifier",
"fallback": "forbidden"
},
{
"domain": "docker_container",
"executor": "host_ansible_executor_with_exact_container_playbook",
"verifier": "container_runtime_independent_verifier",
"fallback": "forbidden"
},
{
"domain": "windows_vmware",
"executor": "Agent99",
"verifier": "agent99_independent_runtime_verifier",
"fallback": "forbidden"
},
{
"domain": "database",
"executor": "db_bounded_executor",
"verifier": "db_independent_verifier",
"fallback": "forbidden"
},
{
"domain": "backup_restore",
"executor": "backup_restore_break_glass",
"verifier": "backup_restore_readback_verifier",
"fallback": "forbidden",
"critical_read_only_default": true
},
{
"domain": "unknown",
"executor": null,
"verifier": null,
"terminal": "asset_identity_unresolved",
"fallback": "forbidden"
},
{
"domain": "control_plane_recovery",
"executor": "Agent99",
"verifier": "cold_start_independent_scorecard_verifier",
"fallback": "forbidden",
"exact_target_required": true
}
],
"immediate_execution_queue": [
{
"position": 1,
"work_item_id": "AIA-SRE-017",
"scope": "alert_chain_delivery_truth_and_emergency_card_attribution",
"blocking_reason": "firing AlertChainBroken invalidates downstream automation and verifier trust",
"terminal_condition": "Alertmanager monotonic delivery metrics are scraped; Agent99 on host99 independently polls the exact host110 active/critical AlertChain signal without a first-hop credential or raw-payload persistence and relays only the reduced envelope over HTTPS; the false NodePort-counter alert resolves; emergency cards state AI/Agent/executor/verifier truth"
},
{
"position": 2,
"work_item_id": "AIA-SRE-002",
"scope": "retire_host110_ollama_identity_and_reconcile_portfolio_inventory",
"blocking_reason": "stale provider identity produces false failover alerts and corrupts provider comparison",
"terminal_condition": "Host110 Ollama is a canonical retired tombstone; GCP-A, GCP-B and host111 are direct identities; host111 LaunchAgent is independently verified from host120/121; exact provider sensors are fresh; no host110 Ollama endpoint remains in runtime or monitoring"
},
{
"position": 3,
"work_item_id": "AIA-SRE-004",
"scope": "critical_fail_closed_and_exact_alert_chain_domain_route",
"blocking_reason": "critical or cross-domain candidates must not reach an executor through legacy fallback",
"terminal_condition": "critical override stays no-write and AlertChainBroken resolves only to host110 Alertmanager container lane"
},
{
"position": 4,
"work_item_id": "AIA-SRE-013",
"scope": "five_lane_paid_provider_canary",
"blocking_reason": "paid calls are allowed only after positions 1-3 have independent readback",
"terminal_condition": "all five providers return comparable safe receipts; public-cloud prompts remain sanitized and tool loops fail closed; the Gate5 authorization UUID is the exact execution run/trace, has an atomic one-use claim and canary bucket proof; Claude/Gemini use a run-owned lease, paired rollback and verified aggregate cost caps"
},
{
"position": 5,
"work_item_id": "AIA-SRE-014",
"scope": "independent_verifier_coverage",
"blocking_reason": "no apply may close without a domain-specific verifier",
"terminal_condition": "every mutating catalog and provider lane has an independent terminal readback"
},
{
"position": 6,
"work_item_id": "AIA-SRE-015",
"scope": "incident_and_learning_writeback",
"blocking_reason": "source, CD or Telegram visibility does not prove learning closure",
"terminal_condition": "same-run incident, KM, RAG, MCP and PlayBook acknowledgements are durable"
}
],
"work_items": [
{
"id": "AIA-SRE-001",
"order": 1,
"priority": "P0",
"phase": "phase_1",
"title": "建立唯一 machine-readable 架構與工作總帳",
"owner_lane": "AIControlPlane",
"risk": "low",
"status": "source_implemented_runtime_pending",
"dependencies": [],
"source_refs": [
"docs/operations/sre-k3s-controlled-automation-work-items.snapshot.json",
"apps/api/src/services/sre_k3s_controlled_automation_work_items.py"
],
"executor": "Gitea controlled CD",
"verifier": "ledger schema and API readback verifier",
"rollback": "revert source commit",
"exit_condition": "production API returns the exact committed ledger and rollups",
"confirmed_truth": [
"the local machine-readable ledger records all 18 ordered work items and keeps source, CD, runtime and recipient-visible evidence separate"
],
"runtime_gaps": [
"bounded production readback at 2026-07-19T06:27:25+08:00 still returned the 2026-07-17T02:00:46+08:00 ledger: AIA-SRE-007/012/016/018 remained planned and AIA-SRE-015/017 remained in_progress, so production does not expose this committed source truth"
],
"next_action": "after the separately owned CI/CD blocker is repaired, integrate once and require the production API generated_at, all 18 item statuses and rollups to match the exact committed ledger"
},
{
"id": "AIA-SRE-002",
"order": 2,
"priority": "P0",
"phase": "phase_1",
"title": "Canonical Service Registry 單一來源與 runtime exact mirror",
"owner_lane": "AssetIdentity",
"risk": "medium",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-001"
],
"source_refs": [
"ops/config/service-registry.yaml",
"ops/monitoring/service-registry.yaml",
"ops/monitoring/generated/prometheus-scrape-generated.yaml",
"ops/monitoring/generated/blackbox-targets-generated.yaml",
"docs/operations/portfolio-infrastructure-asset-reconciliation.snapshot.json",
"docs/operations/portfolio-infrastructure-asset-reconciliation-handoff.md",
"apps/api/src/services/portfolio_infrastructure_asset_reconciliation.py",
"apps/api/src/core/config.py",
"apps/api/src/api/v1/agents.py",
"apps/api/tests/test_config_url_validation.py",
"apps/api/tests/test_ollama_endpoint_resolver.py",
"apps/api/tests/test_portfolio_infrastructure_asset_reconciliation_api.py",
"scripts/ops/retire-host110-ollama-proxy.sh",
"scripts/ops/render-service-registry-configmap.py",
"k8s/awoooi-prod/15-service-registry-configmap.yaml"
],
"executor": "service registry renderer plus Gitea CD",
"verifier": "source hash and exact embedded YAML parity",
"rollback": "reapply previous ConfigMap from prior Gitea SHA",
"exit_condition": "production ConfigMap, monitoring inventory and portfolio asset graph match live canonical identities; Host110 Ollama is present only as a retired tombstone; GCP-A/GCP-B exact targets and the host111 LaunchAgent sensor are fresh; inventory snapshots are never treated as runtime health",
"confirmed_truth": [
"Host110 Ollama has been removed and must remain a retired tombstone, never a fallback endpoint",
"AIA-SRE-002 run-aia-sre-002-20260716-1752 removed the exact stale Nginx 11435 proxy with rollback backup and independently verified listeners 11435/11434=0, Ollama containers=0, stale config paths=0, nginx active/test pass and Prometheus host110 Ollama targets=0",
"The Settings asset gate rejects native and IPv4-mapped IPv6 Host110 Ollama URLs as asset_identity_unresolved before any health probe, fallback decision or Telegram alert; focused URL/order/manifest tests pass locally",
"Host111 is the third provider hop and runs as a macOS LaunchAgent, not systemd",
"GCP-A and GCP-B source targets are public HTTP sanitized candidates only",
"gitea-native source target is not production runtime evidence while the legacy GitHub exporter remains visible"
],
"runtime_gaps": [
"The Host110 stale-endpoint asset gate has source and local-test evidence only; production startup/readback has not yet proven the retired endpoint cannot recur",
"Host111 LaunchAgent lacks completed local plus host120/host121 origin readback and a fresh exact sensor series",
"GCP-A/GCP-B exact Prometheus target series are missing in production",
"gitea-native exact target is missing and legacy GitHub exporter runtime drift is unresolved"
],
"next_action": "deploy the Host110 asset gate with the reconciled tombstone/targets, preserve the verified Host110 absence receipt, then verify Host111 LaunchAgent from local and host120/121, GCP-A/GCP-B exact Prometheus series, gitea-native target freshness and legacy exporter absence under one source SHA"
},
{
"id": "AIA-SRE-003",
"order": 3,
"priority": "P0",
"phase": "phase_1",
"title": "Unknown asset fail-closed 與 deterministic drift work item",
"owner_lane": "AssetIdentity",
"risk": "medium",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-002"
],
"source_refs": [
"apps/api/src/services/service_registry.py",
"apps/api/src/services/auto_repair_service.py"
],
"executor": "none for unresolved identity",
"verifier": "unknown asset replay must return UNRESOLVED and zero candidates",
"rollback": "not applicable; no-write terminal",
"exit_condition": "unknown assets create a stable drift ID and never return AUTO/restart fallback",
"next_action": "production replay one unknown asset without runtime apply"
},
{
"id": "AIA-SRE-004",
"order": 4,
"priority": "P0",
"phase": "phase_1",
"title": "Typed Domain Router 與跨 domain fallback 禁止",
"owner_lane": "AutomationRouter",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-003"
],
"source_refs": [
"apps/api/src/services/controlled_alert_target_router.py",
"apps/api/src/services/agent99_sre_bridge.py",
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/services/awooop_ansible_audit_service.py",
"apps/api/src/services/awooop_ansible_check_mode_service.py",
"apps/api/src/api/v1/agents.py",
"apps/api/tests/test_agent99_sre_bridge.py",
"apps/api/tests/test_agent99_controlled_dispatch_p1.py",
"apps/api/tests/test_sre_typed_domain_router.py",
"ops/monitoring/alerts-unified.yml"
],
"executor": "typed policy router",
"verifier": "domain/catalog/host/canonical ID scope verifier",
"rollback": "revert router source while keeping unknown fail-closed",
"exit_condition": "every candidate is exact-domain scoped; circuit-open never selects another host/domain; GCP-A/GCP-B unsanitized prompts and every cloud tool loop fail closed before a network call",
"confirmed_truth": [
"Agent99 admission rejects unresolved and cross-domain routes for both read-only and controlled-apply events before single-flight claim or transport",
"Explicit circuit_state=open or circuit_open=true is terminal before Agent99 claim/transport and requires a same-domain repair work item instead of fallback",
"BackupCheck persists executor=Agent99 for the exact read-only collector receipt and separately records break_glass_executor=backup_restore_break_glass; controlled apply remains false and the real ledger verifier accepts only the identity-bound BackupCheck evidence set"
],
"runtime_gaps": [
"source tests do not prove production router enforcement",
"Agent99 exact-host Alertmanager pull and reduced HTTPS relay are source-implemented but have no host99 production receipt or primary-path outage verifier"
],
"next_action": "deploy the typed admission gate, then replay one unknown read-only event, one explicit circuit-open event and one BackupCheck event; prove zero Agent99 transport for the first two and an identity-bound read-only receipt for BackupCheck before broader production policy verification"
},
{
"id": "AIA-SRE-005",
"order": 5,
"priority": "P0",
"phase": "phase_1",
"title": "Sentry profiling consumer exact-container bounded recovery",
"owner_lane": "HostContainerAutomation",
"risk": "medium",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"infra/ansible/playbooks/110-sentry-profiling-consumer-recovery.yml",
"apps/api/src/services/awooop_ansible_post_verifier.py"
],
"executor": "awoooi-ansible-executor-broker",
"verifier": "exact container running plus healthy readback",
"rollback": "stop further writes, same-domain check replay and cooldown",
"exit_condition": "INC-20260714-471307 replay selects only host_110 exact playbook and post-verifier passes",
"next_action": "deploy then execute one bounded same-fingerprint controlled replay"
},
{
"id": "AIA-SRE-006",
"order": 6,
"priority": "P0",
"phase": "phase_2",
"title": "Agent99 Host Operations Bridge typed enforcement",
"owner_lane": "Agent99",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"apps/api/src/services/agent99_sre_bridge.py",
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/services/agent99_completion_callback.py",
"apps/api/src/api/v1/agents.py"
],
"executor": "Agent99 for Windows/VMware/control-plane; relay-only elsewhere",
"verifier": "Agent99 callback plus external same-run verifier",
"rollback": "disable typed dispatch route and retain no-write Status readback",
"exit_condition": "all Agent99 dispatches carry canonical typed scope and no legacy catch-all can mutate",
"next_action": "deploy then replay one allowlisted Host112 or cold-start run plus one cross-domain denied candidate"
},
{
"id": "AIA-SRE-007",
"order": 7,
"priority": "P0",
"phase": "phase_2",
"title": "K3s workload 專屬 executor 與 rollout verifier",
"owner_lane": "KubernetesAutomation",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"apps/api/src/services/controlled_alert_target_router.py",
"apps/api/src/services/kubernetes_controlled_executor.py",
"apps/api/src/services/kubernetes_rollout_verifier.py",
"apps/api/src/services/approval_execution.py",
"apps/api/src/services/callback_dispatcher.py",
"apps/api/src/plugins/mcp/providers/k8s_provider.py"
],
"executor": "kubernetes_controlled_executor",
"verifier": "kubernetes_rollout_verifier",
"rollback": "declarative rollout undo or previous immutable revision",
"exit_condition": "K3s incidents never enter Ansible/Agent99 fallback and close with rollout evidence",
"confirmed_truth": [
"controlled mutation accepts only an identity-verified typed kubernetes_workload route bound to kubernetes_controlled_executor and kubernetes_rollout_verifier",
"the bounded mutation catalog permits rollout restart only; scale, delete, raw shell and model-supplied capabilities cannot enter controlled execution",
"namespace, workload kind, workload name, canonical asset, action digest and incident identity are bound into one single-use claim",
"dry-run precedes the bounded apply and the independent rollout verifier compares pre-state, post-state, generation, replicas and observed revision before closure",
"wrong domain, unresolved identity, verifier drift or cross-domain fallback returns a zero-write fail-closed receipt"
],
"runtime_gaps": [
"no production K3s same-run rollout restart, independent rollout verifier, Telegram receipt and learning closure is recorded for this exact source revision"
],
"next_action": "deploy one exact source revision through the authorized release lane, execute one allowlisted K3s rollout canary with its single-use capability, then independently verify rollout state plus same-run Telegram/KM/RAG/MCP/PlayBook receipts; retain zero-write or rollback terminal on any mismatch"
},
{
"id": "AIA-SRE-008",
"order": 8,
"priority": "P0",
"phase": "phase_2",
"title": "Host/systemd 專屬 Ansible executor inventory",
"owner_lane": "HostAutomation",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"infra/ansible/inventory/hosts.yml",
"infra/ansible/playbooks/111-ollama-fallback.yml",
"apps/api/src/services/awooop_ansible_audit_service.py",
"apps/api/src/services/awooop_ansible_check_mode_service.py",
"apps/api/src/services/host_ansible_controlled_executor.py",
"apps/api/src/services/awooop_ansible_post_verifier.py",
"apps/api/src/services/awooop_ansible_learning_writeback.py"
],
"executor": "host_ansible_executor",
"verifier": "host_runtime_independent_verifier",
"rollback": "catalog-specific compensating action",
"exit_condition": "every host action has one exact inventory host, runtime manager identity, check mode, bounded apply and postcondition; host111 uses its LaunchAgent catalog and is verified locally plus from host120/host121",
"confirmed_truth": [
"all 13 auto-apply and all 16 check-mode catalogs are registered with independent postconditions; the only no-check catalog is explicit break-glass restore-password-auth",
"Host111 Ollama uses one exact LaunchAgent catalog with host111 plus host120/host121 verifier origins and no Host110 fallback",
"check-mode and bounded apply are separate receipts, and executor return code alone cannot satisfy the post-verifier",
"inventory transport uses fixed strict host identities and does not permit accept-new host-key behavior"
],
"runtime_gaps": [
"host111 LaunchAgent catalog and sensor are source candidates without production apply receipt",
"host120 and host121 independent origin probes have no same-run production verifier receipt"
],
"next_action": "deploy one exact source revision through the authorized Agent99/Ansible release lane, then check/apply the Host111 LaunchAgent catalog and independently read back host111 plus host120/host121 and the canonical sensor; retain no-write or rollback terminal on any mismatch"
},
{
"id": "AIA-SRE-009",
"order": 9,
"priority": "P0",
"phase": "phase_2",
"title": "Windows/VMware Agent99 single-executor closure",
"owner_lane": "Agent99",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-006"
],
"source_refs": [
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/services/agent99_same_run_reconcile.py",
"apps/api/src/services/agent99_sre_bridge.py",
"apps/api/src/services/agent99_completion_callback.py",
"apps/api/src/services/agent99_outcome_ingestion.py",
"apps/api/src/services/agent99_public_receipts.py",
"apps/api/src/jobs/agent99_controlled_dispatch_reconciler_job.py"
],
"executor": "Agent99",
"verifier": "agent99_independent_runtime_verifier",
"rollback": "allowlisted Status/no-write reconciliation and bounded generation retry",
"exit_condition": "Windows/VMware runs use durable identity, callback, verifier and learning receipt",
"confirmed_truth": [
"Windows/VMware and control-plane recovery dispatches require a typed same-run identity and Agent99 executor binding",
"unknown assets, arbitrary commands and cross-domain fallback fail closed before runtime authorization",
"authenticated completion callback persists outcome and independent verifier evidence before learning writeback",
"terminal closure requires checkpointed Telegram, KM, RAG, MCP and PlayBook acknowledgements on the same run"
],
"runtime_gaps": [
"no production Windows/VMware same-run dispatch, authenticated callback, independent verifier and complete learning closure receipt is recorded for this exact source revision"
],
"next_action": "deploy one exact source revision through the authorized Agent99 release lane, replay one allowlisted Windows/VMware run plus one cross-domain denial, then read back the same-run callback, verifier, Telegram and learning receipts; do not use arbitrary commands or fallback hosts"
},
{
"id": "AIA-SRE-010",
"order": 10,
"priority": "P0",
"phase": "phase_3",
"title": "DB 專屬 verifier 與 bounded executor",
"owner_lane": "DatabaseAutomation",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"ops/config/service-registry.yaml",
"apps/api/src/db/base.py",
"apps/api/src/core/unit_of_work.py",
"apps/api/src/services/db_bounded_executor.py",
"apps/api/src/services/executor_trust_boundary_readback.py",
"agent99-db-bounded-executor.ps1",
"scripts/ops/awoooi-workload-db-identity-bootstrap.sh"
],
"executor": "db_bounded_executor",
"verifier": "db_independent_verifier",
"rollback": "transactional or compensating action only; destructive restore remains break-glass",
"exit_condition": "DB incidents use DB-native preflight/postconditions and never generic host restart",
"runtime_entrypoint_contract": {
"status": "source_ready_runtime_not_applied",
"command_id": "telegram_receipt_index_apply_v1",
"canonical_asset": "public.awooop_outbound_message",
"execution_hub": "host:192.168.0.99",
"dispatch_role": "Agent99_dispatch_only",
"executor_domain": "database",
"runtime_transport": "host99_to_host110_to_host120_fixed_sudo_kubectl_exec",
"host120_installer_required": false,
"check_apply_pod_role": "first_ready_awoooi_api_pod",
"independent_verifier_role": "second_ready_awoooi_api_pod_second_process",
"arbitrary_sql_allowed": false,
"cross_domain_fallback_allowed": false
},
"next_action": "deploy the fixed Agent99 dispatch through the atomic runtime bundle, then use the already allowlisted host120 sudo kubectl path for one same-trace check/apply/second-pod verify of telegram_receipt_index_apply_v1; separately retain cap-2 global session budget and role headroom >= 8 verification"
},
{
"id": "AIA-SRE-011",
"order": 11,
"priority": "P0",
"phase": "phase_3",
"title": "Backup/restore readback 與 critical break-glass",
"owner_lane": "DataProtection",
"risk": "critical",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"apps/api/src/services/backup_restore_signal_automation.py",
"apps/api/src/services/agent99_sre_bridge.py",
"apps/api/src/jobs/agent99_controlled_dispatch_reconciler_job.py",
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/services/agent99_telegram_lifecycle.py",
"apps/api/src/services/agent99_public_receipts.py",
"apps/api/src/services/ai_agent_log_controlled_writeback_consumer_readback.py",
"apps/api/src/services/telegram_alert_learning_context_post_apply_verifier.py",
"apps/api/src/services/telegram_alert_monitoring_coverage_readback.py",
"apps/api/src/repositories/knowledge_repository.py"
],
"executor": "backup_restore_break_glass",
"verifier": "backup_restore_readback_verifier",
"rollback": "no-write default; destructive action requires incident-specific break-glass",
"exit_condition": "freshness, offsite, escrow and restore-drill evidence close without false green",
"confirmed_truth": [
"backup alerts are typed before generic routing and dispatch only Agent99 BackupCheck with controlledApply=false",
"BackupCheck renders a read-only Telegram lifecycle receipt and never uses a controlled-apply receipt",
"same-run closure requires durable Telegram, KM, RAG, MCP, PlayBook and DR scorecard acknowledgements before terminal writeback",
"existing BackupCheck learning assets reconcile idempotently to backup_restore and no-write semantics"
],
"runtime_gaps": [
"no production same-run BackupCheck dispatch, freshness/offsite/escrow/restore-drill verifier receipt or recipient-visible closure is recorded for this source revision"
],
"next_action": "deploy one exact source revision through the authorized release lane, replay one read-only backup signal through Agent99 BackupCheck and verify same-run Telegram/KM/RAG/MCP/PlayBook/DR receipts; do not run backup, restore, delete or retention changes"
},
{
"id": "AIA-SRE-012",
"order": 12,
"priority": "P0",
"phase": "phase_3",
"title": "HolmesGPT Investigator shadow integration",
"owner_lane": "Investigator",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-004"
],
"source_refs": [
"apps/api/src/services/pre_decision_investigator.py",
"apps/api/src/services/holmes_shadow_investigator.py",
"apps/api/src/services/holmes_shadow_replay.py",
"apps/api/src/services/evidence_snapshot.py",
"apps/api/src/core/config.py",
"apps/api/tests/test_holmes_shadow_investigator.py",
"apps/api/tests/test_holmes_shadow_replay.py"
],
"executor": "none in shadow",
"verifier": "offline replay quality and prompt-injection safety scorecard",
"rollback": "remove shadow consumer; active PreDecisionInvestigator remains",
"exit_condition": "internal immutable artifact passes replay, shadow and canary before core replacement",
"confirmed_truth": [
"the complete HolmesGPT shadow investigator and replay source contract passes 17 focused tests with Ruff clean; this is source evidence only and does not prove an approved artifact or runtime promotion",
"the optional PreDecisionInvestigator shadow adapter uses HolmesGPT /api/chat with a strict JSON schema and bounded timeout",
"evidence and returned text are sanitized, protected API key and endpoint are absent from receipts, and only an exact allowlisted internal origin whose runtime artifact header matches the pinned digest can be contacted",
"every trusted post-generation response must include prompt-token, completion-token and cost headers; missing or invalid usage metadata fails closed and the replay scorecard aggregates the bounded totals",
"tool calls, follow-up actions, invalid structured output and any runtime-authority claim are rejected fail closed",
"the advisory receipt merges without replacing existing diagnosis signals and is summarized into the persisted evidence text",
"the sanitized receipt is reused only through the existing 30-second evidence fingerprint cache and is marked as reused, preventing duplicate model calls for an identical cached evidence slice",
"the offline replay evaluator reuses AWOOOI candidate-visible inputs, keeps evaluation labels outside the model request, caps one batch at 50 sequential calls and persists only bounded metrics plus result digests",
"the replay scorecard reports contract, RCA-label, prompt-injection, action-surface and runtime-authority observations but cannot authorize canary promotion or runtime writes"
],
"runtime_gaps": [
"no approved internal immutable HolmesGPT artifact, allowlisted endpoint, modelList alias, runtime usage-header receipt, executed 50-record offline replay scorecard, shadow canary or production receipt exists in this worktree"
],
"next_action": "supply HolmesGPT from an approved internal immutable mirror, configure its protected key/modelList alias/exact allowlist/digest, then execute the bounded 50-record replay against sanitized historical fixtures before any canary"
},
{
"id": "AIA-SRE-013",
"order": 13,
"priority": "P0",
"phase": "phase_3",
"title": "GCP-A/GCP-B/host111 Ollama、Claude、Gemini RCA/critic 與付費 canary/cost gate",
"owner_lane": "ModelRouter",
"risk": "critical",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-012"
],
"source_refs": [
"apps/api/src/services/ai_router.py",
"apps/api/src/services/ollama_failover_manager.py",
"apps/api/src/services/ai_providers/claude.py",
"apps/api/src/services/ai_providers/gemini.py",
"apps/api/src/services/ai_provider_policy.py",
"apps/api/src/services/failover_alerter.py",
"apps/api/src/services/cloud_transport_receipt.py",
"apps/api/src/services/paid_provider_canary_authorization.py",
"apps/api/src/services/paid_provider_canary_gate5_run.py",
"apps/api/src/services/paid_provider_canary_validation.py",
"apps/api/src/jobs/paid_provider_canary_worker.py",
"apps/api/src/services/ai_rate_limiter.py",
"scripts/ops/run-paid-provider-canary.py",
"k8s/awoooi-prod/04-configmap.yaml",
"k8s/awoooi-prod/06-deployment-api.yaml"
],
"executor": "singleton signal-worker paid canary executor plus ordered model router",
"verifier": "provider-attributed replay accuracy, latency and cost readback",
"rollback": "Claude/Gemini are disabled together on any partial five-lane result; remain on the ordered Ollama chain",
"exit_condition": "provider order is enforced; GCP public-HTTP hops remain sanitized candidate-only with cloud tool loops fail-closed until secure transport promotion; paid providers remain 5 percent canary only; the durable Gate5 UUID is the same execution run/trace and is atomically single-use; protected-secret metadata, paired rollback, aggregate cost cap, finalized receipts, latency/token/cost and production readback are verified without secret-value exposure",
"confirmed_truth": [
"The user supplied Claude and Gemini credential material, but raw values are not repository data and must not be copied or read back",
"GCP-A/GCP-B are currently public HTTP boundaries and cannot receive unsanitized or tool-loop payloads",
"Claude and Gemini enablement, canary outcome and rollback form one atomic pair",
"Gate5 run 019f6c09-e77c-7800-35af-9179358c8f37 was durably approved and consumed exactly once; Claude generation authenticated, Gemini authentication was rejected, and both paid gates returned to disabled",
"The first canary accounted USD 0.012099 under the USD 0.25 run cap and persisted neither raw prompt nor raw response",
"The first Gate5 approval emitted approval_hmac_key_not_set_using_dev_fallback; this is a production approval-signing contract drift, not evidence that paid providers are hard locked"
],
"runtime_gaps": [
"GCP-A transport is unreachable and has no same-run transport receipt",
"GCP-B is reachable and has qwen3:14b, but the first canary used an obsolete 120 second outer timeout instead of the 300 second diagnose budget",
"host111 accepts TCP but resets the Ollama HTTP request and requires Agent99 bounded host service diagnosis",
"Claude authenticated but its first response did not satisfy the deterministic OpenClawDecision contract; strict tool schema source fix awaits CD and replay",
"Gemini production secret reference is present but provider authentication is rejected; the user-supplied replacement is not yet present in the protected runtime secret path",
"APPROVAL_HMAC_KEY is not bound to a dedicated protected production secret; the development fallback must be retired after a secret-metadata-only preflight and before production provider promotion",
"singleton approved-run executor handoff and container import-path fixes await Gitea CD",
"provider scorecard KM/RAG/MCP/PlayBook acknowledgement is pending"
],
"next_action": "deploy the executor, strict Claude schema and Ollama timeout fixes; bind APPROVAL_HMAC_KEY through a dedicated protected secret reference without reading its value; repair GCP-A transport and host111 through their typed owner lanes; rotate Gemini through protected secret ingress without repository or log exposure; then create a new single-use Gate5 run and rerun the sanitized five-lane verifier"
},
{
"id": "AIA-SRE-014",
"order": 14,
"priority": "P0",
"phase": "phase_4",
"title": "Independent verifier registry 完整覆蓋",
"owner_lane": "PostExecutionVerifier",
"risk": "high",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-005",
"AIA-SRE-007",
"AIA-SRE-008",
"AIA-SRE-009",
"AIA-SRE-010"
],
"source_refs": [
"apps/api/src/services/independent_verifier_registry.py",
"apps/api/src/services/awooop_ansible_post_verifier.py",
"apps/api/src/services/kubernetes_rollout_verifier.py",
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/services/db_bounded_executor.py",
"apps/api/src/services/backup_restore_signal_automation.py",
"apps/api/src/services/controlled_alert_target_router.py",
"apps/api/src/services/post_execution_verifier.py"
],
"executor": "none",
"verifier": "domain-specific external readback",
"rollback": "failed verifier triggers same-domain retry or compensating action",
"exit_condition": "100 percent of mutating catalogs have independent production postconditions; host111 includes local plus host120/host121 origin verification and provider monitoring requires exact fresh series",
"confirmed_truth": [
"machine-readable registry covers all eight typed domains and keeps executor identity separate from verifier identity for all six mutating domains",
"all 17 Ansible catalogs have registered independent postconditions; every auto-apply catalog is check-mode capable and the sole exception is explicit break-glass",
"K3s, Host/Container, Agent99, DB, Backup/restore and unknown-asset lanes expose domain-specific verifier contracts with cross-domain fallback disabled",
"Backup/restore read-only routing and the verifier registry now bind the same Agent99 evidence collector, backup_restore_readback_verifier and separate backup_restore_break_glass mutation executor; controlled apply and runtime writes remain disabled, with a cross-module contract test",
"Host111 three-origin, GCP exact-series, gitea-native metrics and host99 Alertmanager relay readbacks are registered as runtime receipt requirements rather than source completion evidence"
],
"runtime_gaps": [
"source-level verifier definitions are not production verifier receipts",
"Host111 LaunchAgent, GCP provider targets, gitea-native metrics and the host99 Alertmanager pull/relay remain without independent runtime closure"
],
"next_action": "deploy one exact source revision through the authorized release lane, then attach same-run production receipts for K3s, Ansible, Agent99, DB and Backup readback plus Host111 origins, GCP exact targets, gitea-native metrics and host99 Alertmanager pull/relay; any missing receipt remains fail-closed"
},
{
"id": "AIA-SRE-015",
"order": 15,
"priority": "P0",
"phase": "phase_4",
"title": "Incident closure 與 KM/RAG/MCP/PlayBook durable writeback",
"owner_lane": "LearningControlPlane",
"risk": "medium",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-014"
],
"source_refs": [
"apps/api/src/services/learning_service.py",
"apps/api/src/services/awooop_ansible_learning_writeback.py",
"apps/api/src/jobs/agent99_controlled_dispatch_reconciler_job.py",
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/services/agent99_telegram_lifecycle.py",
"apps/api/src/services/agent99_public_receipts.py",
"apps/api/src/repositories/knowledge_repository.py",
"apps/api/src/services/telegram_alert_learning_context_post_apply_verifier.py",
"apps/api/src/services/telegram_alert_monitoring_coverage_readback.py",
"apps/web/src/app/[locale]/awooop/alerts/page.tsx"
],
"executor": "learning writeback consumer",
"verifier": "durable same-run acknowledgement readback",
"rollback": "append-only correction event; immutable receipts are not overwritten",
"exit_condition": "closure requires all learning acknowledgements on the same run",
"confirmed_truth": [
"source and focused tests bind typed verifier outcomes to same-run Telegram, KM, RAG, MCP, PlayBook and backup DR scorecard acknowledgements",
"Backup/restore recipient-visible recovered cards name Agent99 BackupCheck as the read-only executor and backup_restore_readback_verifier as the typed independent verifier; apply remains not_applicable and the final durable closure receipt is required",
"BackupCheck learning assets are reconciled idempotently as read-only DR evidence rather than controlled repair",
"the Telegram learning verifier now distinguishes alert-card registry receipts from log-controlled consumer receipts, verifies the latter only with an exact successful consumer write receipt, and keeps verifier runtime writes false",
"the monitoring coverage readback no longer lets consumer candidates or alert-card metadata refs bypass the independent verifier; learning readiness requires every declared target write plus an AI Agent context receipt to be verified",
"the independent verifier now rejects metadata-only 0/6 receipts, shrunk 5/5 contracts, duplicated consumer write receipt IDs and selector/receipt ID mismatches; ready requires exactly six unique durable consumer-write receipts bound to the six canonical targets",
"the AwoooP alerts surface distinguishes verified, candidate and blocked learning closure and displays durable write, verified target, AI Agent receipt, evidence-source and blocker readbacks without claiming source evidence as production closure"
],
"runtime_gaps": [
"production exposed six fallback context receipts but the deployed verifier rejected all six under the alert-card-only contract, while the deployed coverage could still treat unverified fallback metadata as ready; this source revision fixes both false-negative and false-green paths but still has no production same-run receipt",
"release c0afb10861cbb76bc232fad3bbeea0d675cd9b80 reached Gitea CD 5384 terminal failure without a deploy marker, so production remained on e614f061f781c6e20a3964875d7ab3c28cf91190",
"bounded production readback at 2026-07-19T06:27:25+08:00 received zero bytes from /api/v1/agents/telegram-alert-monitoring-coverage-readback before the 12-second client deadline, so no Telegram/KM/RAG/MCP/PlayBook runtime closure can be claimed"
],
"next_action": "after the separately owned CI/CD blocker is repaired, integrate this exact tested receipt-contract revision once and read back one same-run Telegram/KM/RAG/MCP/PlayBook/DR closure receipt; source evidence must not be counted as runtime closure"
},
{
"id": "AIA-SRE-016",
"order": 16,
"priority": "P1",
"phase": "phase_5",
"title": "24h replay、MTTA/MTTR、誤判、復發與成本 scorecard",
"owner_lane": "AutomationSLO",
"risk": "low",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-015"
],
"source_refs": [
"apps/api/src/services/automation_slo_scorecard.py",
"apps/api/src/services/ai_automation_runtime_contract.py",
"apps/api/src/api/v1/agents.py"
],
"executor": "offline replay worker",
"verifier": "production aggregate versus same-run receipts",
"rollback": "read-only scorecard",
"exit_condition": "all lanes publish MTTA, MTTR, false positive, recurrence, human intervention, verifier and rollback metrics",
"confirmed_truth": [
"read-only scorecard executes three project-scoped aggregate queries for incident lifecycle, AwoooP runs and provider budget receipts",
"global and per-lane projections publish MTTA, MTTR, false-positive, recurrence, human-intervention, verifier-pass, rollback, freshness and asset-coverage metrics",
"run projection preserves raw total/completed/failed counts while separately publishing non-shadow controlled-with-steps completed and failed counts, so shadow or no-step runs cannot inflate true automation outcomes; provider projection preserves call, token and cost totals",
"missing samples, classifications, verifier evidence or freshness denominators remain partial/no-runtime-sample and cannot report runtime closure"
],
"runtime_gaps": [
"the exact source revision is not deployed and no production 24h aggregate has been compared with same-run receipts or recipient-visible scorecard readback"
],
"next_action": "deploy one exact source revision through the authorized release lane, read /api/v1/agents/sre-automation-slo-scorecard-24h after a complete 24h window, compare aggregate counts with same-run receipts and keep any missing denominator partial"
},
{
"id": "AIA-SRE-017",
"order": 17,
"priority": "P1",
"phase": "phase_5",
"title": "Telegram 與 cockpit 只呈現 canonical lifecycle truth",
"owner_lane": "ProductExperience",
"risk": "medium",
"status": "source_implemented_runtime_pending",
"dependencies": [
"AIA-SRE-015"
],
"source_refs": [
"ops/config/telegram-routing-registry.yaml",
"ops/alertmanager/alertmanager.yml",
"k8s/monitoring/prometheus.yml",
"ops/monitoring/alerts-unified.yml",
"scripts/ops/deploy-alertmanager-runtime-scrape.sh",
"agent99-alertmanager-alertchain-poll.ps1",
"agent99-sre-alert-relay.ps1",
"scripts/reboot-recovery/deploy-agent99-via-windows99-ssh.sh",
"apps/api/src/services/failover_alerter.py",
"apps/api/src/services/telegram_gateway.py",
"apps/api/src/services/notification_matrix.py",
"apps/api/src/services/agent99_sre_bridge.py",
"apps/api/src/services/agent99_telegram_lifecycle.py",
"apps/api/src/services/agent99_controlled_dispatch_ledger.py",
"apps/api/src/jobs/agent99_controlled_dispatch_reconciler_job.py",
"apps/api/src/services/agent99_public_receipts.py",
"apps/api/src/services/telegram_button_registry.py",
"apps/api/src/services/callback_dispatcher.py"
],
"executor": "canonical Telegram gateway",
"verifier": "delivery receipt plus desktop/mobile visible smoke",
"rollback": "suppress duplicate projection; retain incident ledger",
"exit_condition": "no alert is sprayed across groups/bots; every card states AI/provider/Agent action, executor, verifier and receipt truth; AlertChainBroken uses Alertmanager integration-scoped delivery counters stamped with the independently tested sole-webhook receiver contract plus host99 Agent99 exact-host read-only polling independent of the broken webhook; UI status polling never emits failover alerts",
"confirmed_truth": [
"source registry covers 11 products and 26 typed routes; only allowlisted P0/P1 operational routes resolve to the AwoooI SRE war room",
"unknown product, signal, severity or requested destination fails closed and active source contains no raw numeric Telegram destination default",
"Telegram automation cards distinguish durable AI/provider, Agent, executor, verifier and learning receipts from configured-but-unobserved components",
"Agent99 same-run closure is two-phase: an ApprovalRecord fingerprint/hit-count/last-seen recurrence fence must still match, the incident resolution commits first, the recovered Telegram card must receive a durable destination-bound provider acknowledgement, and only then may the run claim runtime_closure_verified",
"information and controlled-action buttons are bound to registered callback actions instead of display-only ghost buttons",
"Telegram monitoring coverage source now limits its complete readback pipeline to a 9.5-second fail-closed budget, serializes four database stages to one connection at a time, disables cache writes for the AI-card coverage read, classifies bounded DB retries as low-risk read-only, and preserves break-glass for runtime canary or lifecycle resume"
],
"runtime_gaps": [
"the host99 Agent99 independent pull/reduced relay has no production deployment, freshness, dedupe or controlled-repair receipt",
"production Alertmanager exposes integration-only notification counters; the bounded check now passes after adding a scrape-time receiver_contract label, but apply and alert resolution remain pending",
"the recipient-visible Agent99 recovered-card closure contract is source-tested only; no production Telegram provider acknowledgement or same-run runtime closure receipt exists yet",
"production runtime e614f061f7 still reports both alert-operation-log and AI-card-delivery DB readbacks unavailable with zero 7-day automation lifecycle receipts; the bounded serial readback fix is source-only until a successful deploy marker",
"bounded /api/v1/telegram/health readback at 2026-07-19T06:27:25+08:00 was configured but degraded_callback_ingress_unverified: interactive_buttons_ready=false, no durable/fresh ingress receipt and no controlled-action receipt, so all registered callback actions correctly remain fail-closed rather than being counted as working buttons"
],
"next_action": "deploy one exact source revision through the authorized release lane, then verify receiver-contract freshness, host99 Agent99 exact-host polling, dedupe, callback execution, current firing alert resolution and recipient-visible same-run delivery receipts; source tests must not be counted as runtime closure"
},
{
"id": "AIA-SRE-018",
"order": 18,
"priority": "P0",
"phase": "phase_6",
"title": "Gitea CD、production runtime 與 visible closure",
"owner_lane": "ReleaseControl",
"risk": "high",
"status": "in_progress",
"dependencies": [
"AIA-SRE-001",
"AIA-SRE-002",
"AIA-SRE-003",
"AIA-SRE-004",
"AIA-SRE-005",
"AIA-SRE-006",
"AIA-SRE-014",
"AIA-SRE-015"
],
"source_refs": [
".gitea/workflows/cd.yaml",
"docs/operations/portfolio-infrastructure-asset-reconciliation.snapshot.json"
],
"executor": "Gitea controlled CD",
"verifier": "deploy marker, image SHA, API readback and desktop/mobile smoke",
"rollback": "previous immutable image revision and ConfigMap source SHA",
"exit_condition": "source/test/CD/runtime/UI evidence all match one deployed SHA; gitea-native exact metrics are fresh and the retired legacy GitHub exporter is absent from production",
"confirmed_truth": [
"the c0afb10861 release candidate was merged with its then-latest observed gitea/main and passed the CD-equivalent non-integration API gate with 5666 passed and 23 skipped, including the 977-test changed-file suite; 4 focused web tests, web typecheck and JSON/YAML parsing also passed",
"stale Telegram tests that required unobserved OpenClaw/NemoTron/ElephantAlpha labels, a promotional check/apply/verify/writeback string or an unreceipted 88 percent confidence were replaced by no-false-AI receipt and actual stage-state contracts; receipt-backed provider, Agent99 executor, verifier and KM/MCP evidence remain visible",
"the Solver, RecommendedAction schema and callback registry now agree that host/systemd actions use the controlled provider and host_ansible_reconcile typed route instead of generic SSH fallback",
"exact source SHA c0afb10861cbb76bc232fad3bbeea0d675cd9b80 was fast-forward pushed once to gitea/main and created the single CD run 5384; no empty commit, retry carrier or force push was used",
"CD run 5384 reached terminal Failure at 2026-07-19T05:05:37+08:00; gitea/main remained c0afb10861cbb76bc232fad3bbeea0d675cd9b80 with no deploy-marker successor",
"production delivery-closure readback generated at 2026-07-19T05:07:28+08:00 still reported runtime build e614f061f781c6e20a3964875d7ab3c28cf91190, so the candidate is not production-visible",
"read-only fetch at 2026-07-19T06:30:36+08:00 confirmed tested functional Telegram vertical-slice head 93f9b4af44 was exactly 0 commits behind and 7 commits ahead of gitea/main c0afb10861; its focused verifier/coverage/work-order regressions pass 71 tests, ledger tests pass 9 tests, and the prior AwoooP UI typecheck/JSON/format/desktop/mobile checks remain source evidence only"
],
"runtime_gaps": [
"no program item is runtime closed for this source SHA",
"gitea-native exact target is missing while legacy GitHub exporter runtime drift remains",
"the exact candidate has a terminal failed CD but no deploy marker, production runtime match, visible UI/Telegram receipt or same-run KM/RAG/MCP/PlayBook acknowledgement",
"authenticated CD logs are unavailable in this lane, so the failed step and sanitized final error remain missing; another carrier would be an unproven retry",
"the Telegram vertical-slice batch is integration-ready against current gitea/main but intentionally remains local until the separately owned CD 5384 failure is reproduced or proven transient"
],
"next_action": "the CI/CD owner must capture CD 5384 failed-step evidence and either reproduce a deterministic failure locally or prove a transient runner fault; only then normal-push the current 0-behind Telegram batch once, and after a successful deploy marker require production API/runtime/UI/Telegram readback plus same-run KM/RAG/MCP/PlayBook acknowledgements"
}
],
"rollups": {
"total_items": 18,
"by_priority": {
"P0": 16,
"P1": 2
},
"by_status": {
"source_implemented_runtime_pending": 17,
"in_progress": 1
},
"source_implemented_items": 17,
"runtime_closed_items": 0,
"program_completion_percent": 0,
"asset_coverage_status": "partial",
"runtime_closure_status": "blocked_cd_5384_failure_runtime_build_e614f061f7"
},
"completion_contract": {
"required_stages": [
"sensor_source_receipt",
"canonical_asset_identity",
"source_truth_diff",
"ai_decision_candidate",
"risk_policy_decision",
"check_mode_receipt",
"bounded_execution_receipt",
"independent_verifier_or_no_write_terminal",
"incident_closure",
"telegram_receipt",
"km_rag_mcp_playbook_writeback_ack"
],
"terminal_without_all_stages": "partial_or_degraded_with_safe_next_action",
"source_test_cd_green_is_runtime_closure": false,
"evidence_layers": {
"source_test": "partial_source_evidence_only",
"gitea_cd_deploy_marker": "pending_for_this_program_sha",
"production_runtime": "zero_work_items_closed",
"visible_ui_telegram": "pending_same_run_receipts",
"learning_writeback": "pending_same_run_durable_acknowledgements"
},
"promotion_rule": "a lower evidence layer cannot promote an incomplete higher evidence layer"
}
}